@article{4788, author = {Hathairat Ketmaneechairat}, title = {A Statistically Validated Large Scale Mathematical Chain of Thought Dataset for Symbolic Reasoning and Large Language Model Benchmarking}, journal = {Journal of Data Processing}, year = {2026}, volume = {16}, number = {3}, doi = {https://doi.org/10.6025/jdp/2026/16/3/161-184}, url = {https://www.dline.info/jdp/fulltext/v16n3/jdpv16n3_3.pdf}, abstract = {Mathematical reasoning poses a significant challenge for Large Language Models (LLMs) due to the necessity for structured symbolic manipulation, multi step deduction, and logically consistent inference. While Chainof Thought (CoT) prompting has substantially advanced reasoning capabilities, existing datasets often lack transparent, verifiable reasoning trajectories and rigorous statistical validation. This study presents a statistically validated, large scale mathematical CoT dataset comprising approximately 100,000 samples spanning six core mathematical domains (Algebra, Geometry, Trigonometry, Matrices, Derivatives, and Integrals) and three predefined difficulty levels. The dataset preserves complete intermediate reasoning traces and final solutions in standardized LaTeX notation. We conducted comprehensive structural, descriptive, and inferential statistical analyses including Chi-square tests, Spearman's rank correlation, two-way ANOVA, and Tukey's HSD post hoc comparisons to validate the dataset's internal consistency. Results demonstrate a perfectly balanced distribution across topics and difficulties, with the mathematical topic emerging as the dominant determinant of reasoning complexity (partial n² = 0.995). Furthermore, the corpus exhibits rich symbolic diversity, authentic mathematical syntax, and structured command cooccurrence networks. This statistically validated corpus provides a reliable and robust benchmark for supervised fine tuning, curriculum learning, process supervision, explainable AI, and evaluating nextgeneration mathematical LLMs.}, }