@article{4778, author = {Pit Pichappan}, title = {Vocabulary Distribution, Lexical Diversity, and Statistical Properties of the Fact Verification Corpus}, journal = {International Journal of Computational Linguistics Research}, year = {2026}, volume = {17}, number = {3}, doi = {https://doi.org/10.6025/ijclr/2026/17/3/172-185}, url = {https://www.dline.info/jcl/fulltext/v17n3/jclv17n3_3.pdf}, abstract = {The proliferation of digital misinformation necessitates robust automated fact verification systems, which rely heavily on high quality benchmark datasets. However, the underlying linguistic and statistical properties of these corpora often remain underexplored. This study presents a comprehensive corpus-linguistic analysis of a fact verification dataset, examining its vocabulary distribution, lexical diversity, and adherence to natural language statistical regularities. Employing a multi metric framework, we evaluated lexical richness using indices such as the Shannon and Simpson Diversity Indices, Type Token Ratio, MTLD, and HDD. Furthermore, we investigated corpus level statistical properties through Zipf's and Heap's laws. Results indicate substantial lexical heterogeneity, with high diversity indices confirming a rich vocabulary inventory suitable for complex semantic reasoning. The rank frequency distribution conforms to Zipf's Law ( = 0.482), revealing a characteristic long tail pattern where domain specific terminology and functional words dominate, alongside numerous infrequent but semantically critical entities. Additionally, Heap's Law analysis ( = 0.637) demonstrates sublinear vocabulary growth, indicating lexical saturation typical of specialized technical corpora. These findings validate the dataset's linguistic coherence and statistical suitability for downstream natural language processing applications, including transformer based language modeling, information retrieval, and retrieval augmented generation. Ultimately, this research establishes a strong empirical foundation for optimizing data preprocessing, model architecture selection, and the development of reliable, evidence aware fact checking systems}, }