@article{4776, author = {K.Kiruthika}, title = {Demographic Inequality and Linguistic Diversity in Indian States and Union Territories}, journal = {International Journal of Computational Linguistics Research}, year = {2026}, volume = {17}, number = {3}, doi = {https://doi.org/10.6025/ijclr/2026/17/3/133-150}, url = {https://www.dline.info/jcl/fulltext/v17n3/jclv17n3_1.pdf}, abstract = {India's multilingual landscape and uneven demographic distribution present unique challenges for governance, yet few studies integrate these dimensions at the sub-national level. This study proposes an integrated analytical framework to examine demographic inequality and linguistic diversity across India's 36 States and Union Territories. The acquired dataset encompasses administrative identifiers, socioeconomic metrics, and widely spoken languages. Demographic concentration is evaluated using the Gini coefficient, Lorenz curve, cumulative distribution, and Pareto analysis. Concurrently, linguistic heterogeneity is quantified through lexical diversity indices, including the Shannon and Simpson Diversity Indices, Type Token Ratio, Herdan's C, and Pielou's Evenness. Additionally, network-based visualizations, such as state language bipartite networks and co-occurrence maps, elucidate structural patterns of multilingual coexistence. Results reveal pronounced demographic inequality (Gini = 0.6149), confirming that a small subset of highly populated states accounts for the vast majority of the national population. In stark contrast, linguistic analysis identifies 43 distinct languages exhibiting high lexical richness and equitable distribution, demonstrating minimal dominance by any single language. These findings highlight a critical divergence between concentrated demographics and broadly distributed linguistic ecosystems. Ultimately, this reproducible computational framework provides policymakers with vital insights for evidence-based governance, equitable resource allocation, educational planning, and language preservation, demonstrating how computational analytics can support the study of complex socio-demographic systems.}, }