@article{1417, author = {Chandan Kundu, Rajib Kumar Das, Kalyan Sengupta}, title = {Implementation of Context Window and Context Identification Array for Identification and Interpretation of Non Standard Word in Bengali News Corpus}, journal = {International Journal of Computational Linguistics Research}, year = {2013}, volume = {4}, number = {4}, doi = {}, url = {http://www.dline.info/jcl/fulltext/v4n4/2.pdf}, abstract = {Non Standard Word (NSW) identification and its interpretation is a major challenge in information retrieval system. Real text not only contains ordinary words and names but also contains non-standard “words” including numbers, abbreviations, dates, months, currency, amounts and different types of numbers. In reality, we can not find NSWs in a dictionary, nor can one find their pronunciation by an application of ordinary “letter-to-sound” rules. Non standard words have greater inclination towards ambiguity in terms of their interpretation and pronunciation than ordinary words. So it is very much required to identify and interpret the NSW properly while we are going for further analysis of text. In real text, NSW are represented in diversified formats. It could be represented by digits, words or combination of both. There are different ways to identify and interpret the NSW such as n-gram language model, decision tree, supervised and unsupervised techniques, weighted finite-state transducers and pattern identification using regular expression. In this paper we presented a novel work to identify and interpret the non standard words in Bengali news corpus. We introduced the coupling of the normal text normalization using the three components of analysis, viz. (i) generation of optimized regular expression for different semiotic classes, (ii) consideration of optimized context window size, and (iii) employ the concept of context identification array (CIA).}, }