{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T14:17:35Z","timestamp":1761401855922},"reference-count":55,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2017,6,7]],"date-time":"2017-06-07T00:00:00Z","timestamp":1496793600000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Cogn Comput"],"published-print":{"date-parts":[[2017,10]]},"DOI":"10.1007\/s12559-017-9479-z","type":"journal-article","created":{"date-parts":[[2017,6,7]],"date-time":"2017-06-07T03:31:12Z","timestamp":1496806272000},"page":"671-688","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":12,"title":["An Efficient Corpus-Based Stemmer"],"prefix":"10.1007","volume":"9","author":[{"given":"Jasmeet","family":"Singh","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Vishal","family":"Gupta","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2017,6,7]]},"reference":[{"issue":"1","key":"9479_CR1","doi-asserted-by":"crossref","first-page":"61","DOI":"10.1145\/267954.267957","volume":"16","author":"J Xu","year":"1998","unstructured":"Xu J, Croft WB. Corpus-based stemming using cooccurrence of word variants. ACM Trans Inf Syst. 1998;16(1):61\u201381.","journal-title":"ACM Trans Inf Syst"},{"issue":"2","key":"9479_CR2","doi-asserted-by":"crossref","first-page":"350","DOI":"10.1109\/TSMCB.2006.885307","volume":"37","author":"NL Bhamidipati","year":"2007","unstructured":"Bhamidipati NL, Pal SK. Stemming via distribution-based word segregation for classification and retrieval. IEEE Trans Syst Man Cybern B Cybern: Publ IEEE Syst Man Cybern Soc. 2007;37(2):350\u201360. http:\/\/www.ncbi.nlm.nih.gov\/pubmed\/17416163","journal-title":"IEEE Trans Syst Man Cybern B Cybern: Publ IEEE Syst Man Cybern Soc"},{"issue":"2","key":"9479_CR3","doi-asserted-by":"crossref","first-page":"261","DOI":"10.1007\/s12559-015-9359-3","volume":"8","author":"V Gupta","year":"2016","unstructured":"Gupta V, Kaur N. A novel hybrid text summarization system for Punjabi text. Cogn Comput. 2016;8(2):261\u201377.","journal-title":"Cogn Comput"},{"key":"9479_CR4","unstructured":"Toutanova K, Suzuki H, Ruopp A. Applying morphology generation models to machine translation. In Association for computational linguistics. 2008;pp. 514\u2013522."},{"key":"9479_CR5","unstructured":"Shrivastava M, Bhattacharyya P. Hindi POS tagger using naive stemming: harnessing morphological information without extensive linguistic knowledge. In Proceedings of International Conference on NLP (ICON08). 2008."},{"key":"9479_CR6","doi-asserted-by":"crossref","unstructured":"Krovetz R. Viewing morphology as an inference process. In Proceedings of the 16th Annual International ACM SIGIR Conference on Research and Development in Information Retrieval. 1993;pp. 191\u2013202.","DOI":"10.1145\/160688.160718"},{"key":"9479_CR7","doi-asserted-by":"crossref","first-page":"757","DOI":"10.1007\/s12559-016-9415-7","volume":"8","author":"K Dashtipour","year":"2016","unstructured":"Dashtipour K, Poria S, Hussain A, Cambria E, Hawalah AYA, Gelbukh A, et al. Multilingual sentiment analysis: state of the art and independent comparison of techniques. Cogn Comput. 2016;8:757\u201371. 1\u201315","journal-title":"Cogn Comput"},{"key":"9479_CR8","first-page":"755","volume":"4","author":"M Hu","year":"2004","unstructured":"Hu M, Liu B. Mining opinion features in customer reviews. AAAI. 2004;4:755\u201360.","journal-title":"AAAI"},{"key":"9479_CR9","doi-asserted-by":"crossref","unstructured":"Almeida TA, Silva TP, Santos I, Hidalgo JMG. Text normalization and semantic indexing to enhance instant messaging and SMS spam filtering. Knowledge-Based Systems. 2016.","DOI":"10.1016\/j.knosys.2016.05.001"},{"key":"9479_CR10","first-page":"22","volume":"11","author":"JB Lovins","year":"1968","unstructured":"Lovins JB. Development of a stemming algorithm. Mech Transl Comput Linguist. 1968;11:22\u201331.","journal-title":"Mech Transl Comput Linguist"},{"issue":"3","key":"9479_CR11","first-page":"33","volume":"2","author":"JL Dawson","year":"1974","unstructured":"Dawson JL. Suffix removal for word conflation. Bull Assoc Lit Linguist Comput. 1974;2(3):33\u201346.","journal-title":"Bull Assoc Lit Linguist Comput"},{"issue":"3","key":"9479_CR12","first-page":"130","volume":"14","author":"MF Porter","year":"1980","unstructured":"Porter MF. An algorithm for suffix stripping. Prog Electron Libr Inf Syst. 1980;14(3):130\u20137.","journal-title":"Prog Electron Libr Inf Syst"},{"issue":"3","key":"9479_CR13","doi-asserted-by":"crossref","first-page":"56","DOI":"10.1145\/101306.101310","volume":"24","author":"CD Paice","year":"1990","unstructured":"Paice CD. Another stemmer. ACM SIGIR Forum. 1990;24(3):56\u201361.","journal-title":"ACM SIGIR Forum"},{"key":"9479_CR14","doi-asserted-by":"crossref","unstructured":"Popovic M, Willett P. The effectiveness of stemming for natural-language access to Slovene textual data. J Am Soc Inf Sci. 1992;43:384\u201390.","DOI":"10.1002\/(SICI)1097-4571(199206)43:5<384::AID-ASI6>3.0.CO;2-L"},{"key":"9479_CR15","doi-asserted-by":"crossref","unstructured":"Kraaij W, Pohlman R. Viewing stemming as recall enhancement. In Proceedings of the 19th annual International ACM SIGIR Conference on Research and Development in Information Retrieval. 1996 ;pp. 40\u201348.","DOI":"10.1145\/243199.243209"},{"key":"9479_CR16","doi-asserted-by":"crossref","unstructured":"Majumder P, Mitra M, Pal D. Bulgarian, Hungarian and Czech stemming using YASS. In Advances in multilingual and multimodal information retrieval. 2008;pp. 49\u201356.","DOI":"10.1007\/978-3-540-85760-0_6"},{"key":"9479_CR17","doi-asserted-by":"crossref","unstructured":"Savoy J, Berger PY. Monolingual, bilingual, and GIRT information retrieval at CLEF-2005. In 6th Workshop of the Cross-Language Evaluation Forum, CLEF 2005. 2006;pp. 131\u2013140.","DOI":"10.1007\/11878773_14"},{"key":"9479_CR18","doi-asserted-by":"crossref","unstructured":"Adam G, Asimakis K, Bouras C, Poulopoulos V. An efficient mechanism for stemming and tagging: the case of Greek language. In Proceedings of the 14th International Conference on Knowledge-Based and Intelligent Information and Engineering Systems. 2010:pp. 389\u2013397.","DOI":"10.1007\/978-3-642-15393-8_44"},{"issue":"12","key":"9479_CR19","doi-asserted-by":"crossref","first-page":"2540","DOI":"10.1002\/asi.21191","volume":"60","author":"L Dolamic","year":"2009","unstructured":"Dolamic L, Savoy J. Indexing and searching strategies for the Russian language. J Am Soc Inf Sci Technol. 2009a;60(12):2540\u20137.","journal-title":"J Am Soc Inf Sci Technol"},{"key":"9479_CR20","doi-asserted-by":"crossref","first-page":"714","DOI":"10.1016\/j.ipm.2009.06.001","volume":"45","author":"L Dolamic","year":"2009","unstructured":"Dolamic L, Savoy J. Indexing and stemming approaches for the Czech language. Inf Process Manag. 2009b;45:714\u201320.","journal-title":"Inf Process Manag"},{"key":"9479_CR21","doi-asserted-by":"crossref","unstructured":"Paik JH, Pal D, Parui SK. A novel corpus-based stemming algorithm using co-occurrence statistics. In Proceedings of the 34th Annual International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR\u201911). New York: ACM. 2011b; pp. 863\u2013872.","DOI":"10.1145\/2009916.2010031"},{"issue":"4","key":"9479_CR22","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/2536736.2536738","volume":"31","author":"JH Paik","year":"2013","unstructured":"Paik JH, Parui SK, Pal D, Robertson SE. Effective and robust query-based stemming. ACM Trans Inf Syst. 2013;31(4):1\u201329. doi: 10.1145\/2536736.2536738 .","journal-title":"ACM Trans Inf Syst"},{"key":"9479_CR23","doi-asserted-by":"crossref","unstructured":"Orad D, Levow G, Cabezas C. CLEF experiments at Maryland: statistical stemming and back off translation. In: Proceedings of the Workshop of Cross-Language Evaluation Forum on Cross-Language Information Retrieval and Evaluation. Berlin: Springer-Verlag. 2001;pp. 176\u2013187.","DOI":"10.1007\/3-540-44645-1_17"},{"issue":"2","key":"9479_CR24","doi-asserted-by":"crossref","first-page":"153","DOI":"10.1162\/089120101750300490","volume":"27","author":"J Goldsmith","year":"2001","unstructured":"Goldsmith J. Unsupervised learning of the morphology of a natural language. J Comput Linguist. 2001;27(2):153\u201398.","journal-title":"J Comput Linguist"},{"issue":"04","key":"9479_CR25","doi-asserted-by":"crossref","first-page":"353","DOI":"10.1017\/S1351324905004055","volume":"12","author":"J Goldsmith","year":"2006","unstructured":"Goldsmith J. An algorithm for the unsupervised learning of morphology. Nat Lang Eng. 2006;12(04):353\u201371.","journal-title":"Nat Lang Eng"},{"key":"9479_CR26","doi-asserted-by":"crossref","unstructured":"Melucci M, Orio N. A novel method for stemmer generation based on hidden Markov models. In Proceedings of the twelfth International Conference on Information and Knowledge Management (CIKM\u201903). 2003;pp. 131\u2013138.","DOI":"10.1145\/956863.956889"},{"issue":"1","key":"9479_CR27","doi-asserted-by":"crossref","first-page":"121","DOI":"10.1016\/j.ipm.2004.04.006","volume":"41","author":"M Bacchin","year":"2005","unstructured":"Bacchin M, Ferro N, Melucci M. A probabilistic model for stemmer generation. Inf Process Manag. 2005;41(1):121\u201337.","journal-title":"Inf Process Manag"},{"key":"9479_CR28","doi-asserted-by":"crossref","unstructured":"Bacchin M, Ferro N, Melucci M. The effectiveness of a graph-based algorithm for stemming. In Digital libraries: people, knowledge, and technology. Springer; 2002. pp. 117\u2013128.","DOI":"10.1007\/3-540-36227-4_12"},{"issue":"1","key":"9479_CR29","doi-asserted-by":"crossref","first-page":"3","DOI":"10.1145\/1187415.1187418","volume":"4","author":"M Creutz","year":"2007","unstructured":"Creutz M, Lagus K. Unsupervised models for morpheme segmentation and morphology learning. ACM Trans Speech Lang Process (TSLP). 2007;4(1):3. article","journal-title":"ACM Trans Speech Lang Process (TSLP)"},{"key":"9479_CR30","doi-asserted-by":"crossref","unstructured":"Creutz M, Lagus K. Unsupervised discovery of morphemes. In Proceedings of the ACL-02 workshop on Morphological and phonological learning. 2002; Vol. 6: pp. 21\u201330.","DOI":"10.3115\/1118647.1118650"},{"key":"9479_CR31","doi-asserted-by":"crossref","unstructured":"Creutz M. Unsupervised segmentation of words using prior distributions of morph length and frequency. In Proceedings of the 41st Annual Meeting of Association for Computational Linguistics. 2003;Vol. 1: pp. 280\u2013287.","DOI":"10.3115\/1075096.1075132"},{"key":"9479_CR32","doi-asserted-by":"crossref","unstructured":"Creutz M, Lagus K. Induction of a simple morphology for highly-inflecting languages. In Proceedings of the 7th Meeting of the ACL Special Interest Group in Computational Phonology: Current Themes in Computational Phonology and Morphology. 2004:pp. 43\u201351.","DOI":"10.3115\/1622153.1622159"},{"key":"9479_CR33","unstructured":"Creutz M, Lagus K. Inducing the morphological lexicon of a natural language from unannotated text. In Proceedings of the International and Interdisciplinary Conference on Adaptive Knowledge Representation and Reasoning (AKRR\u201905). 2005; Vol. 1: pp. 51\u201359."},{"key":"9479_CR34","doi-asserted-by":"crossref","unstructured":"Kohonen O, Virpioja S, Klami M. Allomorfessor: towards unsupervised morpheme analysis. In Evaluating Systems for Multilingual and Multimodal Information Acces. Springer: 2008; pp. 975\u2013982.","DOI":"10.1007\/978-3-642-04447-2_129"},{"issue":"4","key":"9479_CR35","doi-asserted-by":"crossref","first-page":"18","DOI":"10.1145\/1281485.1281489","volume":"25","author":"P Majumder","year":"2007","unstructured":"Majumder P, Mitra M, Parui SK, Kole G, Mitra P, Datta K. YASS: Yet Another Suffix Stripper. ACM Trans Inf Syst. 2007;25(4):18.","journal-title":"ACM Trans Inf Syst"},{"issue":"5\u20137","key":"9479_CR36","doi-asserted-by":"crossref","first-page":"491","DOI":"10.1002\/sim.4780140510","volume":"14","author":"MA Jaro","year":"1995","unstructured":"Jaro MA. Probabilistic linkage of large public health data files. Stat Med. 1995;14(5\u20137):491\u20138.","journal-title":"Stat Med"},{"key":"9479_CR37","unstructured":"Winkler WE. String comparator metrics and enhanced decision rules in the Fellegi-Sunter model of record linkage. 1990."},{"key":"9479_CR38","doi-asserted-by":"crossref","unstructured":"Makin R, Pandey N, Pingali P, Varma V. Approximate string matching techniques for effective CLIR among Indian languages. In International Workshop on Fuzzy Logic and Applications. 2007;pp. 430\u2013437.","DOI":"10.1007\/978-3-540-73400-0_54"},{"issue":"5","key":"9479_CR39","doi-asserted-by":"crossref","first-page":"16","DOI":"10.1109\/MIS.2003.1234765","volume":"18","author":"M Bilenko","year":"2003","unstructured":"Bilenko M, Mooney R, Cohen W, Ravikumar P, Fienberg S. Adaptive name matching in information integration. IEEE Intell Syst. 2003;18(5):16\u201323.","journal-title":"IEEE Intell Syst"},{"key":"9479_CR40","doi-asserted-by":"crossref","unstructured":"Christen P. A comparison of personal name matching: techniques and practical issues. In Sixth IEEE International Conference on Data Mining-Workshops (ICDMW\u201906). 2006;pp. 290\u2013294.","DOI":"10.1109\/ICDMW.2006.2"},{"key":"9479_CR41","unstructured":"Cohen W, Ravikumar P, Fienberg S. A comparison of string metrics for matching names and records. In KDD workshop on data cleaning and object consolidation. 2003 ;Vol. 3: pp. 73\u201378."},{"issue":"4","key":"9479_CR42","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/2037661.2037664","volume":"29","author":"J Paik","year":"2011","unstructured":"Paik J, Mitra M, Parui S, Jarvelin K. GRAS: an effective and efficient stemming algorithm for information retrieval. ACM Trans Inf Syst. 2011a;29(4):1\u201324.","journal-title":"ACM Trans Inf Syst"},{"issue":"2","key":"9479_CR43","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/1967293.1967295","volume":"10","author":"JH Paik","year":"2011","unstructured":"Paik JH, Parui SK. A fast corpus-based stemmer. ACM Trans Asian Lang Inf Process. 2011;10(2):1\u201316. doi: 10.1145\/1967293.1967295 .","journal-title":"ACM Trans Asian Lang Inf Process"},{"key":"9479_CR44","doi-asserted-by":"crossref","unstructured":"Peng F, Lu Y. Context Sensitive Stemming for Web Search. Proceeding SIGIR '07 Proceedings of the 30th annual international ACM SIGIR conference on Research and development in information retrieval. 2007;pp 639\u201346.","DOI":"10.1145\/1277741.1277851"},{"issue":"1","key":"9479_CR45","doi-asserted-by":"crossref","first-page":"68","DOI":"10.1016\/j.ipm.2014.08.006","volume":"51","author":"T Brychc\u00edn","year":"2015","unstructured":"Brychc\u00edn T, Konop\u00edk M. HPS: high precision stemmer. Inf Process Manag. 2015;51(1):68\u201391.","journal-title":"Inf Process Manag"},{"key":"9479_CR46","doi-asserted-by":"crossref","unstructured":"McNamee P, Nicholas C, Mayfield J. Addressing morphological variation in alphabetic languages. In Proceedings of the 32nd international ACM SIGIR conference on Research and development in information retrieval. 2009; pp. 75\u201382.","DOI":"10.1145\/1571941.1571957"},{"issue":"2","key":"9479_CR47","first-page":"2","volume":"7","author":"A Pirkola","year":"2002","unstructured":"Pirkola A, Keskustalo H, Lepp\u00e4nen E, K\u00e4ns\u00e4l\u00e4 A-P, J\u00e4rvelin K. Targeted s-gram matching: a novel n-gram matching technique for cross and monolingual word form variants. Inf Res. 2002;7(2):2\u20137.","journal-title":"Inf Res"},{"key":"9479_CR48","unstructured":"J\u00e4rvelin A. Applications of S-grams in natural language information retrieval. 2014."},{"issue":"3","key":"9479_CR49","doi-asserted-by":"crossref","first-page":"11","DOI":"10.1145\/1838745.1838748","volume":"9","author":"L Dolamic","year":"2010","unstructured":"Dolamic L, Savoy J. Comparative study of indexing and search strategies for the Hindi, Marathi, and Bengali languages. ACM Trans Asian Lang Inf Process. 2010;9(3):11.","journal-title":"ACM Trans Asian Lang Inf Process"},{"issue":"406","key":"9479_CR50","doi-asserted-by":"crossref","first-page":"414","DOI":"10.1080\/01621459.1989.10478785","volume":"84","author":"MA Jaro","year":"1989","unstructured":"Jaro MA. Advances in record-linkage methodology as applied to matching the 1985 census of Tampa, Florida. J Am Stat Assoc. 1989;84(406):414\u201320.","journal-title":"J Am Stat Assoc"},{"issue":"4","key":"9479_CR51","first-page":"467","volume":"18","author":"PF Brown","year":"1992","unstructured":"Brown PF, Desouza PV, Mercer RL, Pietra V, Della J, Lai JC. Class-based n-gram models of natural language. Comput Linguist. 1992;18(4):467\u201379.","journal-title":"Comput Linguist"},{"issue":"3","key":"9479_CR52","doi-asserted-by":"crossref","first-page":"264","DOI":"10.1145\/331499.331504","volume":"31","author":"AK Jain","year":"1999","unstructured":"Jain AK, Murty MN, Flynn PJ. Data clustering: a review. ACM Comput Surv. 1999;31(3):264\u2013323.","journal-title":"ACM Comput Surv"},{"issue":"4","key":"9479_CR53","doi-asserted-by":"crossref","first-page":"357","DOI":"10.1145\/582415.582416","volume":"20","author":"G Amati","year":"2002","unstructured":"Amati G, Van Rijsbergen CJ. Probabilistic models of information retrieval based on measuring the divergence from randomness. ACM Trans Inf Syst (TOIS). 2002;20(4):357\u201389.","journal-title":"ACM Trans Inf Syst (TOIS)"},{"key":"9479_CR54","doi-asserted-by":"publisher","unstructured":"Singh J, Gupta V. A systematic review of text stemming techniques. Artif Intell Rev. 2016:1\u201361. article. doi: 10.1007\/s10462-016-9498-2 .","DOI":"10.1007\/s10462-016-9498-2"},{"issue":"2","key":"9479_CR55","doi-asserted-by":"crossref","first-page":"111","DOI":"10.1145\/1105696.1105699","volume":"4","author":"T Sakai","year":"2005","unstructured":"Sakai T, Manabe T, Koyama M. Flexible pseudo-relevance feedback via selective sampling. ACM Trans Asian Lang Inf Process (TALIP). 2005;4(2):111\u201335.","journal-title":"ACM Trans Asian Lang Inf Process (TALIP)"}],"container-title":["Cognitive Computation"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s12559-017-9479-z\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s12559-017-9479-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s12559-017-9479-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,9,25]],"date-time":"2019-09-25T17:51:17Z","timestamp":1569433877000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s12559-017-9479-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017,6,7]]},"references-count":55,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2017,10]]}},"alternative-id":["9479"],"URL":"https:\/\/doi.org\/10.1007\/s12559-017-9479-z","relation":{},"ISSN":["1866-9956","1866-9964"],"issn-type":[{"value":"1866-9956","type":"print"},{"value":"1866-9964","type":"electronic"}],"subject":[],"published":{"date-parts":[[2017,6,7]]}}}