{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,11]],"date-time":"2025-09-11T19:03:17Z","timestamp":1757617397859,"version":"3.44.0"},"reference-count":41,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2024,12,7]],"date-time":"2024-12-07T00:00:00Z","timestamp":1733529600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,12,7]],"date-time":"2024-12-07T00:00:00Z","timestamp":1733529600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Lang Resources &amp; Evaluation"],"published-print":{"date-parts":[[2025,9]]},"DOI":"10.1007\/s10579-024-09783-3","type":"journal-article","created":{"date-parts":[[2024,12,7]],"date-time":"2024-12-07T09:37:40Z","timestamp":1733564260000},"page":"3215-3241","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["PinLID: a dataset for Pinglish language identiftcation based on code-mixing sentence on unstructured resources"],"prefix":"10.1007","volume":"59","author":[{"given":"Arash","family":"Ghafouri","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hasan","family":"Naderi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mahdi","family":"Firouzmandi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,12,7]]},"reference":[{"key":"9783_CR1","doi-asserted-by":"crossref","unstructured":"Abbasi, M. A., Ghafouri, A., Firouzmandi, M., Naderi, H., & Bidgoli, B. M. (2023). \u201cPersianLLaMA: Towards Building first Persian large language model.\u201d ArXiv Preprint at http:\/\/arxiv.org\/abs\/2312.15713.","DOI":"10.21203\/rs.3.rs-3789059\/v1"},{"key":"9783_CR39","doi-asserted-by":"publisher","unstructured":"Arora, M., & Kansal, V. (2019). Character level embedding with deep convolutional neural network for text normalization of unstructured data for Twitter sentiment analysis. Social Network Analysis and Mining, 9, 12. https:\/\/doi.org\/10.1007\/s13278-019-0557-y","DOI":"10.1007\/s13278-019-0557-y"},{"key":"9783_CR2","doi-asserted-by":"publisher","DOI":"10.29240\/ef.v5i2.2619","author":"A Asrifan","year":"2021","unstructured":"Asrifan, A., Abdullah, H., Muthmainnah, M., & Patil, A. A. P. (2021). An analysis of code mixing in the MOVIE \u2018from London to Bali. Academic Journal of English Language and Education. https:\/\/doi.org\/10.29240\/ef.v5i2.2619","journal-title":"Academic Journal of English Language and Education"},{"key":"9783_CR41","doi-asserted-by":"crossref","unstructured":"Barik, A.M., Mahendra, R. & Adriani, M. (2019). Normalization of Indonesian-English Code-Mixed Twitter Data. In Proceedings of the 5th Workshop on Noisy User-generated Text (W-NUT 2019), Hong Kong, China (pp. 417\u2013424). Association for Computational Linguistics.","DOI":"10.18653\/v1\/D19-5554"},{"issue":"2","key":"9783_CR3","doi-asserted-by":"publisher","first-page":"131","DOI":"10.1177\/13670069000040020101","volume":"4","author":"R Barnett","year":"2000","unstructured":"Barnett, R., Cod\u00f3, E., Eppler, E., Forcadell, M., Gardner-Chloros, P., van Hout, R., Moyer, M., et al. (2000). The LIDES coding manual: A document for preparing and analyzing language interaction data version. International Journal of Bilingualism, 4(2), 131\u2013132. https:\/\/doi.org\/10.1177\/13670069000040020101","journal-title":"International Journal of Bilingualism"},{"key":"9783_CR4","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.747","volume-title":"Unsupervised cross-lingual representation learning at scale","author":"A Conneau","year":"2020","unstructured":"Conneau, A., Khandelwal, K., Goyal, N., Chaudhary, V., Wenzek, G., Guzm\u00e1n, F., Grave, E., Ott, Myle, Zettlemoyer, L., & Stoyanov, V. (2020). Unsupervised cross-lingual representation learning at scale. ACL."},{"key":"9783_CR5","unstructured":"Darvishi, K., Javdan, S., Minaei-Bidgoli, B., & Eetemadi, S. (2022). \u201cPars-{ABSA}: A manually annotated aspect-based sentiment analysis benchmark on {F}arsi Product Reviews.\u201d In Proceedings of the Thirteenth Language Resources and Evaluation Conference, (pp. 7056\u201360). Marseille, France: European Language Resources Association."},{"key":"9783_CR6","unstructured":"Das, A., & Gamb\u00e4ck, B. (2014). \u201cIdentifying languages at the word level in code-mixed indian social media text.\u201d In Dipti Misra Sharma, Rajeev Sangal, and Jyoti D Pawar (Eds.), Proceedings of the 11th International Conference on Natural Language Processing, (pp. 378\u201387). Goa, India: NLP Association of India. https:\/\/aclanthology.org\/W14-5152."},{"key":"9783_CR7","doi-asserted-by":"publisher","unstructured":"Devlin, J., Ming W. C., Kenton, L., and Kristina, T. (2019a) \u201c{BERT:} Pre-training of deep bidirectional transformers for language understanding.\u201d In Jill Burstein, Christy Doran, and Thamar Solorio (Eds.), Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, {NAACL-HLT} 2019, Minneapolis, MN, USA, June 2\u20137, 2019, Volume 1 (Long and Short Papers), (pp. 4171\u201386). Association for Computational Linguistics. https:\/\/doi.org\/10.18653\/v1\/n19-1423.","DOI":"10.18653\/v1\/n19-1423"},{"key":"9783_CR8","unstructured":"Devlin, J., Ming W. C., Kenton, L., and Kristina, T. (2019b). \u201cBERT: Pre-training of deep bidirectional transformers for language understanding.\u201d NAACL HLT 2019 - 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies\u2014Proceedings of the Conference 1: (pp. 4171\u201386)"},{"key":"9783_CR9","unstructured":"Ekbal, A., & Bandyopadhyay, S. (2008). \u201cBengali named entity recognition using support vector machine.\u201d In Proceedings of the {IJCNLP}-08 Workshop on Named Entity Recognition for South and South East {A}sian Languages. https:\/\/aclanthology.org\/I08-5008."},{"key":"9783_CR10","unstructured":"Farahani, M., Gharachorloo, M., Farahani, M., & Manthouri, M. (2020). \u201cParsBERT: Transformer-based model for Persian language understanding.\u201d Preprint at http:\/\/arxiv.org\/abs\/2005.12515."},{"key":"9783_CR11","unstructured":"Gamb\u00e4ck, B., & Das, A. (2016). \u201cComparing the level of code-switching in corpora.\u201d In Nicoletta Calzolari, Khalid Choukri, Thierry Declerck, Sara Goggi, Marko Grobelnik, Bente Maegaard, Joseph Mariani (Eds.), Proceedings of the Tenth International Conference on Language Resources and Evaluation ({LREC}\u201916), (pp. 1850\u201355). Portoro\u017e, Slovenia: European Language Resources Association (ELRA). https:\/\/aclanthology.org\/L16-1292."},{"key":"9783_CR40","doi-asserted-by":"publisher","unstructured":"Gautam, G., & Yadav, D. (2014). Sentiment analysis of twitter data using machine learning approaches and semantic analysis. Seventh International Conference on Contemporary Computing (IC3), Noida, India (pp. 437-442). https:\/\/doi.org\/10.1109\/IC3.2014.6897213","DOI":"10.1109\/IC3.2014.6897213"},{"key":"9783_CR12","doi-asserted-by":"publisher","unstructured":"Ghafouri, A., Abbasi, M. A., & Naderi, H. (2023). \u201cAriaBERT: A pre-trained Persian BERT Model for natural language understanding.\u201d https:\/\/doi.org\/10.21203\/rs.3.rs-3558473\/v1.","DOI":"10.21203\/rs.3.rs-3558473\/v1"},{"key":"9783_CR13","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1429","volume-title":"Metrics for modeling code-switching across corpora","author":"GA Guzm\u00e1n","year":"2017","unstructured":"Guzm\u00e1n, G. A., Ricard, J., Serigos, J., Bullock, B. E., & Toribio, A. J. (2017). Metrics for modeling code-switching across corpora. In Interspeech."},{"key":"9783_CR14","unstructured":"\u201cHistorical Yearly Trends in the Usage Statistics of Content Languages for Websites, May 2024.\u201d (2024). https:\/\/w3techs.com\/technologies\/history_overview\/content_language\/ms\/y."},{"key":"9783_CR15","doi-asserted-by":"publisher","unstructured":"Jayanthi, S. M., Nerella, K., Chandu, K. R., & Black, A. W. (2021). \u201c{C}odemixed{NLP}: An extensible and open {NLP} toolkit for code-mixing.\u201d In Proceedings of the Fifth Workshop on Computational Approaches to Linguistic Code-Switching, (pp. 113\u201318). Online: Association for Computational Linguistics. https:\/\/doi.org\/10.18653\/v1\/2021.calcs-1.14.","DOI":"10.18653\/v1\/2021.calcs-1.14"},{"key":"9783_CR16","unstructured":"Kingma, DP., and J, Ba. (2017). \u201cAdam: A method for stochastic optimization.\u201d"},{"key":"9783_CR17","doi-asserted-by":"crossref","unstructured":"Liu, J., Chen, X., Feng, S., Wang, S., Ouyang, X., Sun, Y., & Su, W. (2020). \u201cKk2018 at SemEval-2020 Task 9: Adversarial training for code-mixing sentiment classification.\u201d Preprint at http:\/\/arxiv.org\/abs\/2009.0.","DOI":"10.18653\/v1\/2020.semeval-1.103"},{"key":"9783_CR18","unstructured":"Liu, Y., M, Ott., N, Goyal., J, Du., M, Joshi., D, Chen., O, Levy., M, Lewis, L, Zettlemoyer., and V, Stoyanov. (2019). \u201cRoBERTa: A robustly optimized BERT pretraining approach.\u201d"},{"key":"9783_CR19","unstructured":"Mandal, S., Mahata, S. K., & Das, D. (2018). \u201cPreparing Bengali\u2013English code-mixed corpus for sentiment analysis of Indian languages.\u201d Preprint at http:\/\/arxiv.org\/abs\/1803.0."},{"key":"9783_CR20","doi-asserted-by":"publisher","unstructured":"Mandal, S., & Singh, A. K. (2018). \u201cLanguage identification in code-mixed data using multichannel neural networks and context capture.\u201d In Proceedings of the 2018 {EMNLP} Workshop W-{NUT}: The 4th Workshop on noisy user-generated text, (pp. 116\u201320). Brussels, Belgium: Association for Computational Linguistics. https:\/\/doi.org\/10.18653\/v1\/W18-6116.","DOI":"10.18653\/v1\/W18-6116"},{"key":"9783_CR21","unstructured":"Mandal, S., Das, S. D., & Das, D. (2019). \u201cLanguage identification of Bengali\u2013English code-mixed data using character & phonetic based LSTM models.\u201d Proceedings of the 11th Forum for Information Retrieval Evaluation"},{"key":"9783_CR22","doi-asserted-by":"publisher","unstructured":"Meishan, Z., Yue, Z., & Guohong, F. (2019). \u201cCross-Lingual dependency parsing using code-mixed {T}ree{B}ank.\u201d In Proceedings of the 2019 Conference on empirical methods in natural language processing and the 9th international joint conference on natural language processing (EMNLP-IJCNLP), (pp. 997\u20131006). Hong Kong, China: Association for Computational Linguistics. https:\/\/doi.org\/10.18653\/v1\/D19-1092.","DOI":"10.18653\/v1\/D19-1092"},{"key":"9783_CR23","doi-asserted-by":"crossref","unstructured":"Patwa, P., Aguilar, G., Kar, S., Pandey, S., Pykl, S., Gamb\u00e4ck, B., & Das, A. (2020). \u201cSemEval-2020 Task 9: Overview of sentiment analysis of code-mixed tweets.\u201d Preprint at http:\/\/arxiv.org\/abs\/2008.04277","DOI":"10.18653\/v1\/2020.semeval-1.100"},{"key":"9783_CR24","unstructured":"Prabhu, A., Joshi, A., Shrivastava, M., & Varma, V. (2016). \u201cTowards sub-word level compositions for sentiment analysis of Hindi-English code mixed text.\u201d CoRR abs\/1611.0."},{"key":"9783_CR25","doi-asserted-by":"publisher","unstructured":"Pratapa, A., Bhat, G., Choudhury, M., Sitaram, S., Dandapat, S., & Bali, K. (2018). \u201cLanguage modeling for code-mixing: The role of linguistic theory based synthetic data.\u201d In Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), (pp. 1543\u201353). Melbourne, Australia: Association for Computational Linguistics. https:\/\/doi.org\/10.18653\/v1\/P18-1143.","DOI":"10.18653\/v1\/P18-1143"},{"key":"9783_CR26","doi-asserted-by":"crossref","unstructured":"Ramanarayanan V, Pugh R, Qian Y, Suendermann-Oeft D (2019) Automatic Turn-Level Language Identification for Code-Switched Spanish-English Dialog, In Luis Fernando DHaro, Rafael E Banchs, and Haizhou Li (Eds.), International Workshop on Spoken Dialogue System Technology, Springer, Cham","DOI":"10.1007\/978-981-13-9443-0_5"},{"key":"9783_CR27","doi-asserted-by":"crossref","unstructured":"Sabri, N., Edalat, A., & Bahrak, B. (2021). \u201cSentiment analysis of Persian-English code-mixed texts.\u201d 2021 26th International Computer Conference, Computer Society of Iran (CSICC), (pp. 1\u20134).","DOI":"10.1109\/CSICC52343.2021.9420605"},{"key":"9783_CR28","doi-asserted-by":"publisher","unstructured":"Shanmugalingam, K., Sumathipala, S., & Premachandra, C. (2018). \u201cWord level language identification of code mixing text in social media using NLP.\u201d In 2018 3rd International Conference on Information Technology Research (ICITR), (pp. 1\u20135). https:\/\/doi.org\/10.1109\/ICITR.2018.8736127.","DOI":"10.1109\/ICITR.2018.8736127"},{"issue":"6","key":"9783_CR29","doi-asserted-by":"publisher","first-page":"2050023","DOI":"10.1142\/S0217984920500864","volume":"34","author":"S Shekhar","year":"2020","unstructured":"Shekhar, S., Sharma, D. K., & Sufyan Beg, M. M. (2020). Language identification framework in code-mixed social media text based on quantum LSTM {\\textemdash} the word belongs to which language? Modern Physics Letters B, 34(6), 2050023\u201386. https:\/\/doi.org\/10.1142\/S0217984920500864","journal-title":"Modern Physics Letters B"},{"key":"9783_CR30","doi-asserted-by":"publisher","unstructured":"Singh, K., Sen, I., Kumaraguru, P. 2018b. \u201cLanguage Identification and Named Entity Recognition in {H}inglish code mixed tweets.\u201d Proceedings of {ACL} 2018, Student Research Workshop, (pp. 52\u201358). Melbourne, Australia: Association for Computational Linguistics. https:\/\/doi.org\/10.18653\/v1\/P18-3008.","DOI":"10.18653\/v1\/P18-3008"},{"key":"9783_CR31","doi-asserted-by":"publisher","unstructured":"Singh K, Sen I, Kumaraguru P. (2018a). \u201cA {T}witter corpus for {H}indi-{E}nglish code mixed {POS} Tagging.\u201d In Proceedings of the sixth international workshop on natural language processing for Social Media, (pp. 12\u201317). Melbourne, Australia: Association for Computational Linguistics. https:\/\/doi.org\/10.18653\/v1\/W18-3503.","DOI":"10.18653\/v1\/W18-3503"},{"key":"9783_CR32","doi-asserted-by":"crossref","unstructured":"Srivastava, V., Singh, M. (2021). \u201cHinGE: A dataset for generation and evaluation of code-mixed Hinglish text.\u201d Preprint at http:\/\/arxiv.org\/abs\/2107.0","DOI":"10.18653\/v1\/2021.eval4nlp-1.20"},{"key":"9783_CR33","unstructured":"Swami, S., Khandelwal, A., Singh, V., Akhtar, SS., Shrivastava, M. (2018). \u201cA corpus of English-Hindi code-mixed tweets for sarcasm detection.\u201d Preprint at http:\/\/arxiv.org\/abs\/1805.1."},{"key":"9783_CR34","doi-asserted-by":"publisher","unstructured":"Tan, S., & Joty, S. (2021). \u201cCode-mixing on sesame street: Dawn of the adversarial polyglots.\u201d In Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, (pp. 3596\u20133616). Association for Computational Linguistics. https:\/\/doi.org\/10.18653\/v1\/2021.naacl-main.282.","DOI":"10.18653\/v1\/2021.naacl-main.282"},{"key":"9783_CR35","doi-asserted-by":"publisher","unstructured":"Utsav, J., Kabaria, D., Vajpeyi, R., Mina, M., & Srivastava, V. (2020). \u201cStance detection in Hindi-English code-mixed data.\u201d In Proceedings of the 7th ACM IKDD CoDS and 25th COMAD, (pp. 359\u2013360). New York, NY, USA: Association for Computing Machinery. https:\/\/doi.org\/10.1145\/3371158.3371226.","DOI":"10.1145\/3371158.3371226"},{"key":"9783_CR36","doi-asserted-by":"crossref","unstructured":"Vyas, Y., Gella, S., Sharma, J., Bali, K., & Choudhury, M (2014). \u201cPOS tagging of English\u2013Hindi code-mixed social media content.\u201d In Proceedings of the 2014 conference on empirical methods in natural language processing (EMNLP), (pp. 974\u201379). Association for Computational Linguistics.","DOI":"10.3115\/v1\/D14-1105"},{"key":"9783_CR37","unstructured":"You, Y., Li, J., Reddi, S., Hseu, J., Kumar, S., Bhojanapalli, S., & Hsieh, C. J. (2020). \u201cLarge batch optimization for deep learning: Training BERT in 76 Minutes.\u201d In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=Syx4wnEtvH."},{"key":"9783_CR38","doi-asserted-by":"publisher","unstructured":"Zampieri, M., Malmasi, S., Nakov, P., Rosenthal, S., Farra, N., & Kumar, R. (2019). \u201cPredicting the type and target of offensive posts in social media\u201d, Proceedings of the 2019 Conference of the North {A}merican chapter of the association for computational linguistics: Human language technologies, Volume 1 (Long and Short Papers), (pp. 1415\u201320). Minneapolis, Minnesota: Association for Computational Linguistics. https:\/\/doi.org\/10.18653\/v1\/N19-1144.","DOI":"10.18653\/v1\/N19-1144"}],"container-title":["Language Resources and Evaluation"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10579-024-09783-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10579-024-09783-3\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10579-024-09783-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,6]],"date-time":"2025-09-06T01:55:27Z","timestamp":1757123727000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10579-024-09783-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,7]]},"references-count":41,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2025,9]]}},"alternative-id":["9783"],"URL":"https:\/\/doi.org\/10.1007\/s10579-024-09783-3","relation":{},"ISSN":["1574-020X","1574-0218"],"issn-type":[{"type":"print","value":"1574-020X"},{"type":"electronic","value":"1574-0218"}],"subject":[],"published":{"date-parts":[[2024,12,7]]},"assertion":[{"value":"27 September 2024","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 December 2024","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no relevant financial or non-financial interests to disclose. The authors have no conflicts of interest to declare that are relevant to the content of this article. All authors certify that they have no affiliations with or involvement in any organization or entity with any financial interest or non-financial interest in the subject or materials discussed in this manuscript. The authors have no financial or proprietary interests in any material discussed in this article.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of Interests"}}]}}