{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T18:05:07Z","timestamp":1784138707252,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":30,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"Taighde &Eacute;ireann &ndash; Research Ireland","award":["18&#x5c;&#x2f;CRT&#x5c;&#x2f;6223"],"award-info":[{"award-number":["18&#x5c;&#x2f;CRT&#x5c;&#x2f;6223"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,20]]},"DOI":"10.1145\/3805712.3808533","type":"proceedings-article","created":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T14:28:19Z","timestamp":1783693699000},"page":"5256-5258","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Towards Building a Standard Benchmark for Low-Resource Nepali Information Retrieval"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5181-9831","authenticated-orcid":false,"given":"Praveen","family":"Acharya","sequence":"first","affiliation":[{"name":"Dublin City University, Dublin, Ireland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0345-8917","authenticated-orcid":false,"given":"Bal Krishna","family":"Bal","sequence":"additional","affiliation":[{"name":"Kathmandu University, Dhulikhel, Nepal"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.21437\/SLTU.2018-19"},{"key":"e_1_3_2_1_2_1","volume-title":"Proceedings of the 21st International Conference on Natural Language Processing (ICON). 515-521","author":"Adhikari Abiral","year":"2024","unstructured":"Abiral Adhikari, Prashant Manandhar, Reewaj Khanal, Samir Wagle, Praveen Acharya, and Bal Krishna Bal. 2024. Profanity and Offensiveness Detection in Nepali Language Using Bi-directional LSTM Models. In Proceedings of the 21st International Conference on Natural Language Processing (ICON). 515-521."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00288"},{"key":"e_1_3_2_1_4_1","volume-title":"Rupak Raj Ghimire, and Praveen Acharya","author":"Bal Bal Krishna","year":"2024","unstructured":"Bal Krishna Bal, Balaram Prasain, Rupak Raj Ghimire, and Praveen Acharya. 2024. Strategies for Corpus Development for Low-Resource Languages: Insights from Nepal. Automatic Speech Recognition and Translation for Low Resource Languages (2024), 297-330."},{"key":"e_1_3_2_1_5_1","volume-title":"A morphological analyzer and a stemmer for Nepali. PAN Localization, working papers","author":"Bal Bal Krishna","year":"2004","unstructured":"Bal Krishna Bal and Prajol Shrestha. 2004. A morphological analyzer and a stemmer for Nepali. PAN Localization, working papers, Vol. 2007 (2004), 324-31."},{"key":"e_1_3_2_1_6_1","first-page":"21","volume-title":"Nepalese Linguistics","volume":"24","author":"Basnet Santa B","unstructured":"Santa B Basnet and Shailesh B Pandey. [n.d.]. Morphological analysis of verbs in Nepali. Nepalese Linguistics, Vol. 24 ([n.d.]), 21-30."},{"key":"e_1_3_2_1_7_1","volume-title":"Nepali Passport Question Answering: A Low-Resource Dataset for Public Service Applications. arXiv preprint arXiv:2603.13320","author":"Begha Funghang Limbu","year":"2026","unstructured":"Funghang Limbu Begha, Praveen Acharya, and Bal Krishna Bal. 2026. Nepali Passport Question Answering: A Low-Resource Dataset for Public Service Applications. arXiv preprint arXiv:2603.13320 (2026)."},{"key":"e_1_3_2_1_8_1","unstructured":"Marta R Costa-Juss\u00e0 James Cross Onur \u00c7elebi Maha Elbayad Kenneth Heafield Kevin Heffernan Elahe Kalbassi Janice Lam Daniel Licht Jean Maillard et al. 2022. No language left behind: Scaling human-centered machine translation. arXiv preprint arXiv:2207.04672 (2022)."},{"key":"e_1_3_2_1_9_1","volume-title":"Proceedings of the 1st International Conference on Language Technologies for All. European Language Resources Association Paris, France, 375-378","author":"Duwal Sharad","year":"2019","unstructured":"Sharad Duwal and Bal Krishna Bal. 2019. Efforts in the development of an aug-mented english-nepali parallel corpus. In Proceedings of the 1st International Conference on Language Technologies for All. European Language Resources Association Paris, France, 375-378."},{"key":"e_1_3_2_1_10_1","volume-title":"Joint Conference of the Information Retrieval Communities in Europe.","author":"Frej Jibril","year":"2020","unstructured":"Jibril Frej, Didier Schwab, and Jean-Pierre Chevallet. 2020a. MLWIKIR: A Python toolkit for building large-scale Wikipedia-based Information Retrieval Datasets in Chinese, English, French, Italian, Japanese, Spanish and more. In Joint Conference of the Information Retrieval Communities in Europe."},{"key":"e_1_3_2_1_11_1","volume-title":"Proceedings of the Twelfth Language Resources and Evaluation Conference. 1926-1933","author":"Frej Jibril","year":"2020","unstructured":"Jibril Frej, Didier Schwab, and Jean-Pierre Chevallet. 2020b. WIKIR: A Python Toolkit for Building a Large-scale Wikipedia-based English Information Retrieval Dataset. In Proceedings of the Twelfth Language Resources and Evaluation Conference. 1926-1933."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.63317\/37edei5qcjb3"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CCIP.2015.7100739"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-54906-9_5"},{"key":"e_1_3_2_1_15_1","unstructured":"Jang Sen Thakuri Khadga Chandra. 2012. English Code-Mixing in The Nepal Samacharpatra Daily. Ph.D. Dissertation. Central department of English Education."},{"key":"e_1_3_2_1_16_1","volume-title":"Pooja Aggarwal, Rajiv Teja Nagipogu, Shachi Dave, et al.","author":"Khanuja Simran","year":"2021","unstructured":"Simran Khanuja, Diksha Bansal, Sarvesh Mehtani, Savya Khosla, Atreyee Dey, Balaji Gopalan, Dilip Kumar Margam, Pooja Aggarwal, Rajiv Teja Nagipogu, Shachi Dave, et al., 2021. Muril: Multilingual representations for indian languages. arXiv preprint arXiv:2103.10730 (2021)."},{"key":"e_1_3_2_1_17_1","volume-title":"Proceedings of the 1st Annual Meeting of the ELRA\/ISCA Special Interest Group on Under-Resourced Languages. 106-111","author":"Maskey Utsav","year":"2022","unstructured":"Utsav Maskey, Manish Bhatta, Shiva Bhatt, Sanket Dhungel, and Bal Krishna Bal. 2022. Nepali encoder transformers: An analysis of auto encoding transformer language models for nepali text classification. In Proceedings of the 1st Annual Meeting of the ELRA\/ISCA Special Interest Group on Under-Resourced Languages. 106-111."},{"key":"e_1_3_2_1_18_1","volume-title":"Languages in Nepal. Kathmandu: National Statistics Office. (National Population and Housing Census","author":"National Statistics Office. 2025.","year":"2021","unstructured":"National Statistics Office. 2025. Languages in Nepal. Kathmandu: National Statistics Office. (National Population and Housing Census 2021)."},{"key":"e_1_3_2_1_19_1","volume-title":"Consolidating and Developing Benchmarking Datasets for the Nepali Natural Language Understanding Tasks. arXiv preprint arXiv:2411.19244","author":"Nyachhyon Jinu","year":"2024","unstructured":"Jinu Nyachhyon, Mridul Sharma, Prajwal Thapa, and Bal Krishna Bal. 2024. Consolidating and Developing Benchmarking Datasets for the Nepali Natural Language Understanding Tasks. arXiv preprint arXiv:2411.19244 (2024)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.3126\/nelta.v24i1-2.27692"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.63317\/4hiw4enkfuxz"},{"key":"e_1_3_2_1_22_1","unstructured":"Balaram Prasain. 2011. A computational analysis of Nepali morphology: A model for natural language Processing. Ph.D. Dissertation. Faculty of Linguistics."},{"key":"e_1_3_2_1_23_1","volume-title":"Part-of-speech Tagset for Nepali. Madan Puraskar Pustakalaya","author":"Prasain B","year":"2008","unstructured":"B Prasain, LP Khatiwada, BK Bal, and P Shrestha. 2008. Part-of-speech Tagset for Nepali. Madan Puraskar Pustakalaya (2008)."},{"key":"e_1_3_2_1_24_1","volume-title":"Mohammad Aliannejadi, Clemencia Siro, and Guglielmo Faggioli.","author":"Rahmani Hossein A","year":"2024","unstructured":"Hossein A Rahmani, Emine Yilmaz, Nick Craswell, Bhaskar Mitra, Paul Thomas, Charles LA Clarke, Mohammad Aliannejadi, Clemencia Siro, and Guglielmo Faggioli. 2024. Llmjudge: Llms for relevance judgments. arXiv preprint arXiv:2408.08896 (2024)."},{"key":"e_1_3_2_1_25_1","volume-title":"Sentence-bert: Sentence embeddings using siamese bert-networks. arXiv preprint arXiv:1908.10084","author":"Reimers Nils","year":"2019","unstructured":"Nils Reimers and Iryna Gurevych. 2019. Sentence-bert: Sentence embeddings using siamese bert-networks. arXiv preprint arXiv:1908.10084 (2019)."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10462-021-10093-1"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.63317\/2kjhyimcfs9n"},{"key":"e_1_3_2_1_28_1","volume-title":"Development of Pre-Trained Transformer-based Models for the Nepali. COLING 2025","author":"Thapa Prajwal","year":"2025","unstructured":"Prajwal Thapa, Jinu Nyachhyon, Mridul Sharma, and Bal Krishna Bal. 2025. Development of Pre-Trained Transformer-based Models for the Nepali. COLING 2025 (2025), 9."},{"key":"e_1_3_2_1_29_1","volume-title":"Multilingual e5 text embeddings: A technical report. arXiv preprint arXiv:2402.05672","author":"Wang Liang","year":"2024","unstructured":"Liang Wang, Nan Yang, Xiaolong Huang, Linjun Yang, Rangan Majumder, and Furu Wei. 2024. Multilingual e5 text embeddings: A technical report. arXiv preprint arXiv:2402.05672 (2024)."},{"key":"e_1_3_2_1_30_1","volume-title":"mT5: A massively multilingual pre-trained text-to-text transformer. arXiv preprint arXiv:2010.11934","author":"Xue Linting","year":"2020","unstructured":"Linting Xue, Noah Constant, Adam Roberts, Mihir Kale, Rami Al-Rfou, Aditya Siddhant, Aditya Barua, and Colin Raffel. 2020. mT5: A massively multilingual pre-trained text-to-text transformer. arXiv preprint arXiv:2010.11934 (2020)."}],"event":{"name":"SIGIR '26: The 49th International ACM SIGIR Conference on Research and Development in Information Retrieval","location":"Melbourne VIC Australia","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 49th International ACM SIGIR Conference on Research and Development in Information Retrieval"],"original-title":[],"deposited":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:14:39Z","timestamp":1784135679000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805712.3808533"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"references-count":30,"alternative-id":["10.1145\/3805712.3808533","10.1145\/3805712"],"URL":"https:\/\/doi.org\/10.1145\/3805712.3808533","relation":{},"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"2026-07-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}