{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T18:08:40Z","timestamp":1784138920606,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":44,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"National Natural Science Foundation of China","award":["62566045"],"award-info":[{"award-number":["62566045"]}]},{"name":"Inner Mongolia Autonomous Region Science and Technology Project","award":["2025KYPT0041"],"award-info":[{"award-number":["2025KYPT0041"]}]},{"name":"Inner Mongolia Autonomous Region Science and Technology Project","award":["2025KYPT0064"],"award-info":[{"award-number":["2025KYPT0064"]}]},{"name":"Inner Mongolia Autonomous Region First-Class Discipline Scientific Research Special Projec","award":["YLXKZX-ND-036"],"award-info":[{"award-number":["YLXKZX-ND-036"]}]},{"name":"Science and Technology Major Project of Inner Mongolia Autonomous Region","award":["2025ZDSF0029"],"award-info":[{"award-number":["2025ZDSF0029"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,20]]},"DOI":"10.1145\/3805712.3808632","type":"proceedings-article","created":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T14:28:19Z","timestamp":1783693699000},"page":"3175-3182","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["TM-Bench: Benchmarking Large Language Models on Low-Resource Traditional Mongolian"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-6287-2472","authenticated-orcid":false,"given":"Zhenjie","family":"Gao","sequence":"first","affiliation":[{"name":"College of Computer Science, Inner Mongolia University, Hohhot, Inner Mongolia, China, National &amp;#38; Local Joint Engineering Research Center of Intelligent Information Processing Technology for Mongolian, Hohhot, Inner Mongolia, China, and Inner Mongolia Key Laboratory of Multilingual Artificial Intelligence Technology, Hohhot, Inner Mongolia, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7312-1629","authenticated-orcid":false,"given":"Feilong","family":"Bao","sequence":"additional","affiliation":[{"name":"College of Computer Science, Inner Mongolia University, Hohhot, Inner Mongolia, China, National &amp;#38; Local Joint Engineering Research Center of Intelligent Information Processing Technology for Mongolian, Hohhot, Inner Mongolia, China, and Inner Mongolia Key Laboratory of Multilingual Artificial Intelligence Technology, Hohhot, Inner Mongolia, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-9388-5828","authenticated-orcid":false,"given":"Aruukhan","family":"Bai","sequence":"additional","affiliation":[{"name":"College of Computer Science, Inner Mongolia University, Hohhot, Inner Mongolia, China, National &amp;#38; Local Joint Engineering Research Center of Intelligent Information Processing Technology for Mongolian, Hohhot, Inner Mongolia, China, and Inner Mongolia Key Laboratory of Multilingual Artificial Intelligence Technology, Hohhot, Inner Mongolia, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-8514-0363","authenticated-orcid":false,"given":"Ruichen","family":"Hou","sequence":"additional","affiliation":[{"name":"College of Computer Science, Inner Mongolia University, Hohhot, Inner Mongolia, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-0720-5647","authenticated-orcid":false,"given":"Xieqi","family":"Ji","sequence":"additional","affiliation":[{"name":"College of Computer Science, Inner Mongolia University, Hohhot, Inner Mongolia, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-8542-1336","authenticated-orcid":false,"given":"Dabalgan","family":"Wang","sequence":"additional","affiliation":[{"name":"College of Computer Science, Inner Mongolia University, Hohhot, Inner Mongolia, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-4993-5854","authenticated-orcid":false,"given":"Hugjil","family":"Ming","sequence":"additional","affiliation":[{"name":"College of Computer Science, Inner Mongolia University, Hohhot, Inner Mongolia, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-0802-9203","authenticated-orcid":false,"given":"Yuan","family":"Li","sequence":"additional","affiliation":[{"name":"College of Computer Science, Inner Mongolia University, Hohhot, Inner Mongolia, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al.","author":"Achiam Josh","year":"2023","unstructured":"Josh Achiam, Steven Adler, Sandhini Agarwal, Lama Ahmad, Ilge Akkaya, Florencia Leoni Aleman, Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al., 2023. Gpt-4 technical report. arXiv preprint arXiv:2303.08774 (2023)."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.3724\/2096-7004.di.2025.0008"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.914"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.376"},{"key":"e_1_3_2_1_5_1","unstructured":"Tom Brown Benjamin Mann Nick Ryder Melanie Subbiah Jared D Kaplan Prafulla Dhariwal Arvind Neelakantan Pranav Shyam Girish Sastry Amanda Askell et al. 2020. Language models are few-shot learners. Advances in neural information processing systems Vol. 33 (2020) 1877-1901."},{"key":"e_1_3_2_1_6_1","volume-title":"Contemporary Mongolian population distribution, migration, cultural change, and identity. China's Minorities on the Move","author":"Burjgin Jirgal","year":"2015","unstructured":"Jirgal Burjgin and Naran Bilik. 2015. Contemporary Mongolian population distribution, migration, cultural change, and identity. China's Minorities on the Move (2015), 53-68."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICSDA.2009.5278368"},{"key":"e_1_3_2_1_8_1","volume-title":"Imran Razzak, and Usman Naseem.","author":"Dey Krishno","year":"2024","unstructured":"Krishno Dey, Prerona Tarannum, Md Arid Hasan, Imran Razzak, and Usman Naseem. 2024. Better to ask in english: Evaluation of large language models on english, low-resource and cross-lingual settings. arXiv preprint arXiv:2410.13153 (2024)."},{"key":"e_1_3_2_1_9_1","volume-title":"How Large Language Models Enhance Low-Resource Mongolian-Chinese Machine Translation? Data Intelligence","author":"Gao Zhenjie","year":"2025","unstructured":"Zhenjie Gao, Feilong Bao, Yuan Li, Ruichen Hou, and Yibo Han. 2025. How Large Language Models Enhance Low-Resource Mongolian-Chinese Machine Translation? Data Intelligence (2025)."},{"key":"e_1_3_2_1_10_1","unstructured":"Gemma Team et al. 2025. Gemma 3 Technical Report. arXiv:2503.19786 [cs.CL] https:\/\/arxiv.org\/abs\/2503.19786"},{"key":"e_1_3_2_1_11_1","unstructured":"Aaron Grattafiori Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Alex Vaughan et al. 2024. The llama 3 herd of models. arXiv preprint arXiv:2407.21783 (2024)."},{"key":"e_1_3_2_1_12_1","volume-title":"The false promise of imitating proprietary llms. arXiv preprint arXiv:2305.15717","author":"Gudibande Arnav","year":"2023","unstructured":"Arnav Gudibande, Eric Wallace, Charlie Snell, Xinyang Geng, Hao Liu, Pieter Abbeel, Sergey Levine, and Dawn Song. 2023. The false promise of imitating proprietary llms. arXiv preprint arXiv:2305.15717 (2023)."},{"key":"e_1_3_2_1_13_1","volume-title":"Chitta Baral, and Swaroop Mishra.","author":"Gupta Himanshu","year":"2023","unstructured":"Himanshu Gupta, Kevin Scaria, Ujjwala Anantheswaran, Shreyas Verma, Mihir Parmar, Saurabh Arjun Sawant, Chitta Baral, and Swaroop Mishra. 2023. Targen: Targeted data generation with large language models. arXiv preprint arXiv:2310.17876 (2023)."},{"key":"e_1_3_2_1_14_1","volume-title":"Measuring massive multitask language understanding. arXiv preprint arXiv:2009.03300","author":"Hendrycks Dan","year":"2020","unstructured":"Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, and Jacob Steinhardt. 2020. Measuring massive multitask language understanding. arXiv preprint arXiv:2009.03300 (2020)."},{"key":"e_1_3_2_1_15_1","volume-title":"Over-tokenized transformer: Vocabulary is generally worth scaling. arXiv preprint arXiv:2501.16975","author":"Huang Hongzhi","year":"2025","unstructured":"Hongzhi Huang, Defa Zhu, Banggu Wu, Yutao Zeng, Ya Wang, Qiyang Min, and Xun Zhou. 2025. Over-tokenized transformer: Vocabulary is generally worth scaling. arXiv preprint arXiv:2501.16975 (2025)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.560"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.484"},{"key":"e_1_3_2_1_18_1","first-page":"224","article-title":"A comparative ethnolinguistic study of Mongolian and Inuit languages. TPM-Testing, Psychometrics","volume":"32","author":"Lkhagvasuren Uulensolongo","year":"2025","unstructured":"Uulensolongo Lkhagvasuren and Nurjan Khavdsyelyem. 2025. A comparative ethnolinguistic study of Mongolian and Inuit languages. TPM-Testing, Psychometrics, Methodology in Applied Psychology, Vol. 32, S8 (2025), 224-227.","journal-title":"Methodology in Applied Psychology"},{"key":"e_1_3_2_1_19_1","volume-title":"A survey on large language models with some insights on their capabilities and limitations. arXiv preprint arXiv:2501.04040","author":"Matarazzo Andrea","year":"2025","unstructured":"Andrea Matarazzo and Riccardo Torlone. 2025. A survey on large language models with some insights on their capabilities and limitations. arXiv preprint arXiv:2501.04040 (2025)."},{"key":"e_1_3_2_1_20_1","volume-title":"Overcoming data scarcity in generative language modelling for low-resource languages: A systematic review. arXiv preprint arXiv:2505.04531","author":"McGiff Josh","year":"2025","unstructured":"Josh McGiff and Nikola S Nikolov. 2025. Overcoming data scarcity in generative language modelling for low-resource languages: A systematic review. arXiv preprint arXiv:2505.04531 (2025)."},{"key":"e_1_3_2_1_21_1","volume-title":"Large language models: A survey. arXiv preprint arXiv:2402.06196","author":"Minaee Shervin","year":"2024","unstructured":"Shervin Minaee, Tomas Mikolov, Narjes Nikzad, Meysam Chenaghlu, Richard Socher, Xavier Amatriain, and Jianfeng Gao. 2024. Large language models: A survey. arXiv preprint arXiv:2402.06196 (2024)."},{"key":"e_1_3_2_1_22_1","volume-title":"Pre-training Language Model for Mongolian with Agglutinative Linguistic Knowledge Injection. In 2024 International Joint Conference on Neural Networks (IJCNN). IEEE, 1-8.","author":"Na Muhan","year":"2024","unstructured":"Muhan Na, Rui Liu, Feilong Bao, and Guanglai Gao. 2024. Pre-training Language Model for Mongolian with Agglutinative Linguistic Knowledge Injection. In 2024 International Joint Conference on Neural Networks (IJCNN). IEEE, 1-8."},{"key":"e_1_3_2_1_23_1","volume-title":"Proceedings of the 40th annual meeting of the Association for Computational Linguistics. 311-318","author":"Papineni Kishore","year":"2002","unstructured":"Kishore Papineni, Salim Roukos, Todd Ward, and Wei-Jing Zhu. 2002. Bleu: a method for automatic evaluation of machine translation. In Proceedings of the 40th annual meeting of the Association for Computational Linguistics. 311-318."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W17-4770"},{"key":"e_1_3_2_1_25_1","volume-title":"Patterns","volume":"6","author":"Qin Libo","year":"2025","unstructured":"Libo Qin, Qiguang Chen, Yuhang Zhou, Zhi Chen, Yinghui Li, Lizi Liao, Min Li, Wanxiang Che, and Philip S Yu. 2025. A survey of multilingual large language models. Patterns, Vol. 6, 1 (2025)."},{"key":"e_1_3_2_1_26_1","first-page":"4411","volume-title":"Proceedings of the International Conference on Machine Learning","volume":"2020","author":"Siddhant Aditya","year":"2020","unstructured":"Aditya Siddhant, Junjie Hu, Melvin Johnson, Orhan Firat, and Sebastian Ruder. 2020. Xtreme: A massively multilingual multi-task benchmark for evaluating cross-lingual generalization. In Proceedings of the International Conference on Machine Learning, Vol. 2020. 4411-4421."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/PIERE62470.2024.10804991"},{"key":"e_1_3_2_1_28_1","unstructured":"Tuguldur Tugstugi et al. 2020. mongolian-nlp: Mongolian NLP Resources and Tools. https:\/\/github.com\/tugstugi\/mongolian-nlp Accessed: 2025-10-08."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W18-5446"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.2514626122"},{"key":"e_1_3_2_1_31_1","volume-title":"Tokenization matters! degrading large language models through challenging their tokenization. arXiv preprint arXiv:2405.17067","author":"Wang Dixuan","year":"2024","unstructured":"Dixuan Wang, Yanda Li, Junyuan Jiang, Zepeng Ding, Ziqin Luo, Guochao Jiang, Jiaqing Liang, and Deqing Yang. 2024a. Tokenization matters! degrading large language models through challenging their tokenization. arXiv preprint arXiv:2405.17067 (2024)."},{"key":"e_1_3_2_1_32_1","unstructured":"Ke Wang Jiahui Zhu Minjie Ren Zeming Liu Shiwei Li Zongye Zhang Chenkai Zhang Xiaoyu Wu Qiqi Zhan Qingjie Liu et al. 2024b. A survey on data synthesis and augmentation for large language models. arXiv preprint arXiv:2410.12896 (2024)."},{"key":"e_1_3_2_1_33_1","unstructured":"Jason Wei Yi Tay Rishi Bommasani Colin Raffel Barret Zoph Sebastian Borgeaud Dani Yogatama Maarten Bosma Denny Zhou Donald Metzler et al. 2022a. Emergent abilities of large language models. arXiv preprint arXiv:2206.07682 (2022)."},{"key":"e_1_3_2_1_34_1","volume-title":"Denny Zhou, et al.","author":"Wei Jason","year":"2022","unstructured":"Jason Wei, Xuezhi Wang, Dale Schuurmans, Maarten Bosma, Fei Xia, Ed Chi, Quoc V Le, Denny Zhou, et al., 2022b. Chain-of-thought prompting elicits reasoning in large language models. Advances in neural information processing systems, Vol. 35 (2022), 24824-24837."},{"key":"e_1_3_2_1_35_1","volume-title":"Unigen: A unified framework for textual dataset generation using large language models. arXiv e-prints","author":"Wu Siyuan","year":"2024","unstructured":"Siyuan Wu, Yue Huang, Chujie Gao, Dongping Chen, Qihui Zhang, Yao Wan, Tianyi Zhou, Xiangliang Zhang, Jianfeng Gao, Chaowei Xiao, et al., 2024. Unigen: A unified framework for textual dataset generation using large language models. arXiv e-prints (2024), arXiv-2406."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.emnlp-main.622"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.coling-main.419"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.emnlp-main.79"},{"key":"e_1_3_2_1_39_1","unstructured":"An Yang Anfeng Li Baosong Yang Beichen Zhang Binyuan Hui Bo Zheng Bowen Yu Chang Gao Chengen Huang Chenxu Lv et al. 2025. Qwen3 technical report. arXiv preprint arXiv:2505.09388 (2025)."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","unstructured":"Shen Yingli Bao Wugedele and Zhao Xiaobing. 2022. Mongolian-Chinese Machine Translation Correction Dataset. doi:10.11922\/sciencedb.j00001.00354","DOI":"10.11922\/sciencedb.j00001.00354"},{"key":"e_1_3_2_1_41_1","volume-title":"But With Whom? MENA Values Benchmark for Evaluating Cultural Alignment and Multilingual Bias in LLMs. arXiv preprint arXiv:2510.13154","author":"Zahraei Pardis Sadat","year":"2025","unstructured":"Pardis Sadat Zahraei and Ehsaneddin Asgari. 2025. I Am Aligned, But With Whom? MENA Values Benchmark for Evaluating Cultural Alignment and Multilingual Bias in LLMs. arXiv preprint arXiv:2510.13154 (2025)."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1472"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.479"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.795"}],"event":{"name":"SIGIR '26: The 49th International ACM SIGIR Conference on Research and Development in Information Retrieval","location":"Melbourne VIC Australia","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 49th International ACM SIGIR Conference on Research and Development in Information Retrieval"],"original-title":[],"deposited":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:23:17Z","timestamp":1784136197000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805712.3808632"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"references-count":44,"alternative-id":["10.1145\/3805712.3808632","10.1145\/3805712"],"URL":"https:\/\/doi.org\/10.1145\/3805712.3808632","relation":{},"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"2026-07-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}