{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,21]],"date-time":"2026-04-21T15:53:43Z","timestamp":1776786823250,"version":"3.51.2"},"reference-count":24,"publisher":"Association for Computing Machinery (ACM)","issue":"11","license":[{"start":{"date-parts":[[2024,11,21]],"date-time":"2024-11-21T00:00:00Z","timestamp":1732147200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"crossref","award":["lzujbky-2024-jdzx15"],"award-info":[{"award-number":["lzujbky-2024-jdzx15"]}],"id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100012166","name":"National Key R&D Program of China","doi-asserted-by":"crossref","award":["2020YFC0832500"],"award-info":[{"award-number":["2020YFC0832500"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["62266037"],"award-info":[{"award-number":["62266037"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"name":"Science and Technology Plan of Qinghai Province","award":["2020-GX-164"],"award-info":[{"award-number":["2020-GX-164"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":["ACM Trans. Asian Low-Resour. Lang. Inf. Process."],"published-print":{"date-parts":[[2024,11,30]]},"abstract":"<jats:p>\n            Tibetan language processing is crucial for preserving its rich cultural heritage and reducing communication barriers between different languages. However, as a low-resource language, the development of Tibetan natural language processing has lagged behind. To address the unique and complex structural information of Tibetan, this article improves the embedding model based on fundamental Tibetan Component-and-Character-and-Word-based Embedding (TCCWE) to enhance the effectiveness of word vector representation. We incorporate position information into the training of Tibetan word vectors, developing models based on components, characters, and their integration. Furthermore, to evaluate the effectiveness of these word vectors, we propose an intrinsic evaluation set, wordsimT, based on\n            <jats:italic>k<\/jats:italic>\n            -means clustering. Experimental results demonstrate that the character-based positional vector integration model achieves a Spearman's rank correlation coefficient of 79.99% on the wordsimT benchmark, outperforming the baseline TCCWE model by 1.51%. Additionally, we validate the proposed models in downstream text classification tasks. These findings underscore the importance of incorporating positional information in Tibetan word vectors.\n          <\/jats:p>","DOI":"10.1145\/3681787","type":"journal-article","created":{"date-parts":[[2024,8,1]],"date-time":"2024-08-01T11:53:30Z","timestamp":1722513210000},"page":"1-21","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Improved Tibetan Word Vectors Models Based on Position Information Fusion"],"prefix":"10.1145","volume":"23","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0623-3556","authenticated-orcid":false,"given":"Hui","family":"Lv","sequence":"first","affiliation":[{"name":"School of Information Science and Engineering, Lanzhou University, Lanzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5775-5576","authenticated-orcid":false,"given":"Hao","family":"Lv","sequence":"additional","affiliation":[{"name":"School of Information Science and Engineering, Lanzhou University, Lanzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-0313-2184","authenticated-orcid":false,"given":"Liu","family":"Yang","sequence":"additional","affiliation":[{"name":"School of Mathematics and Statistics, Lanzhou University, Lanzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9403-7140","authenticated-orcid":false,"given":"Jun","family":"Shen","sequence":"additional","affiliation":[{"name":"School of Computing and Information Technology, University of Wollongong, Wollongong, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4511-7152","authenticated-orcid":false,"given":"La","family":"Duo","sequence":"additional","affiliation":[{"name":"The State Key Laboratory of Tibetan Intelligent Information Processing and Application, Qinghai Normal University, Xining, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7960-627X","authenticated-orcid":false,"given":"Yan","family":"Li","sequence":"additional","affiliation":[{"name":"School of Information Science and Engineering, Lanzhou University, Lanzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8054-5446","authenticated-orcid":false,"given":"Qingguo","family":"Zhou","sequence":"additional","affiliation":[{"name":"School of Information Science and Engineering, Lanzhou University, Lanzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6945-9122","authenticated-orcid":false,"given":"Binbin","family":"Yong","sequence":"additional","affiliation":[{"name":"School of Information Science and Engineering, Lanzhou University, Lanzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,11,21]]},"reference":[{"key":"e_1_3_1_2_2","unstructured":"J. Devlin M. W. Chang K. Lee and K. Toutanova. 2019. BERT: Pre-trained of deep bidirectional transformers for language understanding. arXiv:1810.04805. https:\/\/arxiv.org\/abs\/1810.04805"},{"key":"e_1_3_1_3_2","unstructured":"A. Radford K. Narasimhan T. Salimans and I. Sutskever. 2018. Improving language understanding by generative pre-training. 1--12."},{"key":"e_1_3_1_4_2","article-title":"Efficient estimation of word representations in vector space","author":"Mikolov Tomas","year":"2013","unstructured":"Tomas Mikolov, Kai Chen, Greg Corrado, and Jeffrey Dean. 2013. Efficient estimation of word representations in vector space. In Proceedings of the ICLR Workshop Papers. 1--12. https:\/\/arxiv.org\/abs\/1301.3781","journal-title":"Proceedings of the ICLR Workshop Papers"},{"key":"e_1_3_1_5_2","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/D14-1162"},{"issue":"05","key":"e_1_3_1_6_2","first-page":"44","article-title":"A tibetan word embedding representation method based on multi-primitives joint training","volume":"34","author":"Zhijie Cai","year":"2020","unstructured":"Cai Zhijie, Cai Rangzhuoma, and Sun Maosong. 2020. A tibetan word embedding representation method based on multi-primitives joint training. Journal of Chinese Information Processing 34, 05 (2020), 44\u201349 (in Chinese).","journal-title":"Journal of Chinese Information Processing"},{"key":"e_1_3_1_7_2","doi-asserted-by":"publisher","DOI":"10.23919\/TST.2017.8195342"},{"key":"e_1_3_1_8_2","first-page":"406","article-title":"Placing search in context: The concept revisited","author":"Finkelstein L.","year":"2001","unstructured":"L. Finkelstein, E. Gabrilovich, Y. Matias, E. Rivlin, Z. Solan, G. Wolfman, and E. Ruppin. 2001. Placing search in context: The concept revisited. In Proceedings of the 10th International Conference on the World Wide Web, 406\u2013414.","journal-title":"Proceedings of the 10th International Conference on the World Wide Web"},{"key":"e_1_3_1_9_2","first-page":"374","article-title":"SemEval-2012 task 4: Evaluating Chinese word similarity","author":"Jin P.","year":"2012","unstructured":"P. Jin and Y. Wu. 2012. SemEval-2012 task 4: Evaluating Chinese word similarity. In Proceedings of the 1st Joint Conference on Lexical and Computational Semantics (SEM), 374\u2013377.","journal-title":"Proceedings of the 1st Joint Conference on Lexical and Computational Semantics (SEM)"},{"issue":"07","key":"e_1_3_1_10_2","first-page":"81 \u2013 87 + 100","article-title":"Construction of Tibetan words embedding similarity and relevance evaluation set","volume":"33","author":"Zhijie Cai","year":"2019","unstructured":"Cai Zhijie, Sun Maosong, and Cai Rangzhuoma. 2019. Construction of Tibetan words embedding similarity and relevance evaluation set. Journal of Chinese Information Processing 33, 07 (2019), 81 \u2013 87 + 100 (in Chinese).","journal-title":"Journal of Chinese Information Processing"},{"key":"e_1_3_1_11_2","first-page":"873","article-title":"Improving word representations via global context and multiple word prototypes","author":"Huang Eric H.","year":"2012","unstructured":"Eric H. Huang, Richard Socher, Christopher D. Manning, and Andrew Y. Ng. 2012. Improving word representations via global context and multiple word prototypes. In Proceedings of the 50th Annual Meeting of the Association for Computational Linguistics: Long Papers - Volume 1 (ACL \u201912). Association for Computational Linguistics, 873\u2013882.","journal-title":"Proceedings of the 50th Annual Meeting of the Association for Computational Linguistics: Long Papers - Volume 1 (ACL \u201912)"},{"key":"e_1_3_1_12_2","doi-asserted-by":"publisher","DOI":"10.5555\/944919.944966"},{"key":"e_1_3_1_13_2","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3065201"},{"key":"e_1_3_1_14_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.561"},{"key":"e_1_3_1_15_2","doi-asserted-by":"publisher","DOI":"10.1145\/3511600"},{"key":"e_1_3_1_16_2","doi-asserted-by":"publisher","DOI":"10.1145\/3527663"},{"issue":"10","key":"e_1_3_1_17_2","first-page":"33-38+50","article-title":"Construction of Tibetan knowledge base of semantic similar words based on word vectors","volume":"34","author":"Long Congjun","year":"2020","unstructured":"Congjun Long, Huidan Liu, and Maoke Zhou. 2020. Construction of Tibetan knowledge base of semantic similar words based on word vectors. Journal of Chinese Information Processing 34, 10 (2020), 33-38+50 (in Chinese).","journal-title":"Journal of Chinese Information Processing"},{"key":"e_1_3_1_18_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D17-1027"},{"key":"e_1_3_1_19_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.12029"},{"key":"e_1_3_1_20_2","doi-asserted-by":"publisher","DOI":"10.23919\/TST.2017.8195345"},{"key":"e_1_3_1_21_2","doi-asserted-by":"publisher","DOI":"10.1109\/SMC53654.2022.9945074"},{"key":"e_1_3_1_22_2","doi-asserted-by":"publisher","DOI":"10.1109\/TCSS.2024.3374633"},{"key":"e_1_3_1_23_2","doi-asserted-by":"crossref","first-page":"472","DOI":"10.1007\/978-3-319-69005-6_39","article-title":"End-to-end neural text classification for Tibetan","author":"Qun N.","year":"2017","unstructured":"N. Qun, X. Li, X. Qiu, and X. Huang. 2017. End-to-end neural text classification for Tibetan. In Proceedings of Chinese Computational Linguistics and Natural Language Processing Based on Naturally Annotated Big Data: 16th China National Conference (CCL 2017) and the 5th International Symposium NLP-NABD 2017. Springer International Publishing, 472\u2013480.","journal-title":"Proceedings of Chinese Computational Linguistics and Natural Language Processing Based on Naturally Annotated Big Data: 16th China National Conference (CCL 2017) and the 5th International Symposium NLP-NABD 2017"},{"key":"e_1_3_1_24_2","unstructured":"UTibetNLP. 2023. ProSubCINO. GitHub repository. Retrieved March 15 2023 from https:\/\/github.com\/UTibetNLP\/ProSubCINO"},{"key":"e_1_3_1_25_2","unstructured":"P. Liu X. Qiu and X. Huang. 2016. Recurrent neural network for text classification with multi-task learning. arXiv:1605.05101. https:\/\/arxiv.org\/abs\/1605.05101"}],"container-title":["ACM Transactions on Asian and Low-Resource Language Information Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3681787","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3681787","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:10:03Z","timestamp":1750295403000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3681787"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,21]]},"references-count":24,"journal-issue":{"issue":"11","published-print":{"date-parts":[[2024,11,30]]}},"alternative-id":["10.1145\/3681787"],"URL":"https:\/\/doi.org\/10.1145\/3681787","relation":{},"ISSN":["2375-4699","2375-4702"],"issn-type":[{"value":"2375-4699","type":"print"},{"value":"2375-4702","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,21]]},"assertion":[{"value":"2023-05-15","order":0,"name":"received","label":"Received","group":{"name":"publication_history","label":"Publication History"}},{"value":"2024-07-16","order":2,"name":"accepted","label":"Accepted","group":{"name":"publication_history","label":"Publication History"}},{"value":"2024-11-21","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}