{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,10]],"date-time":"2026-05-10T01:12:34Z","timestamp":1778375554321,"version":"3.51.4"},"publisher-location":"Singapore","reference-count":38,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819527243","type":"print"},{"value":"9789819527250","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,11,1]],"date-time":"2025-11-01T00:00:00Z","timestamp":1761955200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,11,1]],"date-time":"2025-11-01T00:00:00Z","timestamp":1761955200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-981-95-2725-0_12","type":"book-chapter","created":{"date-parts":[[2025,10,31]],"date-time":"2025-10-31T05:19:49Z","timestamp":1761887989000},"page":"177-193","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["RankLLM: A Multi-Criteria Decision-Making Method for\u00a0LLM Performance Evaluation in\u00a0Sentiment Analysis"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3681-7907","authenticated-orcid":false,"given":"Huzhi","family":"Xue","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9278-6685","authenticated-orcid":false,"given":"Butian","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3357-4006","authenticated-orcid":false,"given":"Haihua","family":"Xie","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5358-3893","authenticated-orcid":false,"given":"Zeyu","family":"Sun","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,11,1]]},"reference":[{"issue":"1","key":"12_CR1","first-page":"31","volume":"1","author":"M Aruldoss","year":"2013","unstructured":"Aruldoss, M., Lakshmi, T.M., Venkatesan, V.P.: A survey on multi criteria decision making methods and its applications. Am. J. Inf. Syst. 1(1), 31\u201343 (2013)","journal-title":"Am. J. Inf. Syst."},{"key":"12_CR2","doi-asserted-by":"publisher","DOI":"10.1016\/j.dajour.2021.100021","volume":"2","author":"S Chakraborty","year":"2022","unstructured":"Chakraborty, S.: TOPSIS and modified TOPSIS: a comparative analysis. Decis. Anal. J. 2, 100021 (2022)","journal-title":"Decis. Anal. J."},{"issue":"3","key":"12_CR3","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3641289","volume":"15","author":"Y Chang","year":"2024","unstructured":"Chang, Y., et al.: A survey on evaluation of large language models. ACM Trans. Intell. Syst. Technol. 15(3), 1\u201345 (2024)","journal-title":"ACM Trans. Intell. Syst. Technol."},{"key":"12_CR4","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2020.114186","volume":"168","author":"P Chen","year":"2021","unstructured":"Chen, P.: Effects of the entropy weight on TOPSIS. Expert Syst. Appl. 168, 114186 (2021)","journal-title":"Expert Syst. Appl."},{"key":"12_CR5","unstructured":"Chu, Z., et al.: Timebench: a comprehensive evaluation of temporal reasoning abilities in large language models. arXiv preprint arXiv:2311.17667 (2023)"},{"key":"12_CR6","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2022.119045","volume":"214","author":"S Corrente","year":"2023","unstructured":"Corrente, S., Tasiou, M.: A robust TOPSIS method for decision making problems with hierarchical and non-monotonic criteria. Expert Syst. Appl. 214, 119045 (2023)","journal-title":"Expert Syst. Appl."},{"key":"12_CR7","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2022.119292","volume":"213","author":"S Du","year":"2023","unstructured":"Du, S.: Hybrid kano-dematel-topsis model based benefit distribution of multiple logistics service providers considering consumer service evaluation of segmented task. Expert Syst. Appl. 213, 119292 (2023)","journal-title":"Expert Syst. Appl."},{"key":"12_CR8","unstructured":"Gani, H., Bhat, S.F., Naseer, M., Khan, S., Wonka, P.: LLM blueprint: enabling text-to-image generation with complex and detailed prompts. arXiv preprint arXiv:2310.10640 (2023)"},{"key":"12_CR9","doi-asserted-by":"crossref","unstructured":"Gao, M., Hu, X., Ruan, J., Pu, X., Wan, X.: LLM-based NLG evaluation: current status and challenges (2024)","DOI":"10.1162\/coli_a_00561"},{"key":"12_CR10","doi-asserted-by":"crossref","unstructured":"Hu, X., et al.: Are LLM-based evaluators confusing NLG quality criteria? arXiv preprint arXiv:2402.12055 (2024)","DOI":"10.18653\/v1\/2024.acl-long.516"},{"key":"12_CR11","doi-asserted-by":"crossref","unstructured":"Huang, H., Qu, Y., Liu, J., Yang, M., Zhao, T.: An empirical study of LLM-as-a-judge for LLM evaluation: fine-tuned judge models are task-specific classifiers. arXiv preprint arXiv:2403.02839 (2024)","DOI":"10.18653\/v1\/2025.findings-acl.306"},{"key":"12_CR12","doi-asserted-by":"crossref","unstructured":"Hwang, C.L., Yoon, K., Hwang, C.L., Yoon, K.: Methods for multiple attribute decision making. Multiple attribute decision making: methods and applications a state-of-the-art survey, pp. 58\u2013191 (1981)","DOI":"10.1007\/978-3-642-48318-9_3"},{"key":"12_CR13","doi-asserted-by":"publisher","DOI":"10.1016\/j.techfore.2022.121524","volume":"177","author":"M Irfan","year":"2022","unstructured":"Irfan, M., Elavarasan, R.M., Ahmad, M., Mohsin, M., Dagar, V., Hao, Y.: Prioritizing and overcoming biomass energy barriers: application of AHP and G-TOPSIS approaches. Technol. Forecast. Soc. Chang. 177, 121524 (2022)","journal-title":"Technol. Forecast. Soc. Chang."},{"key":"12_CR14","doi-asserted-by":"crossref","unstructured":"Khurana, A., Subramonyam, H., Chilana, P.K.: Why and when LLM-based assistants can go wrong: investigating the effectiveness of prompt-based interactions for software help-seeking. In: Proceedings of the 29th International Conference on Intelligent User Interfaces, pp. 288\u2013303 (2024)","DOI":"10.1145\/3640543.3645200"},{"key":"12_CR15","doi-asserted-by":"crossref","unstructured":"Li, M., et al.: Newsbench: a systematic evaluation framework for assessing editorial capabilities of large language models in Chinese journalism. In: Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 9993\u201310014 (2024)","DOI":"10.18653\/v1\/2024.acl-long.538"},{"key":"12_CR16","doi-asserted-by":"publisher","first-page":"564","DOI":"10.1016\/j.renene.2021.11.112","volume":"184","author":"Z Li","year":"2022","unstructured":"Li, Z., Luo, Z., Wang, Y., Fan, G., Zhang, J.: Suitability evaluation system for the shallow geothermal energy implementation in region by entropy weight method and TOPSIS method. Renewable Energy 184, 564\u2013576 (2022)","journal-title":"Renewable Energy"},{"key":"12_CR17","unstructured":"Li, Z., Shi, Y., Liu, Z., Yang, F., Liu, N., Du, M.: Quantifying multilingual performance of large language models across languages (2024)"},{"key":"12_CR18","unstructured":"Lin, C.Y.: Rouge: a package for automatic evaluation of summaries. In: Text Summarization Branches Out, pp. 74\u201381 (2004)"},{"key":"12_CR19","unstructured":"Liu, Y., et al.: Trustworthy LLMs: a survey and guideline for evaluating large language models\u2019 alignment. arXiv preprint arXiv:2308.05374 (2023)"},{"key":"12_CR20","doi-asserted-by":"crossref","unstructured":"Miao, M., Meng, F., Liu, Y., Zhou, X.H., Zhou, J.: Prevent the language model from being overconfident in neural machine translation. arXiv preprint arXiv:2105.11098 (2021)","DOI":"10.18653\/v1\/2021.acl-long.268"},{"key":"12_CR21","doi-asserted-by":"crossref","unstructured":"Mizrahi, M., Kaplan, G., Malkin, D., Dror, R., Shahaf, D., Stanovsky, G.: State of what art? A call for multi-prompt LLM evaluation. arXiv preprint arXiv:2401.00595 (2023)","DOI":"10.1162\/tacl_a_00681"},{"key":"12_CR22","doi-asserted-by":"publisher","first-page":"933","DOI":"10.1162\/tacl_a_00681","volume":"12","author":"M Mizrahi","year":"2024","unstructured":"Mizrahi, M., Kaplan, G., Malkin, D., Dror, R., Shahaf, D., Stanovsky, G.: State of what art? A call for multi-prompt LLM evaluation. Trans. Assoc. Comput. Linguist. 12, 933\u2013949 (2024)","journal-title":"Trans. Assoc. Comput. Linguist."},{"key":"12_CR23","doi-asserted-by":"crossref","unstructured":"Pan, Q., et al.: Human-centered design recommendations for LLM-as-a-judge. arXiv preprint arXiv:2407.03479 (2024)","DOI":"10.18653\/v1\/2024.hucllm-1.2"},{"key":"12_CR24","doi-asserted-by":"crossref","unstructured":"Papineni, K., Roukos, S., Ward, T., Zhu, W.J.: Bleu: a method for automatic evaluation of machine translation. In: Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics, pp. 311\u2013318 (2002)","DOI":"10.3115\/1073083.1073135"},{"key":"12_CR25","doi-asserted-by":"crossref","unstructured":"Shankar, S., Zamfirescu-Pereira, J., Hartmann, B., Parameswaran, A.G., Arawjo, I.: Who validates the validators? Aligning LLM-assisted evaluation of LLM outputs with human preferences. arXiv preprint arXiv:2404.12272 (2024)","DOI":"10.1145\/3654777.3676450"},{"key":"12_CR26","doi-asserted-by":"publisher","DOI":"10.1016\/j.aei.2025.103570","volume":"67","author":"W Song","year":"2025","unstructured":"Song, W., Xue, H., Rong, W.: An integrated method for resilient-sustainable supplier selection based on action-oriented practices. Adv. Eng. Inform. 67, 103570 (2025)","journal-title":"Adv. Eng. Inform."},{"key":"12_CR27","doi-asserted-by":"publisher","first-page":"1560","DOI":"10.1016\/j.egyr.2021.03.007","volume":"7","author":"F Sun","year":"2021","unstructured":"Sun, F., Yu, J.: Improved energy performance evaluating and ranking approach for office buildings using simple-normalization, entropy-based TOPSIS and K-means method. Energy Rep. 7, 1560\u20131570 (2021)","journal-title":"Energy Rep."},{"key":"12_CR28","doi-asserted-by":"crossref","unstructured":"Sun, S., Zhuang, S., Wang, S., Zuccon, G.: An investigation of prompt variations for zero-shot LLM-based rankers. arXiv preprint arXiv:2406.14117 (2024)","DOI":"10.1007\/978-3-031-88711-6_12"},{"key":"12_CR29","unstructured":"Tian, X., et al.: Examining LLM prompting strategies for automatic evaluation of learner-created computational artifacts (2024)"},{"key":"12_CR30","doi-asserted-by":"crossref","unstructured":"Wang, L., et al.: Prompt engineering in consistency and reliability with the evidence-based guideline for LLMs. NPJ Digit. Med. 7(1), 41 (2024)","DOI":"10.1038\/s41746-024-01029-4"},{"key":"12_CR31","doi-asserted-by":"crossref","unstructured":"Wilkins, G., Keshav, S., Mortier, R.: Offline energy-optimal LLM serving: workload-based energy models for LLM inference on heterogeneous systems. arXiv preprint arXiv:2407.04014 (2024)","DOI":"10.1145\/3727200.3727217"},{"key":"12_CR32","doi-asserted-by":"crossref","unstructured":"Yang, D., Chen, F., Fang, H.: Behavior alignment: a new perspective of evaluating LLM-based conversational recommendation systems. In: Proceedings of the 47th International ACM SIGIR Conference on Research and Development in Information Retrieval, pp. 2286\u20132290 (2024)","DOI":"10.1145\/3626772.3657924"},{"key":"12_CR33","unstructured":"Zhang, T., Kishore, V., Wu, F., Weinberger, K.Q., Artzi, Y.: Bertscore: evaluating text generation with BERT. arXiv preprint arXiv:1904.09675 (2019)"},{"key":"12_CR34","doi-asserted-by":"crossref","unstructured":"Zhang, W., Deng, Y., Liu, B., Pan, S.J., Bing, L.: Sentiment analysis in the era of large language models: a reality check. arXiv preprint arXiv:2305.15005 (2023)","DOI":"10.18653\/v1\/2024.findings-naacl.246"},{"key":"12_CR35","doi-asserted-by":"crossref","unstructured":"Zhang, Y., et al.: Llmeval: a preliminary study on how to evaluate large language models. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a038, pp. 19615\u201319622 (2024)","DOI":"10.1609\/aaai.v38i17.29934"},{"key":"12_CR36","doi-asserted-by":"crossref","unstructured":"Zhao, W., Peyrard, M., Liu, F., Gao, Y., Meyer, C.M., Eger, S.: Moverscore: text generation evaluating with contextualized embeddings and earth mover distance. arXiv preprint arXiv:1909.02622 (2019)","DOI":"10.18653\/v1\/D19-1053"},{"key":"12_CR37","unstructured":"Zheng, L., et al.: Judging LLM-as-a-judge with MT-bench and chatbot arena. In: Advances in Neural Information Processing Systems, vol. 36 (2024)"},{"key":"12_CR38","doi-asserted-by":"crossref","unstructured":"Zhou, K., Jurafsky, D., Hashimoto, T.: Navigating the grey area: expressions of overconfidence and uncertainty in language models. arxiv abs\/2302.13439 (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.335"}],"container-title":["Lecture Notes in Computer Science","Chinese Computational Linguistics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-95-2725-0_12","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,31]],"date-time":"2025-10-31T05:20:07Z","timestamp":1761888007000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-95-2725-0_12"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,1]]},"ISBN":["9789819527243","9789819527250"],"references-count":38,"URL":"https:\/\/doi.org\/10.1007\/978-981-95-2725-0_12","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,11,1]]},"assertion":[{"value":"1 November 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"CCL","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China National Conference on Chinese Computational Linguistics","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Jinan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"11 August 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"14 August 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"24","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"cncl2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/link.springer.com\/conference\/cncl","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}