{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,19]],"date-time":"2026-05-19T22:06:36Z","timestamp":1779228396246,"version":"3.51.4"},"reference-count":31,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100013088","name":"Qinglan Project of Jiangsu Province of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100013088","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62272201"],"award-info":[{"award-number":["62272201"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Computer Speech &amp; Language"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.csl.2026.101957","type":"journal-article","created":{"date-parts":[[2026,2,14]],"date-time":"2026-02-14T00:18:49Z","timestamp":1771028329000},"page":"101957","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["An exhaustive evaluation method for open-domain LLM dialogue by constructing recursive CoT"],"prefix":"10.1016","volume":"100","author":[{"given":"Shengjie","family":"Zhao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9481-9599","authenticated-orcid":false,"given":"Zhenping","family":"Xie","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.csl.2026.101957_bib0001","unstructured":"Adiwardana, D., Luong, M.T., So, D.R., Hall, J., Fiedel, N., Thoppilan, R., Yang, Z., Kulshreshtha, A., Nemade, G., & Lu, Y. (2020). Towards a human-like open-domain chatbot. arXiv preprint arXiv:2001.09977."},{"key":"10.1016\/j.csl.2026.101957_bib0002","series-title":"Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization","article-title":"METEOR: an automatic metric for MT evaluation with improved correlation with human judgments","author":"Banerjee","year":"2005"},{"key":"10.1016\/j.csl.2026.101957_bib0003","series-title":"Findings of the Association for Computational Linguistics: EMNLP 2023","article-title":"MEEP: is this engaging? prompting large language models for dialogue evaluation in multilingual settings","author":"Ferron","year":"2023"},{"key":"10.1016\/j.csl.2026.101957_bib0004","unstructured":"Fu, J., Ng, S.K., Jiang, Z., & Liu, P. (2023). Gptscore: evaluate as you desire. arXiv preprint arXiv:2302.04166."},{"key":"10.1016\/j.csl.2026.101957_bib0005","doi-asserted-by":"crossref","unstructured":"Ghazarian, S., Wei, J.T.Z., Galstyan, A., & Peng, N. (2019). Better automatic evaluation of open-domain dialogue systems with contextualized embeddings. arXiv preprint arXiv:1904.10635.","DOI":"10.18653\/v1\/W19-2310"},{"key":"10.1016\/j.csl.2026.101957_bib0006","doi-asserted-by":"crossref","unstructured":"Huang, L., Ye, Z., Qin, J., Lin, L., & Liang, X. (2020). GRADE: automatic graph-enhanced coherence metric for evaluating open-domain dialogue systems. arXiv preprint arXiv:2010.03994.","DOI":"10.18653\/v1\/2020.emnlp-main.742"},{"key":"10.1016\/j.csl.2026.101957_bib0007","series-title":"Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing","article-title":"IM^ 2: an interpretable and multi-category integrated metric framework for automatic dialogue evaluation","author":"Jiang","year":"2022"},{"issue":"1","key":"10.1016\/j.csl.2026.101957_bib0008","first-page":"1","article-title":"Pone: a novel automatic evaluation metric for open-domain generative dialogue systems","volume":"39","author":"Lan","year":"2020","journal-title":"ACM Transact. Informat. Syst. (TOIS)"},{"key":"10.1016\/j.csl.2026.101957_bib0009","doi-asserted-by":"crossref","unstructured":"Li, Z., Zhang, J., Fei, Z., Feng, Y., & Zhou, J. (2021). Conversations are not flat: modeling the dynamic information flow across dialogue utterances. arXiv preprint arXiv:2106.02227.","DOI":"10.18653\/v1\/2021.acl-long.11"},{"key":"10.1016\/j.csl.2026.101957_bib0010","unstructured":"Lin, C.Y. (2004). Rouge: a package for automatic evaluation of summaries. Text summarization branches out."},{"key":"10.1016\/j.csl.2026.101957_bib0011","doi-asserted-by":"crossref","unstructured":"Lin, Y.T., & Chen, Y.N. (2023). Llm-eval: unified multi-dimensional automatic evaluation for open-domain conversations with large language models. arXiv preprint arXiv:2305.13711.","DOI":"10.18653\/v1\/2023.nlp4convai-1.5"},{"key":"10.1016\/j.csl.2026.101957_bib0012","doi-asserted-by":"crossref","unstructured":"Liu, C.W., Lowe, R., Serban, I.V., Noseworthy, M., Charlin, L., & Pineau, J. (2016). How not to evaluate your dialogue system: an empirical study of unsupervised evaluation metrics for dialogue response generation. arXiv preprint arXiv:1603.08023.","DOI":"10.18653\/v1\/D16-1230"},{"key":"10.1016\/j.csl.2026.101957_bib0013","doi-asserted-by":"crossref","unstructured":"Liu, Y., Iter, D., Xu, Y., Wang, S., Xu, R., & Zhu, C. (2023). G-eval: nlg evaluation using gpt-4 with better human alignment. arXiv preprint arXiv:2303.16634.","DOI":"10.18653\/v1\/2023.emnlp-main.153"},{"key":"10.1016\/j.csl.2026.101957_bib0014","doi-asserted-by":"crossref","unstructured":"Lowe, R., Noseworthy, M., Serban, I.V., Angelard-Gontier, N., Bengio, Y., & Pineau, J. (2017). Towards an automatic turing test: learning to evaluate dialogue responses. arXiv preprint arXiv:1708.07149.","DOI":"10.18653\/v1\/P17-1103"},{"key":"10.1016\/j.csl.2026.101957_bib0015","doi-asserted-by":"crossref","unstructured":"Mehri, S., & Eskenazi, M. (2020a). Unsupervised evaluation of interactive dialog with DialoGPT. arXiv preprint arXiv:2006.12719.","DOI":"10.18653\/v1\/2020.sigdial-1.28"},{"key":"10.1016\/j.csl.2026.101957_bib0016","doi-asserted-by":"crossref","unstructured":"Mehri, S., & Eskenazi, M. (2020b). USR: an unsupervised and reference free evaluation metric for dialog generation. arXiv preprint arXiv:2005.00456.","DOI":"10.18653\/v1\/2020.acl-main.64"},{"key":"10.1016\/j.csl.2026.101957_bib0017","unstructured":"Mendon\u00e7a, J., Pereira, P., Moniz, H., Carvalho, J.P., Lavie, A., & Trancoso, I. (2023). Simple LLM prompting is state-of-the-art for robust and multilingual dialogue evaluation. arXiv preprint arXiv:2308.16797."},{"key":"10.1016\/j.csl.2026.101957_bib0018","doi-asserted-by":"crossref","unstructured":"Pang, B., Nijkamp, E., Han, W., Zhou, L., Liu, Y., & Tu, K. (2020). Towards holistic and automatic evaluation of open-domain dialogue generation.","DOI":"10.18653\/v1\/2020.acl-main.333"},{"key":"10.1016\/j.csl.2026.101957_bib0019","series-title":"Proceedings of the 40th annual meeting of the Association for Computational Linguistics","article-title":"Bleu: a method for automatic evaluation of machine translation","author":"Papineni","year":"2002"},{"key":"10.1016\/j.csl.2026.101957_bib0020","doi-asserted-by":"crossref","unstructured":"Phy, V., Zhao, Y., & Aizawa, A. (2020). Deconstruct to reconstruct a configurable evaluation metric for open-domain dialogue systems. arXiv preprint arXiv:2011.00483.","DOI":"10.18653\/v1\/2020.coling-main.368"},{"key":"10.1016\/j.csl.2026.101957_bib0021","unstructured":"Pl\u00e1tek, O., Hude\u010dek, V., Schmidtov\u00e1, P., Lango, M., & Du\u0161ek, O. (2023). Three Ways of Using Large Language Models to Evaluate Chat. arXiv preprint arXiv:2308.06502."},{"key":"10.1016\/j.csl.2026.101957_bib0022","unstructured":"Rodr\u00edguez-Cantelar, M., Zhang, C., Tang, C., Shi, K., Ghazarian, S., Sedoc, J., D'Haro, L.F., & Rudnicky, A. (2023). Overview of robust and multilingual automatic evaluation metrics for open-domain dialogue systems at dstc 11 track 4. arXiv preprint arXiv:2306.12794."},{"key":"10.1016\/j.csl.2026.101957_bib0023","doi-asserted-by":"crossref","unstructured":"See, A., Roller, S., Kiela, D., & Weston, J. (2019). What makes a good conversation? how controllable attributes affect human judgments. arXiv preprint arXiv:1902.08654.","DOI":"10.18653\/v1\/N19-1170"},{"key":"10.1016\/j.csl.2026.101957_bib0024","series-title":"Proceedings of the AAAI conference on artificial intelligence","article-title":"Ruber: an unsupervised method for automatic evaluation of open-domain dialog systems","author":"Tao","year":"2018"},{"key":"10.1016\/j.csl.2026.101957_bib0025","first-page":"24824","article-title":"Chain-of-thought prompting elicits reasoning in large language models","volume":"35","author":"Wei","year":"2022","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"10.1016\/j.csl.2026.101957_bib0026","doi-asserted-by":"crossref","unstructured":"Yeh, Y.T., Eskenazi, M., & Mehri, S. (2021). A comprehensive assessment of dialog evaluation metrics. arXiv preprint arXiv:2106.03706.","DOI":"10.18653\/v1\/2021.eancs-1.3"},{"key":"10.1016\/j.csl.2026.101957_bib0027","doi-asserted-by":"crossref","unstructured":"Zhang, C., Chen, Y., D'Haro, L.F., Zhang, Y., Friedrichs, T., Lee, G., & Li, H. (2021). DynaEval: unifying turn and dialogue level evaluation. arXiv preprint arXiv:2106.01112.","DOI":"10.18653\/v1\/2021.acl-long.441"},{"key":"10.1016\/j.csl.2026.101957_bib0028","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","article-title":"A comprehensive analysis of the effectiveness of large language models as automatic dialogue evaluators","author":"Zhang","year":"2024"},{"key":"10.1016\/j.csl.2026.101957_bib0029","doi-asserted-by":"crossref","unstructured":"Zhang, C., D\u2019Haro, L.F., Banchs, R.E., Friedrichs, T., & Li, H. (2021). Deep am-fm: toolkit for automatic dialogue evaluation. Conversational Dialogue Systems For the Next Decade, 53\u201369.","DOI":"10.1007\/978-981-15-8395-7_5"},{"key":"10.1016\/j.csl.2026.101957_bib0030","unstructured":"Zhang, P., Hu, X., Yu, K., Wang, J., Han, S., Liu, C., & Yuan, C. (2022). MME-CRS: multi-metric evaluation based on correlation re-scaling for evaluating open-domain dialogue. arXiv preprint arXiv:2206.09403."},{"key":"10.1016\/j.csl.2026.101957_bib0031","doi-asserted-by":"crossref","unstructured":"Zhang, S. (2018). Personalizing dialogue agents: i have a dog, do you have pets too. arXiv preprint arXiv:1801.07243.","DOI":"10.18653\/v1\/P18-1205"}],"container-title":["Computer Speech &amp; Language"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0885230826000203?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0885230826000203?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,5,19]],"date-time":"2026-05-19T21:14:18Z","timestamp":1779225258000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0885230826000203"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":31,"alternative-id":["S0885230826000203"],"URL":"https:\/\/doi.org\/10.1016\/j.csl.2026.101957","relation":{},"ISSN":["0885-2308"],"issn-type":[{"value":"0885-2308","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"An exhaustive evaluation method for open-domain LLM dialogue by constructing recursive CoT","name":"articletitle","label":"Article Title"},{"value":"Computer Speech & Language","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.csl.2026.101957","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Published by Elsevier Ltd.","name":"copyright","label":"Copyright"}],"article-number":"101957"}}