{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,6]],"date-time":"2026-06-06T07:35:21Z","timestamp":1780731321742,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":40,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,4,22]],"date-time":"2025-04-22T00:00:00Z","timestamp":1745280000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62201475, 61972425, 62302400, 62376227, U24A20250"],"award-info":[{"award-number":["62201475, 61972425, 62302400, 62376227, U24A20250"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Sichuan Science and Technology Program","award":["2024NSFSC1460, 2024NSFSC1436, 2023NSFSC0032"],"award-info":[{"award-number":["2024NSFSC1460, 2024NSFSC1436, 2023NSFSC0032"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,4,28]]},"DOI":"10.1145\/3696410.3714674","type":"proceedings-article","created":{"date-parts":[[2025,4,22]],"date-time":"2025-04-22T23:08:29Z","timestamp":1745363309000},"page":"2981-2991","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":10,"title":["MDEval: Evaluating and Enhancing Markdown Awareness in Large Language Models"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2123-3490","authenticated-orcid":false,"given":"Zhongpu","family":"Chen","sequence":"first","affiliation":[{"name":"Southwestern University of Finance and Economics, Chengdu, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-3031-5647","authenticated-orcid":false,"given":"Yinfeng","family":"Liu","sequence":"additional","affiliation":[{"name":"Southwestern University of Finance and Economics, Chengdu, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1378-1312","authenticated-orcid":false,"given":"Long","family":"Shi","sequence":"additional","affiliation":[{"name":"Southwestern University of Finance and Economics, Chengdu, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6865-7899","authenticated-orcid":false,"given":"Zhi-Jie","family":"Wang","sequence":"additional","affiliation":[{"name":"Chongqing University, Chongqing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0861-2100","authenticated-orcid":false,"given":"Xingyan","family":"Chen","sequence":"additional","affiliation":[{"name":"Southwestern University of Finance and Economics, Chengdu, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8454-0025","authenticated-orcid":false,"given":"Yu","family":"Zhao","sequence":"additional","affiliation":[{"name":"Southwestern University of Finance and Economics, Chengdu, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4860-9184","authenticated-orcid":false,"given":"Fuji","family":"Ren","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,4,22]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Confident AI. 2024. DeepEval. https:\/\/github.com\/confident-ai\/deepeval. Accessed: 2024-09-04."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.23919\/CISTI.2019.8760889"},{"key":"e_1_3_2_1_3_1","volume-title":"Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization. 65--72","author":"Banerjee Satanjeev","year":"2005","unstructured":"Satanjeev Banerjee and Alon Lavie. 2005. METEOR: An automatic metric for MT evaluation with improved correlation with human judgments. In Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization. 65--72."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1111\/nyas.15007"},{"key":"e_1_3_2_1_5_1","unstructured":"Linzheng Chai Shukai Liu Jian Yang Yuwei Yin Ke Jin Jiaheng Liu Tao Sun Ge Zhang Changyu Ren Hongcheng Guo et al. 2024. McEval: Massively Multilingual Code Evaluation. arXiv preprint arXiv:2406.07436 (2024)."},{"key":"e_1_3_2_1_6_1","volume-title":"Forty-first International Conference on Machine Learning.","author":"Chiang Wei-Lin","year":"2024","unstructured":"Wei-Lin Chiang, Lianmin Zheng, Ying Sheng, Anastasios Nikolas Angelopoulos, Tianle Li, Dacheng Li, Banghua Zhu, Hao Zhang, Michael Jordan, Joseph E Gonzalez, et al. 2024. Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference. In Forty-first International Conference on Machine Learning."},{"key":"e_1_3_2_1_7_1","volume-title":"Proceedings of the Workshop on Text Summarization Branches Out","author":"Chin-Yew Lin","year":"2004","unstructured":"Lin Chin-Yew. 2004. Rouge: A package for automatic evaluation of summaries. In Proceedings of the Workshop on Text Summarization Branches Out, 2004."},{"key":"e_1_3_2_1_8_1","volume-title":"Think you have solved question answering? try arc, the ai2 reasoning challenge. arXiv preprint arXiv:1803.05457","author":"Clark Peter","year":"2018","unstructured":"Peter Clark, Isaac Cowhey, Oren Etzioni, Tushar Khot, Ashish Sabharwal, Carissa Schoenick, and Oyvind Tafjord. 2018. Think you have solved question answering? try arc, the ai2 reasoning challenge. arXiv preprint arXiv:1803.05457 (2018)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.2318124121"},{"key":"e_1_3_2_1_10_1","unstructured":"Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Amy Yang Angela Fan et al. 2024. The llama 3 herd of models. arXiv preprint arXiv:2407.21783 (2024)."},{"key":"e_1_3_2_1_11_1","volume-title":"Ana Mar\u00eda Iglesias Maqueda, and Jorge Luis Morato Lara","author":"Elahi Ehsan","year":"2022","unstructured":"Ehsan Elahi, Ana Mar\u00eda Iglesias Maqueda, and Jorge Luis Morato Lara. 2022. Web Readability Challenges. In Proceedings of the Computational Methods in Systems and Software. Springer, 446--454."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00373"},{"key":"e_1_3_2_1_13_1","volume-title":"Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers). 6556--6576","author":"Ng Kiong","year":"2024","unstructured":"Jinlan Fu, See Kiong Ng, Zhengbao Jiang, and Pengfei Liu. 2024. GPTScore: Evaluate as You Desire. In Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers). 6556--6576."},{"key":"e_1_3_2_1_14_1","volume-title":"Llm-based nlg evaluation: Current status and challenges. arXiv preprint arXiv:2402.01383","author":"Gao Mingqi","year":"2024","unstructured":"Mingqi Gao, Xinyu Hu, Jie Ruan, Xiao Pu, and Xiaojun Wan. 2024. Llm-based nlg evaluation: Current status and challenges. arXiv preprint arXiv:2402.01383 (2024)."},{"key":"e_1_3_2_1_15_1","volume-title":"Designing instructional text","author":"Hartley James","unstructured":"James Hartley. 2013. Designing instructional text. Routledge."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICACS.2018.8333281"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/1498759.1498827"},{"key":"e_1_3_2_1_18_1","volume-title":"International conference on machine learning. PMLR, 957--966","author":"Kusner Matt","year":"2015","unstructured":"Matt Kusner, Yu Sun, Nicholas Kolkin, and Kilian Weinberger. 2015. From word embeddings to document distances. In International conference on machine learning. PMLR, 957--966."},{"key":"e_1_3_2_1_19_1","volume-title":"MT-Eval: A Multi-Turn Capabilities Evaluation Benchmark for Large Language Models. arXiv preprint arXiv:2401.16745","author":"Kwan Wai-Chung","year":"2024","unstructured":"Wai-Chung Kwan, Xingshan Zeng, Yuxin Jiang, Yufei Wang, Liangyou Li, Lifeng Shang, Xin Jiang, Qun Liu, and Kam-Fai Wong. 2024. MT-Eval: A Multi-Turn Capabilities Evaluation Benchmark for Large Language Models. arXiv preprint arXiv:2401.16745 (2024)."},{"key":"e_1_3_2_1_20_1","volume-title":"Datasets for large language models: A comprehensive survey. arXiv preprint arXiv:2402.18041","author":"Liu Yang","year":"2024","unstructured":"Yang Liu, Jiahuan Cao, Chongyu Liu, Kai Ding, and Lianwen Jin. 2024. Datasets for large language models: A comprehensive survey. arXiv preprint arXiv:2402.18041 (2024)."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.153"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.557"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00681"},{"key":"e_1_3_2_1_24_1","volume-title":"scannable, and objective: How to write for the Web. Useit. com","author":"Morkes John","year":"1997","unstructured":"John Morkes and Jakob Nielsen. 1997. Concise, scannable, and objective: How to write for the Web. Useit. com, Vol. 51, 1 (1997), 1--17."},{"key":"e_1_3_2_1_25_1","volume-title":"CLUES: Few-Shot Learning Evaluation in Natural Language Understanding. In NeurIPS","author":"Mukherjee Subhabrata","year":"2021","unstructured":"Subhabrata (Subho) Mukherjee, Xiaodong Liu, Guoqing Zheng, Saghar Hosseini, Hao Cheng, Greg Yang, Chris Meek, Ahmed Awadallah, and Jianfeng Gao. 2021. CLUES: Few-Shot Learning Evaluation in Natural Language Understanding. In NeurIPS 2021."},{"key":"e_1_3_2_1_26_1","volume-title":"Peer-review-in-LLMs: Automatic Evaluation Method for LLMs in Open-environment. arXiv preprint arXiv:2402.01830","author":"Ning Kun-Peng","year":"2024","unstructured":"Kun-Peng Ning, Shuo Yang, Yu-Yang Liu, Jia-Yu Yao, Zhen-Hui Liu, Yu Wang, Ming Pang, and Li Yuan. 2024. Peer-review-in-LLMs: Automatic Evaluation Method for LLMs in Open-environment. arXiv preprint arXiv:2402.01830 (2024)."},{"key":"e_1_3_2_1_27_1","volume-title":"Proceedings of the 40th annual meeting of the Association for Computational Linguistics. 311--318","author":"Papineni Kishore","year":"2002","unstructured":"Kishore Papineni, Salim Roukos, Todd Ward, and Wei-Jing Zhu. 2002. Bleu: a method for automatic evaluation of machine translation. In Proceedings of the 40th annual meeting of the Association for Computational Linguistics. 311--318."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.704"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3543507.3583506"},{"key":"e_1_3_2_1_30_1","unstructured":"Hugo Touvron Louis Martin Kevin Stone Peter Albert Amjad Almahairi Yasmine Babaei Nikolay Bashlykov Soumya Batra Prajjwal Bhargava Shruti Bhosale et al. 2023. Llama 2: Open foundation and fine-tuned chat models. arXiv preprint arXiv:2307.09288 (2023)."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W18-5446"},{"key":"e_1_3_2_1_32_1","unstructured":"Wikipedia. [n. d.]. Elo rating system. https:\/\/en.wikipedia.org\/wiki\/Elo_rating_system. Accessed: 2024-09--21."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/JAS.2023.123618"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3589334.3645363"},{"key":"e_1_3_2_1_35_1","volume-title":"Harnessing the power of llms in practice: A survey on chatgpt and beyond. ACM Transactions on Knowledge Discovery from Data","author":"Yang Jingfeng","year":"2024","unstructured":"Jingfeng Yang, Hongye Jin, Ruixiang Tang, Xiaotian Han, Qizhang Feng, Haoming Jiang, Shaochen Zhong, Bing Yin, and Xia Hu. 2024. Harnessing the power of llms in practice: A survey on chatgpt and beyond. ACM Transactions on Knowledge Discovery from Data, Vol. 18, 6 (2024), 1--32."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3589334.3645676"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.411"},{"key":"e_1_3_2_1_38_1","volume-title":"BERTScore: Evaluating Text Generation with BERT. In 8th International Conference on Learning Representations, ICLR","author":"Zhang Tianyi","year":"2020","unstructured":"Tianyi Zhang, Varsha Kishore, Felix Wu, Kilian Q. Weinberger, and Yoav Artzi. 2020. BERTScore: Evaluating Text Generation with BERT. In 8th International Conference on Learning Representations, ICLR 2020."},{"key":"e_1_3_2_1_39_1","first-page":"46595","article-title":"Judging LLM-as-a-judge with mt-bench and chatbot arena","volume":"36","author":"Zheng Lianmin","year":"2023","unstructured":"Lianmin Zheng, Wei-Lin Chiang, Ying Sheng, Siyuan Zhuang, Zhanghao Wu, Yonghao Zhuang, Zi Lin, Zhuohan Li, Dacheng Li, Eric Xing, et al. 2023. Judging LLM-as-a-judge with mt-bench and chatbot arena. Advances in Neural Information Processing Systems, Vol. 36 (2023), 46595--46623.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-demos.38"}],"event":{"name":"WWW '25: The ACM Web Conference 2025","location":"Sydney NSW Australia","acronym":"WWW '25","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web"]},"container-title":["Proceedings of the ACM on Web Conference 2025"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3696410.3714674","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3696410.3714674","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:18:56Z","timestamp":1750295936000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3696410.3714674"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,4,22]]},"references-count":40,"alternative-id":["10.1145\/3696410.3714674","10.1145\/3696410"],"URL":"https:\/\/doi.org\/10.1145\/3696410.3714674","relation":{},"subject":[],"published":{"date-parts":[[2025,4,22]]},"assertion":[{"value":"2025-04-22","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}