{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,19]],"date-time":"2026-06-19T15:00:58Z","timestamp":1781881258628,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":45,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3754571","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T06:47:18Z","timestamp":1761374838000},"page":"5394-5403","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["Enhancing the Geometric Problem-Solving Ability of Multimodal LLMs via Symbolic-Neural Integration"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-0233-2417","authenticated-orcid":false,"given":"Yicheng","family":"Pan","sequence":"first","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9211-0749","authenticated-orcid":false,"given":"Zhenrong","family":"Zhang","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-3345-605X","authenticated-orcid":false,"given":"Pengfei","family":"Hu","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2416-3720","authenticated-orcid":false,"given":"Jiefeng","family":"Ma","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2387-0389","authenticated-orcid":false,"given":"Jun","family":"Du","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2713-2535","authenticated-orcid":false,"given":"Jianshu","family":"Zhang","sequence":"additional","affiliation":[{"name":"iFLYTEK Research, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-1116-9046","authenticated-orcid":false,"given":"Quan","family":"Liu","sequence":"additional","affiliation":[{"name":"iFLYTEK Research, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5575-4940","authenticated-orcid":false,"given":"Jianqing","family":"Gao","sequence":"additional","affiliation":[{"name":"iFLYTEK Research, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-4837-8276","authenticated-orcid":false,"given":"Feng","family":"Ma","sequence":"additional","affiliation":[{"name":"iFLYTEK Research, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Large language models for mathematical reasoning: Progresses and challenges. arXiv preprint arXiv:2402.00157","author":"Ahn Janice","year":"2024","unstructured":"Janice Ahn, Rishu Verma, Renze Lou, Di Liu, Rui Zhang, and Wenpeng Yin. 2024. Large language models for mathematical reasoning: Progresses and challenges. arXiv preprint arXiv:2402.00157 (2024)."},{"key":"e_1_3_2_1_2_1","unstructured":"Shuai Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge Sibo Song Kai Dang Peng Wang Shijie Wang Jun Tang et al. 2025. Qwen2. 5-vl technical report. arXiv preprint arXiv:2502.13923 (2025)."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.44"},{"key":"e_1_3_2_1_4_1","unstructured":"Jie Cao and Jing Xiao. 2022. An Augmented Benchmark Dataset for Geometric Question Answering through Dual Parallel Text Encoding. In Proceedings of the 29th International Conference on Computational Linguistics Nicoletta Calzolari Chu-Ren Huang Hansaem Kim James Pustejovsky Leo Wanner Key-Sun Choi Pum-Mo Ryu Hsin-Hsi Chen Lucia Donatelli Heng Ji et al. (Eds.). International Committee on Computational Linguistics Gyeongju Republic of Korea 1511-1520. https:\/\/aclanthology.org\/2022.coling-1.130\/"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"crossref","unstructured":"Jiaqi Chen Tong Li Jinghui Qin Pan Lu Liang Lin Chongyu Chen and Xiaodan Liang. 2022. UniGeo: Unifying Geometry Logical Reasoning via Reformulating Mathematical Expression. In EMNLP.","DOI":"10.18653\/v1\/2022.emnlp-main.218"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"crossref","unstructured":"Jiaqi Chen Jianheng Tang Jinghui Qin Xiaodan Liang Lingbo Liu Eric P Xing and Liang Lin. 2021. GeoQA: A geometric question answering benchmark towards multimodal numerical reasoning. In Findings of ACL.","DOI":"10.18653\/v1\/2021.findings-acl.46"},{"key":"e_1_3_2_1_7_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 24185-24198","author":"Chen Zhe","year":"2024","unstructured":"Zhe Chen, Jiannan Wu, Wenhai Wang, Weijie Su, Guo Chen, Sen Xing, Muyan Zhong, Qinglong Zhang, Xizhou Zhu, Lewei Lu, et al. 2024. Internvl: Scaling up vision foundation models and aligning for generic visual-linguistic tasks. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 24185-24198."},{"key":"e_1_3_2_1_8_1","volume-title":"Gold-medalist Performance in Solving Olympiad Geometry with AlphaGeometry2. arXiv preprint arXiv:2502.03544","author":"Chervonyi Yuri","year":"2025","unstructured":"Yuri Chervonyi, Trieu H Trinh, Miroslav Ol\u0161\u00e1k, Xiaomeng Yang, Hoang Nguyen, Marcelo Menegali, Junehyuk Jung, Vikas Verma, Quoc V Le, and Thang Luong. 2025. Gold-medalist Performance in Solving Olympiad Geometry with AlphaGeometry2. arXiv preprint arXiv:2502.03544 (2025)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.5555\/56948.56949"},{"key":"e_1_3_2_1_10_1","volume-title":"R-cot: Reverse chain-of-thought problem generation for geometric reasoning in large multimodal models. arXiv preprint arXiv:2410.17885","author":"Deng Linger","year":"2024","unstructured":"Linger Deng, Yuliang Liu, Bohan Li, Dongliang Luo, Liang Wu, Chengquan Zhang, Pengyuan Lyu, Ziyang Zhang, Gang Zhang, Errui Ding, et al. 2024. R-cot: Reverse chain-of-thought problem generation for geometric reasoning in large multimodal models. arXiv preprint arXiv:2410.17885 (2024)."},{"key":"e_1_3_2_1_11_1","volume-title":"International Conference on Learning Representations.","author":"Dosovitskiy Alexey","year":"2020","unstructured":"Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, Xiaohua Zhai, Thomas Unterthiner, Mostafa Dehghani, Matthias Minderer, Georg Heigold, Sylvain Gelly, et al. 2020. An image is worth 16x16 words: Transformers for image recognition at scale. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_12_1","volume-title":"G-llava: Solving geometric problem with multi-modal large language model. arXiv preprint arXiv:2312.11370","author":"Gao Jiahui","year":"2023","unstructured":"Jiahui Gao, Renjie Pi, Jipeng Zhang, Jiacheng Ye, Wanjun Zhong, Yufei Wang, Lanqing Hong, Jianhua Han, Hang Xu, Zhenguo Li, et al. 2023. G-llava: Solving geometric problem with multi-modal large language model. arXiv preprint arXiv:2312.11370 (2023)."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICPR56361.2022.9956397"},{"key":"e_1_3_2_1_14_1","unstructured":"Aaron Hurst Adam Lerer Adam P Goucher Adam Perelman Aditya Ramesh Aidan Clark AJ Ostrow Akila Welihinda Alan Hayes Alec Radford et al. 2024. Gpt-4o system card. arXiv preprint arXiv:2410.21276 (2024)."},{"key":"e_1_3_2_1_15_1","volume-title":"A survey on large language models for code generation. arXiv preprint arXiv:2406.00515","author":"Jiang Juyong","year":"2024","unstructured":"Juyong Jiang, Fan Wang, Jiasi Shen, Sungju Kim, and Sunghun Kim. 2024. A survey on large language models for code generation. arXiv preprint arXiv:2406.00515 (2024)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2312.12241"},{"key":"e_1_3_2_1_17_1","volume-title":"Eagle: Elevating geometric reasoning through llm-empowered visual instruction tuning. arXiv preprint arXiv:2408.11397","author":"Li Zhihao","year":"2024","unstructured":"Zhihao Li, Yao Du, Yang Liu, Yan Zhang, Yufang Liu, Mengdi Zhang, and Xunliang Cai. 2024. Eagle: Elevating geometric reasoning through llm-empowered visual instruction tuning. arXiv preprint arXiv:2408.11397 (2024)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.153"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.18653\/V1\/2023.EMNLP-MAIN.440"},{"key":"e_1_3_2_1_20_1","unstructured":"Haotian Liu Chunyuan Li Yuheng Li Bo Li Yuanhan Zhang Sheng Shen and Yong Jae Lee. 2024. LLaVA-NeXT: Improved reasoning OCR and world knowledge. https:\/\/llava-vl.github.io\/blog\/2024-01-30-llava-next\/"},{"key":"e_1_3_2_1_21_1","volume-title":"International Conference on Learning Representations.","author":"Lu Pan","year":"2023","unstructured":"Pan Lu, Hritik Bansal, Tony Xia, Jiacheng Liu, Chunyuan Li, Hannaneh Hajishirzi, Hao Cheng, Kai-Wei Chang, Michel Galley, and Jianfeng Gao. 2023. Mathvista: Evaluating mathematical reasoning of foundation models in visual contexts. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"crossref","unstructured":"Pan Lu Ran Gong Shibiao Jiang Liang Qiu Siyuan Huang Xiaodan Liang and Song-Chun Zhu. 2021. Inter-GPS: Interpretable geometry problem solving with formal language and symbolic reasoning. In ACL.","DOI":"10.18653\/v1\/2021.acl-long.528"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findingsemnlp.248"},{"key":"e_1_3_2_1_24_1","volume-title":"Maths: Multimodal Transformer-Based Human-Readable Solver. In 2024 IEEE International Conference on Multimedia and Expo (ICME). IEEE, 1-6.","author":"Pan Yicheng","year":"2024","unstructured":"Yicheng Pan, Zhenrong Zhang, Jiefeng Ma, Pengfei Hu, Jun Du, Qing Wang, Jianshu Zhang, Dan Liu, and Si Wei. 2024. Maths: Multimodal Transformer-Based Human-Readable Solver. In 2024 IEEE International Conference on Multimedia and Expo (ICME). IEEE, 1-6."},{"key":"e_1_3_2_1_25_1","volume-title":"Multimath: Bridging visual and mathematical reasoning for large language models. arXiv preprint arXiv:2409.00147","author":"Peng Shuai","year":"2024","unstructured":"Shuai Peng, Di Fu, Liangcai Gao, Xiuqin Zhong, Hongguang Fu, and Zhi Tang. 2024. Multimath: Bridging visual and mathematical reasoning for large language models. arXiv preprint arXiv:2409.00147 (2024)."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-acl.850"},{"key":"e_1_3_2_1_27_1","unstructured":"Minjoon Seo Hannaneh Hajishirzi Ali Farhadi Oren Etzioni and Clint Malcolm. 2015. Solving geometry problems: Combining text and diagram interpretation. In EMNLP."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.268"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-023-06747-5"},{"key":"e_1_3_2_1_30_1","volume-title":"Denny Zhou, et al.","author":"Wei Jason","year":"2022","unstructured":"Jason Wei, Xuezhi Wang, Dale Schuurmans, Maarten Bosma, Fei Xia, Ed Chi, Quoc V Le, Denny Zhou, et al. 2022. Chain-of-thought prompting elicits reasoning in large language models. Advances in neural information processing systems 35 (2022), 24824-24837."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11280-024-01291-2"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01312"},{"key":"e_1_3_2_1_33_1","unstructured":"Renqiu Xia Mingsheng Li Hancheng Ye Wenjie Wu Hongbin Zhou Jiakang Yuan Tianshuo Peng Xinyu Cai Xiangchao Yan Bin Wang et al. 2024. GeoX: Geometric Problem Solving Through Unified Formalized Vision-Language Pretraining. arXiv preprint arXiv:2412.11863 (2024)."},{"key":"e_1_3_2_1_34_1","first-page":"6559","volume-title":"Proceedings of the Thirty-Third International Joint Conference on Artificial Intelligence, IJCAI 2024","author":"Xiao Tong","year":"2024","unstructured":"Tong Xiao, Jiayu Liu, Zhenya Huang, Jinze Wu, Jing Sha, Shijin Wang, and Enhong Chen. 2024. Learning to Solve Geometry Problems via Simulating Human Dual-Reasoning Process. In Proceedings of the Thirty-Third International Joint Conference on Artificial Intelligence, IJCAI 2024, Jeju, South Korea, August 3-9, 2024. ijcai.org, 6559-6568. https:\/\/www.ijcai.org\/proceedings\/2024\/725"},{"key":"e_1_3_2_1_35_1","unstructured":"An Yang Baosong Yang Binyuan Hui Bo Zheng Bowen Yu Chang Zhou Chengpeng Li Chengyuan Li Dayiheng Liu Fei Huang et al. 2024. Qwen2 Technical Report. arXiv preprint arXiv:2407.10671 (2024)."},{"key":"e_1_3_2_1_36_1","unstructured":"An Yang Baosong Yang Beichen Zhang Binyuan Hui Bo Zheng Bowen Yu Chengyuan Li Dayiheng Liu Fei Huang Haoran Wei et al. 2024. Qwen2. 5 technical report. arXiv preprint arXiv:2412.15115 (2024)."},{"key":"e_1_3_2_1_37_1","volume-title":"reason and verify: Geometry problem solving with parsed clauses from diagram. arXiv preprint arXiv:2407.07327","author":"Zhang Ming-Liang","year":"2024","unstructured":"Ming-Liang Zhang, Zhong-Zhi Li, Fei Yin, Liang Lin, and Cheng-Lin Liu. 2024. Fuse, reason and verify: Geometry problem solving with parsed clauses from diagram. arXiv preprint arXiv:2407.07327 (2024)."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2022\/228"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"crossref","unstructured":"Ming-Liang Zhang Fei Yin and Cheng-Lin Liu. 2023. A Multi-Modal Neural Geometric Solver with Textual Clauses Parsed from Diagram. In IJCAI.","DOI":"10.24963\/ijcai.2023\/376"},{"key":"e_1_3_2_1_40_1","volume-title":"Mavis: Mathematical visual instruction tuning. arXiv e-prints","author":"Zhang Renrui","year":"2024","unstructured":"Renrui Zhang, Xinyu Wei, Dongzhi Jiang, Yichi Zhang, Ziyu Guo, Chengzhuo Tong, Jiaming Liu, Aojun Zhou, Bin Wei, Shanghang Zhang, et al. 2024. Mavis: Mathematical visual instruction tuning. arXiv e-prints (2024), arXiv-2407."},{"key":"e_1_3_2_1_41_1","unstructured":"Xiaokai Zhang Na Zhu Yiming He Jia Zou Qike Huang Xiaoxiao Jin Yanjun Guo Chenyang Mao Yang Li Zhe Zhu et al. 2024. FormalGeo: An Extensible Formalized Framework for Olympiad Geometric Problem Solving. arXiv:2310.18021 [cs.AI] https:\/\/arxiv.org\/abs\/2310.18021"},{"key":"e_1_3_2_1_42_1","volume-title":"FGeo-HyperGNet: Geometric Problem Solving Integrating Formal Symbolic System and Hypergraph Neural Network. arXiv preprint arXiv:2402.11461","author":"Zhang Xiaokai","year":"2024","unstructured":"Xiaokai Zhang, Na Zhu, Cheng Qin, Yang Li, Zhenbing Zeng, and Tuo Leng. 2024. FGeo-HyperGNet: Geometric Problem Solving Integrating Formal Symbolic System and Hypergraph Neural Network. arXiv preprint arXiv:2402.11461 (2024)."},{"key":"e_1_3_2_1_43_1","volume-title":"Pi-GPS: Enhancing Geometry Problem Solving by Unleashing the Power of Diagrammatic Information. arXiv preprint arXiv:2503.05543","author":"Zhao Junbo","year":"2025","unstructured":"Junbo Zhao, Ting Zhang, Jiayu Sun, Mi Tian, and Hua Huang. 2025. Pi-GPS: Enhancing Geometry Problem Solving by Unleashing the Power of Diagrammatic Information. arXiv preprint arXiv:2503.05543 (2025)."},{"key":"e_1_3_2_1_44_1","unstructured":"Wayne Xin Zhao Kun Zhou Junyi Li Tianyi Tang Xiaolei Wang Yupeng Hou Yingqian Min Beichen Zhang Junjie Zhang Zican Dong et al. 2023. A survey of large language models. arXiv preprint arXiv:2303.18223 1 2 (2023)."},{"key":"e_1_3_2_1_45_1","volume-title":"Math-puma: Progressive upward multimodal alignment to enhance mathematical reasoning. arXiv preprint arXiv:2408.08640","author":"Zhuang Wenwen","year":"2024","unstructured":"Wenwen Zhuang, Xin Huang, Xiantao Zhang, and Jin Zeng. 2024. Math-puma: Progressive upward multimodal alignment to enhance mathematical reasoning. arXiv preprint arXiv:2408.08640 (2024)."}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3754571","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:11:16Z","timestamp":1765339876000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3754571"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":45,"alternative-id":["10.1145\/3746027.3754571","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3754571","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}