{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,23]],"date-time":"2026-06-23T00:48:18Z","timestamp":1782175698925,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":80,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755703","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T06:56:44Z","timestamp":1761375404000},"page":"11071-11080","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["Embodied-R: Collaborative Framework for Activating Embodied Spatial Reasoning in Foundation Models via Reinforcement Learning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9653-3316","authenticated-orcid":false,"given":"Baining","family":"Zhao","sequence":"first","affiliation":[{"name":"Shenzhen International Graduate School, Tsinghua University, Shenzhen, China and Pengcheng Laboratory, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-2297-1830","authenticated-orcid":false,"given":"Ziyou","family":"Wang","sequence":"additional","affiliation":[{"name":"Northeastern University, Qinghuangdao, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-2599-9929","authenticated-orcid":false,"given":"Jianjie","family":"Fang","sequence":"additional","affiliation":[{"name":"Northeastern University, Qinhuangdao, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7561-5646","authenticated-orcid":false,"given":"Chen","family":"Gao","sequence":"additional","affiliation":[{"name":"BNRist, Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5830-0685","authenticated-orcid":false,"given":"Fanhang","family":"Man","sequence":"additional","affiliation":[{"name":"Shenzhen International Graduate School, Tsinghua University, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7833-1876","authenticated-orcid":false,"given":"Jinqiang","family":"Cui","sequence":"additional","affiliation":[{"name":"Pengcheng Laboratory, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0351-2939","authenticated-orcid":false,"given":"Xin","family":"Wang","sequence":"additional","affiliation":[{"name":"Department of Computer Science and Technology, Tsinghua University, Beijing, China and BNRist, Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8271-5023","authenticated-orcid":false,"given":"Xinlei","family":"Chen","sequence":"additional","affiliation":[{"name":"Shenzhen International Graduate School, Tsinghua University, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5617-1659","authenticated-orcid":false,"given":"Yong","family":"Li","sequence":"additional","affiliation":[{"name":"Department of Electronic Engineering, Tsinghua University, Beijing, China and BNRist, Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2236-9290","authenticated-orcid":false,"given":"Wenwu","family":"Zhu","sequence":"additional","affiliation":[{"name":"Department of Computer Science and Technology, Tsinghua University, Beijing, China and BNRist, Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"Nguyen Bach, Amit Bahree, Arash Bakhtiari, Jianmin Bao, Harkirat Behl, et al.","author":"Abdin Marah","year":"2024","unstructured":"Marah Abdin, Jyoti Aneja, Hany Awadalla, Ahmed Awadallah, Ammar Ahmad Awan, Nguyen Bach, Amit Bahree, Arash Bakhtiari, Jianmin Bao, Harkirat Behl, et al., 2024. Phi-3 technical report: A highly capable language model locally on your phone. arXiv preprint arXiv:2404.14219 (2024)."},{"key":"e_1_3_2_2_2_1","volume-title":"Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al.","author":"Achiam Josh","year":"2023","unstructured":"Josh Achiam, Steven Adler, Sandhini Agarwal, Lama Ahmad, Ilge Akkaya, Florencia Leoni Aleman, Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al., 2023. Gpt-4 technical report. arXiv preprint arXiv:2303.08774 (2023)."},{"key":"e_1_3_2_2_3_1","volume-title":"Keerthana Gopalakrishnan, Karol Hausman, Brian Ichter, Alex Irpan, Nikhil Joshi, Ryan Julian, et al.","author":"Ahn Michael","year":"2024","unstructured":"Michael Ahn, Debidatta Dwibedi, Chelsea Finn, Montse Gonzalez Arenas, Keerthana Gopalakrishnan, Karol Hausman, Brian Ichter, Alex Irpan, Nikhil Joshi, Ryan Julian, et al., 2024. Autort: Embodied foundation models for large scale orchestration of robotic agents. arXiv preprint arXiv:2401.12963 (2024)."},{"key":"e_1_3_2_2_4_1","first-page":"393","volume-title":"Nature","volume":"602","author":"Aubin Cameron A","year":"2022","unstructured":"Cameron A Aubin, Benjamin Gorissen, Edoardo Milana, Philip R Buskohl, Nathan Lazarus, Geoffrey A Slipher, Christoph Keplinger, Josh Bongard, Fumiya Iida, Jennifer A Lewis, et al., 2022. Towards enduring autonomous robots via embodied energy. Nature, Vol. 602, 7897 (2022), 393-402."},{"key":"e_1_3_2_2_5_1","unstructured":"Shuai Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge Sibo Song Kai Dang Peng Wang Shijie Wang Jun Tang et al. 2025. Qwen2. 5-vl technical report. arXiv preprint arXiv:2502.13923 (2025)."},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.52202\/079017-1684"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01370"},{"key":"e_1_3_2_2_8_1","unstructured":"Liang Chen Lei Li Haozhe Zhao Yifan Song and Vinci. 2025. R1-V: Reinforcing Super Generalization Ability in Vision-Language Models with Less Than $3. https:\/\/github.com\/Deep-Agent\/R1-V. Accessed: 2025-02-02."},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/2809695.2809724"},{"key":"e_1_3_2_2_10_1","volume-title":"Ddl: Empowering delivery drones with large-scale urban sensing capability","author":"Chen Xuecheng","year":"2024","unstructured":"Xuecheng Chen, Haoyang Wang, Yuhan Cheng, Haohao Fu, Yuxuan Liu, Fan Dang, Yunhao Liu, Jinqiang Cui, and Xinlei Chen. 2024a. Ddl: Empowering delivery drones with large-scale urban sensing capability. IEEE Journal of Selected Topics in Signal Processing (2024)."},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/JIOT.2024.3389771"},{"key":"e_1_3_2_2_12_1","volume-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 24185-24198","author":"Chen Zhe","year":"2024","unstructured":"Zhe Chen, Jiannan Wu, Wenhai Wang, Weijie Su, Guo Chen, Sen Xing, Muyan Zhong, Qinglong Zhang, Xizhou Zhu, Lewei Lu, et al., 2024b. Internvl: Scaling up vision foundation models and aligning for generic visual-linguistic tasks. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 24185-24198."},{"key":"e_1_3_2_2_13_1","volume-title":"Videgothink: Assessing egocentric video understanding capabilities for embodied ai. arXiv preprint arXiv:2410.11623","author":"Cheng Sijie","year":"2024","unstructured":"Sijie Cheng, Kechen Fang, Yangyang Yu, Sicheng Zhou, Bohao Li, Ye Tian, Tingguang Li, Lei Han, and Yang Liu. 2024. Videgothink: Assessing egocentric video understanding capabilities for embodied ai. arXiv preprint arXiv:2410.11623 (2024)."},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1002\/cne.902980205"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1142\/S2301385024500109"},{"key":"e_1_3_2_2_16_1","first-page":"1481","volume-title":"Science","volume":"344","author":"Donoso Ma\u00ebl","year":"2014","unstructured":"Ma\u00ebl Donoso, Anne GE Collins, and Etienne Koechlin. 2014. Foundations of human reasoning in the prefrontal cortex. Science, Vol. 344, 6191 (2014), 1481-1486."},{"key":"e_1_3_2_2_17_1","volume-title":"Corey Lynch, Aakanksha Chowdhery, Ayzaan Wahid, Jonathan Tompson, Quan Vuong, Tianhe Yu, Wenlong Huang, et al.","author":"Driess Danny","year":"2023","unstructured":"Danny Driess, Fei Xia, Mehdi SM Sajjadi, Corey Lynch, Aakanksha Chowdhery, Ayzaan Wahid, Jonathan Tompson, Quan Vuong, Tianhe Yu, Wenlong Huang, et al., 2023. Palm-e: An embodied multimodal language model. (2023)."},{"key":"e_1_3_2_2_18_1","volume-title":"VL-Nav: Real-time Vision-Language Navigation with Spatial Reasoning. arXiv preprint arXiv:2502.00931","author":"Du Yi","year":"2025","unstructured":"Yi Du, Taimeng Fu, Zhuoqun Chen, Bowen Li, Shaoshu Su, Zhipeng Zhao, and Chen Wang. 2025. VL-Nav: Real-time Vision-Language Navigation with Spatial Reasoning. arXiv preprint arXiv:2502.00931 (2025)."},{"key":"e_1_3_2_2_19_1","volume-title":"Video-of-thought: Step-by-step video reasoning from perception to cognition. arXiv preprint arXiv:2501.03230","author":"Fei Hao","year":"2024","unstructured":"Hao Fei, Shengqiong Wu, Wei Ji, Hanwang Zhang, Meishan Zhang, Mong-Li Lee, and Wynne Hsu. 2024. Video-of-thought: Step-by-step video reasoning from perception to cognition. arXiv preprint arXiv:2501.03230 (2024)."},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1038\/s41467-022-30761-2"},{"key":"e_1_3_2_2_21_1","first-page":"662","volume-title":"Science","volume":"308","author":"Fogassi Leonardo","year":"2005","unstructured":"Leonardo Fogassi, Pier Francesco Ferrari, Benno Gesierich, Stefano Rozzi, Fabian Chersi, and Giacomo Rizzolatti. 2005. Parietal lobe: from action organization to intention understanding. Science, Vol. 308, 5722 (2005), 662-667."},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1002\/wcs.1226"},{"key":"e_1_3_2_2_23_1","unstructured":"Chen Gao Baining Zhao Weichen Zhang Jinzhu Mao Jun Zhang Zhiheng Zheng Fanhang Man Jianjie Fang Zile Zhou Jinqiang Cui et al. 2024. EmbodiedCity: A Benchmark Platform for Embodied Agent in Real-world City Environment. arXiv preprint arXiv:2410.09604 (2024)."},{"key":"e_1_3_2_2_24_1","unstructured":"Google. 2024. Gemini API. https:\/\/ai.google.dev\/gemini-api. Accessed: 2025-04-12."},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01842"},{"key":"e_1_3_2_2_26_1","volume-title":"MR-COGraphs: Communication-efficient Multi-Robot Open-vocabulary Mapping System via 3D Scene Graphs","author":"Gu Qiuyi","year":"2025","unstructured":"Qiuyi Gu, Zhaocheng Ye, Jincheng Yu, Jiahao Tang, Tinghao Yi, Yuhan Dong, Jian Wang, Jinqiang Cui, Xinlei Chen, and Yu Wang. 2025. MR-COGraphs: Communication-efficient Multi-Robot Open-vocabulary Mapping System via 3D Scene Graphs. IEEE Robotics and Automation Letters (2025)."},{"key":"e_1_3_2_2_27_1","volume-title":"Yifei Liu, Ning Shang, Youran Sun, Yi Zhu, Fan Yang, and Mao Yang.","author":"Guan Xinyu","year":"2025","unstructured":"Xinyu Guan, Li Lyna Zhang, Yifei Liu, Ning Shang, Youran Sun, Yi Zhu, Fan Yang, and Mao Yang. 2025. rStar-Math: Small LLMs Can Master Math Reasoning with Self-Evolved Deep Thinking. arXiv preprint arXiv:2501.04519 (2025)."},{"key":"e_1_3_2_2_28_1","unstructured":"Daya Guo Dejian Yang Haowei Zhang Junxiao Song Ruoyu Zhang Runxin Xu Qihao Zhu Shirong Ma Peiyi Wang Xiao Bi et al. 2025. Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning. arXiv preprint arXiv:2501.12948 (2025)."},{"key":"e_1_3_2_2_29_1","volume-title":"Mathprompter: Mathematical reasoning using large language models. arXiv preprint arXiv:2303.05398","author":"Imani Shima","year":"2023","unstructured":"Shima Imani, Liang Du, and Harsh Shrivastava. 2023. Mathprompter: Mathematical reasoning using large language models. arXiv preprint arXiv:2303.05398 (2023)."},{"key":"e_1_3_2_2_30_1","volume-title":"The spatial resolution of visual attention. Cognitive psychology","author":"Intriligator James","year":"2001","unstructured":"James Intriligator and Patrick Cavanagh. 2001. The spatial resolution of visual attention. Cognitive psychology, Vol. 43, 3 (2001), 171-216."},{"key":"e_1_3_2_2_31_1","volume-title":"ICLR 2025 Workshop on Embodied Intelligence with Large Language Models In Open City Environment.","author":"Jian Zhuozhu","unstructured":"Zhuozhu Jian, Xuran Pu, Jianjie Fang, Zhiyuan Deng, Xueqian Wang, Xinlei Chen, et al., [n.d.]. A Large Language Model-Driven Heterogeneous Air-Ground Search Swarm. In ICLR 2025 Workshop on Embodied Intelligence with Large Language Models In Open City Environment."},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02095"},{"key":"e_1_3_2_2_33_1","volume-title":"Purifying large language models by ensembling a small language model. arXiv preprint arXiv:2402.14845","author":"Li Tianlin","year":"2024","unstructured":"Tianlin Li, Qian Liu, Tianyu Pang, Chao Du, Qing Guo, Yang Liu, and Min Lin. 2024a. Purifying large language models by ensembling a small language model. arXiv preprint arXiv:2402.14845 (2024)."},{"key":"e_1_3_2_2_34_1","volume-title":"Video-llava: Learning united visual representation by alignment before projection. arXiv preprint arXiv:2311.10122","author":"Lin Bin","year":"2023","unstructured":"Bin Lin, Yang Ye, Bin Zhu, Jiaxi Cui, Munan Ning, Peng Jin, and Li Yuan. 2023. Video-llava: Learning united visual representation by alignment before projection. arXiv preprint arXiv:2311.10122 (2023)."},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00566"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.3390\/app15010107"},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1142\/S230138502441019X"},{"key":"e_1_3_2_2_38_1","volume-title":"Aligning cyber space with physical world: A comprehensive survey on embodied ai. arXiv preprint arXiv:2407.06886","author":"Liu Yang","year":"2024","unstructured":"Yang Liu, Weixing Chen, Yongjie Bai, Xiaodan Liang, Guanbin Li, Wen Gao, and Liang Lin. 2024a. Aligning cyber space with physical world: A comprehensive survey on embodied ai. arXiv preprint arXiv:2407.06886 (2024)."},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3643832.3661872"},{"key":"e_1_3_2_2_40_1","volume-title":"Visual-rft: Visual reinforcement fine-tuning. arXiv preprint arXiv:2503.01785","author":"Liu Ziyu","year":"2025","unstructured":"Ziyu Liu, Zeyi Sun, Yuhang Zang, Xiaoyi Dong, Yuhang Cao, Haodong Duan, Dahua Lin, and Jiaqi Wang. 2025a. Visual-rft: Visual reinforcement fine-tuning. arXiv preprint arXiv:2503.01785 (2025)."},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASE.2025.3534143"},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.136"},{"key":"e_1_3_2_2_43_1","first-page":"46212","article-title":"Egoschema: A diagnostic benchmark for very long-form video language understanding","volume":"36","author":"Mangalam Karttikeya","year":"2023","unstructured":"Karttikeya Mangalam, Raiymbek Akshulakov, and Jitendra Malik. 2023. Egoschema: A diagnostic benchmark for very long-form video language understanding. Advances in Neural Information Processing Systems, Vol. 36 (2023), 46212-46244.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_44_1","unstructured":"Yingqian Min Zhipeng Chen Jinhao Jiang Jie Chen Jia Deng Yiwen Hu Yiru Tang Jiapeng Wang Xiaoxue Cheng Huatong Song et al. 2024. Imitate explore and self-improve: A reproduction report on slow-thinking reasoning systems. arXiv preprint arXiv:2412.09413 (2024)."},{"key":"e_1_3_2_2_45_1","unstructured":"OpenAI. 2024a. GPT-4o API. https:\/\/openai.com\/api\/. Accessed: 2025-04-12."},{"key":"e_1_3_2_2_46_1","unstructured":"OpenAI. 2024b. Learning to Reason with LLMs. https:\/\/openai.com\/index\/learning-to-reason-with-llms\/ Accessed: 2025-03-04."},{"key":"e_1_3_2_2_47_1","unstructured":"OpenAI. 2025. OpenAI o3-mini. https:\/\/openai.com\/index\/openai-o3-mini\/ Accessed: 2025-04-15."},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.scitotenv.2020.136546"},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.1126\/scirobotics.abg1188"},{"key":"e_1_3_2_2_50_1","first-page":"43000","article-title":"Rl on incorrect synthetic data scales the efficiency of llm math reasoning by eight-fold","volume":"37","author":"Setlur Amrith","year":"2025","unstructured":"Amrith Setlur, Saurabh Garg, Xinyang Geng, Naman Garg, Virginia Smith, and Aviral Kumar. 2025. Rl on incorrect synthetic data scales the efficiency of llm math reasoning by eight-fold. Advances in Neural Information Processing Systems, Vol. 37 (2025), 43000-43031.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_51_1","volume-title":"Conference on robot learning. PMLR, 492-504","author":"Shah Dhruv","year":"2023","unstructured":"Dhruv Shah, B\u0142a\u017cej Osi'nski, Sergey Levine, et al., 2023. Lm-nav: Robotic navigation with large pre-trained models of language, vision, and action. In Conference on robot learning. PMLR, 492-504."},{"key":"e_1_3_2_2_52_1","volume-title":"Ioannis Papaioannou, Arash Eshghi, Ioannis Konstas, and Oliver Lemon.","author":"Suglia Alessandro","year":"2024","unstructured":"Alessandro Suglia, Claudio Greco, Katie Baker, Jose L Part, Ioannis Papaioannou, Arash Eshghi, Ioannis Konstas, and Oliver Lemon. 2024. Alanavlm: A multimodal embodied ai foundation model for egocentric video understanding. arXiv preprint arXiv:2406.13807 (2024)."},{"key":"e_1_3_2_2_53_1","volume-title":"video-SALMONN-o1: Reasoning-enhanced Audio-visual Large Language Model. arXiv preprint arXiv:2502.11775","author":"Sun Guangzhi","year":"2025","unstructured":"Guangzhi Sun, Yudong Yang, Jimin Zhuang, Changli Tang, Yixuan Li, Wei Li, Zejun MA, and Chao Zhang. 2025. video-SALMONN-o1: Reasoning-enhanced Audio-visual Large Language Model. arXiv preprint arXiv:2502.11775 (2025)."},{"key":"e_1_3_2_2_54_1","unstructured":"Gemini Team Rohan Anil Sebastian Borgeaud Jean-Baptiste Alayrac Jiahui Yu Radu Soricut Johan Schalkwyk Andrew M Dai Anja Hauth Katie Millican et al. 2023. Gemini: a family of highly capable multimodal models. arXiv preprint arXiv:2312.11805 (2023)."},{"key":"e_1_3_2_2_55_1","unstructured":"Kimi Team Angang Du Bofei Gao Bowei Xing Changjiu Jiang Cheng Chen Cheng Li Chenjun Xiao Chenzhuang Du Chonghua Liao et al. 2025. Kimi k1. 5: Scaling reinforcement learning with llms. arXiv preprint arXiv:2501.12599 (2025)."},{"key":"e_1_3_2_2_56_1","unstructured":"Qwen Team. 2024a. Qwen-VL-Max. https:\/\/qwenlm.github.io\/blog\/qwen-vl-max\/. Accessed: 2025-04-12."},{"key":"e_1_3_2_2_57_1","unstructured":"Qwen Team. 2024b. QwQ: Reflect Deeply on the Boundaries of the Unknown. https:\/\/qwenlm.github.io\/blog\/qwq-32b-preview\/"},{"key":"e_1_3_2_2_58_1","volume-title":"Rao Muhammad Anwer, et al","author":"Thawakar Omkar","year":"2025","unstructured":"Omkar Thawakar, Dinura Dissanayake, Ketan More, Ritesh Thawkar, Ahmed Heakl, Noor Ahsan, Yuhao Li, Mohammed Zumri, Jean Lahoud, Rao Muhammad Anwer, et al., 2025. Llamav-o1: Rethinking step-by-step visual reasoning in llms. arXiv preprint arXiv:2501.06186 (2025)."},{"key":"e_1_3_2_2_59_1","volume-title":"Calibrating large language models using their generations only. arXiv preprint arXiv:2403.05973","author":"Ulmer Dennis","year":"2024","unstructured":"Dennis Ulmer, Martin Gubri, Hwaran Lee, Sangdoo Yun, and Seong Joon Oh. 2024. Calibrating large language models using their generations only. arXiv preprint arXiv:2403.05973 (2024)."},{"key":"e_1_3_2_2_60_1","unstructured":"Fali Wang Zhiwei Zhang Xianren Zhang Zongyu Wu Tzuhao Mo Qiuhao Lu Wanjing Wang Rui Li Junjie Xu Xianfeng Tang et al. 2024 e. A comprehensive survey of small language models in the era of large language models: Techniques enhancements applications collaboration with llms and trustworthiness. arXiv preprint arXiv:2411.03350 (2024)."},{"key":"e_1_3_2_2_61_1","doi-asserted-by":"publisher","DOI":"10.1145\/3715014.3722048"},{"key":"e_1_3_2_2_62_1","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM52122.2024.10621375"},{"key":"e_1_3_2_2_63_1","first-page":"75392","article-title":"Is a picture worth a thousand words? delving into spatial reasoning for vision language models","volume":"37","author":"Wang Jiayu","year":"2024","unstructured":"Jiayu Wang, Yifei Ming, Zhenmei Shi, Vibhav Vineet, Xin Wang, Sharon Li, and Neel Joshi. 2024c. Is a picture worth a thousand words? delving into spatial reasoning for vision language models. Advances in Neural Information Processing Systems, Vol. 37 (2024), 75392-75421.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_64_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01868"},{"key":"e_1_3_2_2_65_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72980-5_17"},{"key":"e_1_3_2_2_66_1","volume-title":"Denny Zhou, et al.","author":"Wei Jason","year":"2022","unstructured":"Jason Wei, Xuezhi Wang, Dale Schuurmans, Maarten Bosma, Fei Xia, Ed Chi, Quoc V Le, Denny Zhou, et al., 2022. Chain-of-thought prompting elicits reasoning in large language models. Advances in neural information processing systems, Vol. 35 (2022), 24824-24837."},{"key":"e_1_3_2_2_67_1","volume-title":"Logic-RL: Unleashing LLM Reasoning with Rule-Based Reinforcement Learning. arXiv preprint arXiv:2502.14768","author":"Xie Tian","year":"2025","unstructured":"Tian Xie, Zitian Gao, Qingnan Ren, Haoming Luo, Yuqian Hong, Bryan Dai, Joey Zhou, Kai Qiu, Zhirong Wu, and Chong Luo. 2025. Logic-RL: Unleashing LLM Reasoning with Rule-Based Reinforcement Learning. arXiv preprint arXiv:2502.14768 (2025)."},{"key":"e_1_3_2_2_68_1","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC59245.2023.00014"},{"key":"e_1_3_2_2_69_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.567"},{"key":"e_1_3_2_2_70_1","doi-asserted-by":"publisher","DOI":"10.1145\/3636534.3694730"},{"key":"e_1_3_2_2_71_1","unstructured":"An Yang Beichen Zhang Binyuan Hui Bofei Gao Bowen Yu Chengpeng Li Dayiheng Liu Jianhong Tu Jingren Zhou Junyang Lin et al. 2024b. Qwen2. 5-math technical report: Toward mathematical expert model via self-improvement. arXiv preprint arXiv:2409.12122 (2024)."},{"key":"e_1_3_2_2_72_1","volume-title":"Thinking in space: How multimodal large language models see, remember, and recall spaces. arXiv preprint arXiv:2412.14171","author":"Yang Jihan","year":"2024","unstructured":"Jihan Yang, Shusheng Yang, Anjali W Gupta, Rilyn Han, Li Fei-Fei, and Saining Xie. 2024a. Thinking in space: How multimodal large language models see, remember, and recall spaces. arXiv preprint arXiv:2412.14171 (2024)."},{"key":"e_1_3_2_2_73_1","unstructured":"Weihao Zeng Yuzhen Huang Wei Liu Keqing He Qian Liu Zejun Ma and Junxian He. 2025. 7B Model and 8K Examples: Emerging Reasoning with Reinforcement Learning is Both Effective and Efficient. https:\/\/hkust-nlp.notion.site\/simplerl-reason. Notion Blog."},{"key":"e_1_3_2_2_74_1","volume-title":"How to enable llm with 3d capacity? a survey of spatial reasoning in llm. arXiv preprint arXiv:2504.05786","author":"Zha Jirong","year":"2025","unstructured":"Jirong Zha, Yuxuan Fan, Xiao Yang, Chen Gao, and Xinlei Chen. 2025. How to enable llm with 3d capacity? a survey of spatial reasoning in llm. arXiv preprint arXiv:2504.05786 (2025)."},{"key":"e_1_3_2_2_75_1","first-page":"64735","article-title":"Rest-mcts*: Llm self-training via process reward guided tree search","volume":"37","author":"Zhang Dan","year":"2025","unstructured":"Dan Zhang, Sining Zhoubian, Ziniu Hu, Yisong Yue, Yuxiao Dong, and Jie Tang. 2025b. Rest-mcts*: Llm self-training via process reward guided tree search. Advances in Neural Information Processing Systems, Vol. 37 (2025), 64735-64772.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_76_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.1511"},{"key":"e_1_3_2_2_77_1","volume-title":"Effective prompt extraction from language models. arXiv preprint arXiv:2307.06865","author":"Zhang Yiming","year":"2023","unstructured":"Yiming Zhang, Nicholas Carlini, and Daphne Ippolito. 2023. Effective prompt extraction from language models. arXiv preprint arXiv:2307.06865 (2023)."},{"key":"e_1_3_2_2_78_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.1558"},{"key":"e_1_3_2_2_79_1","unstructured":"Theodore Zhao Mu Wei J Samuel Preston and Hoifung Poon. 2023. Automatic Calibration and Error Correction for Generative Large Language Models via Pareto Optimal Self-Supervision. (2023)."},{"key":"e_1_3_2_2_80_1","doi-asserted-by":"publisher","DOI":"10.1038\/nrn2776"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755703","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:35:06Z","timestamp":1765308906000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755703"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":80,"alternative-id":["10.1145\/3746027.3755703","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755703","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}