{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T13:02:44Z","timestamp":1785502964483,"version":"3.56.0"},"publisher-location":"New York, NY, USA","reference-count":53,"publisher":"ACM","funder":[{"name":"National Natural Science Foundation of China","award":["62272437"],"award-info":[{"award-number":["62272437"]}]},{"name":"National Natural Science Foundation of China","award":["72271083"],"award-info":[{"award-number":["72271083"]}]},{"name":"National Natural Science Foundation of China","award":["62402470"],"award-info":[{"award-number":["62402470"]}]},{"name":"Fundamental Research Funds for the Central Universities of China","award":["PA2025IISL0099"],"award-info":[{"award-number":["PA2025IISL0099"]}]},{"name":"Fundamental Research Funds for the Central Universities of China","award":["PA2024GDSK0107"],"award-info":[{"award-number":["PA2024GDSK0107"]}]},{"name":"Fundamental Research Funds for the Central Universities of China","award":["WK2100000053"],"award-info":[{"award-number":["WK2100000053"]}]},{"name":"Postdoctoral Fellowship Program of CPSF","award":["GZC20241643"],"award-info":[{"award-number":["GZC20241643"]}]},{"name":"Anhui Postdoctoral Scientific Research Program Foundation","award":["2025B1063"],"award-info":[{"award-number":["2025B1063"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,8,9]]},"DOI":"10.1145\/3770854.3780206","type":"proceedings-article","created":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T12:07:40Z","timestamp":1785499660000},"page":"49-58","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["MGFRec: Towards Reinforced Reasoning Recommendation with Multiple Groundings and Feedback"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-6894-364X","authenticated-orcid":false,"given":"Shihao","family":"Cai","sequence":"first","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5187-9196","authenticated-orcid":false,"given":"Chongming","family":"Gao","sequence":"additional","affiliation":[{"name":"Intelligent Interconnected Systems Laboratory of Anhui Province (Hefei University of Technology), University of Science and Technology of China, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3249-2164","authenticated-orcid":false,"given":"Haoyan","family":"Liu","sequence":"additional","affiliation":[{"name":"Intelligent Interconnected Systems Laboratory of Anhui Province (Hefei University of Technology), University of Science and Technology of China, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2616-6880","authenticated-orcid":false,"given":"Wentao","family":"Shi","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2981-5812","authenticated-orcid":false,"given":"Jianshan","family":"Sun","sequence":"additional","affiliation":[{"name":"Hefei University of Technology, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9224-2431","authenticated-orcid":false,"given":"Ruiming","family":"Tang","sequence":"additional","affiliation":[{"name":"Kuaishou Inc., Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5828-9842","authenticated-orcid":false,"given":"Fuli","family":"Feng","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,4,20]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3661383"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/3716393"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.589"},{"key":"e_1_3_2_2_4_1","first-page":"4844","volume-title":"ACL","author":"Cai Shihao","year":"2025","unstructured":"Shihao Cai, Chongming Gao, Yang Zhang, Wentao Shi, Jizhi Zhang, Keqin Bao, Qifan Wang, and Fuli Feng. 2025a. K-order Ranking Preference Optimization for Large Language Models. In Findings of the Association for Computational Linguistics, ACL 2025,. 4844-4859."},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3726302.3729893"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3701551.3703572"},{"key":"e_1_3_2_2_7_1","unstructured":"Mingyang Chen Tianpeng Li Haoze Sun Yijie Zhou Chenzheng Zhu Haofen Wang Jeff Z Pan Wen Zhang Huajun Chen Fan Yang et al. 2025b. Learning to reason with search for llms via reinforcement learning. arXiv preprint arXiv:2503.19470 (2025)."},{"key":"e_1_3_2_2_8_1","volume-title":"Fine-grained List-wise Alignment for Generative Medication Recommendation. In The Thirty-ninth Annual Conference on Neural Information Processing Systems.","author":"Fan Chenxiao","year":"2025","unstructured":"Chenxiao Fan, Chongming Gao, Wentao Shi, Yaxin Gong, Zhao Zihao, and Fuli Feng. 2025. Fine-grained List-wise Alignment for Generative Medication Recommendation. In The Thirty-ninth Annual Conference on Neural Information Processing Systems."},{"key":"e_1_3_2_2_9_1","unstructured":"Runnan Fang Shihao Cai Baixuan Li Jialong Wu Guangyu Li Wenbiao Yin Xinyu Wang Xiaobin Wang Liangcai Su Zhen Zhang et al. 2025a. Towards General Agentic Intelligence via Environment Scaling. arXiv preprint arXiv:2509.13311 (2025)."},{"key":"e_1_3_2_2_10_1","volume-title":"Reason4Rec: Large Language Models for Recommendation with Deliberative User Preference Alignment. arXiv preprint arXiv:2502.02061","author":"Fang Yi","year":"2025","unstructured":"Yi Fang, Wenjie Wang, Yang Zhang, Fengbin Zhu, Qifan Wang, Fuli Feng, and Xiangnan He. 2025b. Reason4Rec: Large Language Models for Recommendation with Deliberative User Preference Alignment. arXiv preprint arXiv:2502.02061 (2025)."},{"key":"e_1_3_2_2_11_1","volume-title":"Retool: Reinforcement learning for strategic tool use in llms. arXiv preprint arXiv:2504.11536","author":"Feng Jiazhan","year":"2025","unstructured":"Jiazhan Feng, Shijue Huang, Xingwei Qu, Ge Zhang, Yujia Qin, Baoquan Zhong, Chengquan Jiang, Jinxin Chi, and Wanjun Zhong. 2025. Retool: Reinforcement learning for strategic tool use in llms. arXiv preprint arXiv:2504.11536 (2025)."},{"key":"e_1_3_2_2_12_1","volume-title":"From llm reasoning to autonomous ai agents: A comprehensive review. arXiv preprint arXiv:2504.19678","author":"Ferrag Mohamed Amine","year":"2025","unstructured":"Mohamed Amine Ferrag, Norbert Tihanyi, and Merouane Debbah. 2025. From llm reasoning to autonomous ai agents: A comprehensive review. arXiv preprint arXiv:2504.19678 (2025)."},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3696410.3714524"},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3726302.3729981"},{"key":"e_1_3_2_2_15_1","unstructured":"Daya Guo Dejian Yang Haowei Zhang Junxiao Song Ruoyu Zhang Runxin Xu Qihao Zhu Shirong Ma Peiyi Wang Xiao Bi et al. 2025. Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning. arXiv preprint arXiv:2501.12948 (2025)."},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3583780.3614949"},{"key":"e_1_3_2_2_17_1","volume-title":"Session-based Recommendations with Recurrent Neural Networks. In 4th International Conference on Learning Representations.","author":"Hidasi Bal\u00e1zs","year":"2016","unstructured":"Bal\u00e1zs Hidasi, Alexandros Karatzoglou, Linas Baltrunas, and Domonkos Tikk. 2016. Session-based Recommendations with Recurrent Neural Networks. In 4th International Conference on Learning Representations."},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/3543507.3583434"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3534678.3539381"},{"key":"e_1_3_2_2_20_1","unstructured":"Aaron Jaech Adam Kalai Adam Lerer Adam Richardson Ahmed El-Kishky Aiden Low Alec Helyar Aleksander Madry Alex Beutel Alex Carney et al. 2024. Openai o1 system card. arXiv preprint arXiv:2412.16720 (2024)."},{"key":"e_1_3_2_2_21_1","volume-title":"Search-r1: Training llms to reason and leverage search engines with reinforcement learning. arXiv preprint arXiv:2503.09516","author":"Jin Bowen","year":"2025","unstructured":"Bowen Jin, Hansi Zeng, Zhenrui Yue, Jinsung Yoon, Sercan Arik, Dong Wang, Hamed Zamani, and Jiawei Han. 2025. Search-r1: Training llms to reason and leverage search engines with reinforcement learning. arXiv preprint arXiv:2503.09516 (2025)."},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.5555\/1622737.1622748"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDM.2018.00035"},{"key":"e_1_3_2_2_24_1","unstructured":"Timo Kaufmann Paul Weng Viktor Bengs and Eyke H\u00fcllermeier. 2024. A survey of reinforcement learning from human feedback. (2024)."},{"key":"e_1_3_2_2_25_1","volume-title":"Towards general text embeddings with multi-stage contrastive learning. arXiv preprint arXiv:2308.03281","author":"Li Zehan","year":"2023","unstructured":"Zehan Li, Xin Zhang, Yanzhao Zhang, Dingkun Long, Pengjun Xie, and Meishan Zhang. 2023. Towards general text embeddings with multi-stage contrastive learning. arXiv preprint arXiv:2308.03281 (2023)."},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3657690"},{"key":"e_1_3_2_2_27_1","volume-title":"Rec-r1: Bridging generative large language models and user-centric recommendation systems via reinforcement learning. arXiv preprint arXiv:2503.24289","author":"Lin Jiacheng","year":"2025","unstructured":"Jiacheng Lin, Tian Wang, and Kun Qian. 2025. Rec-r1: Bridging generative large language models and user-centric recommendation systems via reinforcement learning. arXiv preprint arXiv:2503.24289 (2025)."},{"key":"e_1_3_2_2_28_1","volume-title":"Jinpeng Wang, Sheng Chen, and Ji-Rong Wen.","author":"Liu Enze","year":"2025","unstructured":"Enze Liu, Bowen Zheng, Xiaolei Wang, Wayne Xin Zhao, Jinpeng Wang, Sheng Chen, and Ji-Rong Wen. 2025. LARES: Latent Reasoning for Sequential Recommendation. arXiv preprint arXiv:2505.16865 (2025)."},{"key":"e_1_3_2_2_29_1","unstructured":"OpenAI. 2025. Introducing GPT-4.1 in the API. In technical report. https:\/\/openai.com\/index\/gpt-4-1\/"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"crossref","unstructured":"Long Ouyang Jeffrey Wu Xu Jiang Diogo Almeida Carroll Wainwright Pamela Mishkin Chong Zhang Sandhini Agarwal Katarina Slama Alex Ray et al. 2022. Training language models to follow instructions with human feedback. Advances in neural information processing systems Vol. 35 (2022) 27730-27744.","DOI":"10.52202\/068431-2011"},{"key":"e_1_3_2_2_31_1","volume-title":"Direct preference optimization: Your language model is secretly a reward model. Advances in neural information processing systems","author":"Rafailov Rafael","year":"2023","unstructured":"Rafael Rafailov, Archit Sharma, Eric Mitchell, Christopher D Manning, Stefano Ermon, and Chelsea Finn. 2023. Direct preference optimization: Your language model is secretly a reward model. Advances in neural information processing systems, Vol. 36 (2023), 53728-53741."},{"key":"e_1_3_2_2_32_1","volume-title":"Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347","author":"Schulman John","year":"2017","unstructured":"John Schulman, Filip Wolski, Prafulla Dhariwal, Alec Radford, and Oleg Klimov. 2017. Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347 (2017)."},{"key":"e_1_3_2_2_33_1","volume-title":"Deepseekmath: Pushing the limits of mathematical reasoning in open language models. arXiv preprint arXiv:2402.03300","author":"Shao Zhihong","year":"2024","unstructured":"Zhihong Shao, Peiyi Wang, Qihao Zhu, Runxin Xu, Junxiao Song, Xiao Bi, Haowei Zhang, Mingchuan Zhang, YK Li, Yang Wu, et al., 2024. Deepseekmath: Pushing the limits of mathematical reasoning in open language models. arXiv preprint arXiv:2402.03300 (2024)."},{"key":"e_1_3_2_2_34_1","unstructured":"Haozhan Shen Peng Liu Jingcheng Li Chunxin Fang Yibo Ma Jiajia Liao Qiaoli Shen Zilun Zhang Kangjia Zhao Qianqian Zhang et al. 2025. Vlm-r1: A stable and generalizable r1-style large vision-language model. arXiv preprint arXiv:2504.07615 (2025)."},{"key":"e_1_3_2_2_35_1","volume-title":"A Human-Central Recommendation Framework with Large Language Models. arXiv preprint arXiv:2308.09904","author":"Shu Yubo","year":"2023","unstructured":"Yubo Shu, Hansu Gu, Peng Zhang, Haonan Zhang, Tun Lu, Dongsheng Li, and Ning Gu. 2023. RAH! RecSys-Assistant-Human: A Human-Central Recommendation Framework with Large Language Models. arXiv preprint arXiv:2308.09904 (2023)."},{"key":"e_1_3_2_2_36_1","volume-title":"Lei Fang, and Ji-Rong Wen.","author":"Song Huatong","year":"2025","unstructured":"Huatong Song, Jinhao Jiang, Yingqian Min, Jie Chen, Zhipeng Chen, Wayne Xin Zhao, Lei Fang, and Ji-Rong Wen. 2025. R1-searcher: Incentivizing the search capability in llms via reinforcement learning. arXiv preprint arXiv:2503.05592 (2025)."},{"key":"e_1_3_2_2_37_1","volume-title":"Think before recommend: Unleashing the latent reasoning power for sequential recommendation. arXiv preprint arXiv:2503.22675","author":"Tang Jiakai","year":"2025","unstructured":"Jiakai Tang, Sunhao Dai, Teng Shi, Jun Xu, Xu Chen, Wen Chen, Wu Jian, and Yuning Jiang. 2025a. Think before recommend: Unleashing the latent reasoning power for sequential recommendation. arXiv preprint arXiv:2503.22675 (2025)."},{"key":"e_1_3_2_2_38_1","volume-title":"Think before recommend: Unleashing the latent reasoning power for sequential recommendation. arXiv preprint arXiv:2503.22675","author":"Tang Jiakai","year":"2025","unstructured":"Jiakai Tang, Sunhao Dai, Teng Shi, Jun Xu, Xu Chen, Wen Chen, Wu Jian, and Yuning Jiang. 2025b. Think before recommend: Unleashing the latent reasoning power for sequential recommendation. arXiv preprint arXiv:2503.22675 (2025)."},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3159652.3159656"},{"key":"e_1_3_2_2_40_1","volume-title":"September","author":"Team Qwen","year":"2024","unstructured":"Qwen Team. 2024. Qwen2. 5: A party of foundation models, September 2024. https:\/\/qwenlm. github. io\/blog\/qwen2, Vol. 5, 4 (2024)."},{"key":"e_1_3_2_2_41_1","unstructured":"Tongyi DeepResearch Team. 2025. Tongyi DeepResearch: A New Era of Open-Source AI Researchers. https:\/\/github.com\/Alibaba-NLP\/DeepResearch."},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.944"},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i17.29887"},{"key":"e_1_3_2_2_44_1","unstructured":"Yan Wang Zhixuan Chu Xin Ouyang Simeng Wang Hongyan Hao Yue Shen Jinjie Gu Siqiao Xue James Y Zhang Qing Cui et al. 2023. Enhancing recommender systems with large language model reasoning graphs. arXiv preprint arXiv:2308.10835 (2023)."},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11280-024-01291-2"},{"key":"e_1_3_2_2_46_1","volume-title":"Table-r1: Region-based reinforcement learning for table understanding. arXiv preprint arXiv:2505.12415","author":"Wu Zhenhe","year":"2025","unstructured":"Zhenhe Wu, Jian Yang, Jiaheng Liu, Xianjie Wu, Changzai Pan, Jie Zhang, Yu Zhao, Shuangyong Song, Yongxiang Li, and Zhoujun Li. 2025. Table-r1: Region-based reinforcement learning for table understanding. arXiv preprint arXiv:2505.12415 (2025)."},{"key":"e_1_3_2_2_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/3539618.3591624"},{"key":"e_1_3_2_2_48_1","volume-title":"Towards Large Recommender Models with Reasoning. arXiv preprint arXiv:2505.16994","author":"You Runyang","year":"2025","unstructured":"Runyang You, Yongqi Li, Xinyu Lin, Xin Zhang, Wenjie Wang, Wenjie Li, and Liqiang Nie. 2025. R$^2$ec: Towards Large Recommender Models with Reasoning. arXiv preprint arXiv:2505.16994 (2025)."},{"key":"e_1_3_2_2_49_1","volume-title":"Yu Chen, and Ji-Rong Wen.","author":"Zhang Junjie","year":"2025","unstructured":"Junjie Zhang, Beichen Zhang, Wenqi Sun, Hongyu Lu, Wayne Xin Zhao, Yu Chen, and Ji-Rong Wen. 2025c. Slow Thinking for Sequential Recommendation. arXiv preprint arXiv:2504.09627 (2025)."},{"key":"e_1_3_2_2_50_1","volume-title":"Collm: Integrating collaborative embeddings into large language models for recommendation","author":"Zhang Yang","year":"2025","unstructured":"Yang Zhang, Fuli Feng, Jizhi Zhang, Keqin Bao, Qifan Wang, and Xiangnan He. 2025a. Collm: Integrating collaborative embeddings into large language models for recommendation. IEEE Transactions on Knowledge and Data Engineering (2025)."},{"key":"e_1_3_2_2_51_1","volume-title":"Reinforced Latent Reasoning for LLM-based Recommendation. arXiv preprint arXiv:2505.19092","author":"Zhang Yang","year":"2025","unstructured":"Yang Zhang, Wenxin Xu, Xiaoyan Zhao, Wenjie Wang, Fuli Feng, Xiangnan He, and Tat-Seng Chua. 2025b. Reinforced Latent Reasoning for LLM-based Recommendation. arXiv preprint arXiv:2505.19092 (2025)."},{"key":"e_1_3_2_2_52_1","volume-title":"Reason-to-Recommend: Using Interaction-of-Thought Reasoning to Enhance LLM Recommendation. arXiv preprint arXiv:2506.05069","author":"Zhao Keyu","year":"2025","unstructured":"Keyu Zhao, Fengli Xu, and Yong Li. 2025. Reason-to-Recommend: Using Interaction-of-Thought Reasoning to Enhance LLM Recommendation. arXiv preprint arXiv:2506.05069 (2025)."},{"key":"e_1_3_2_2_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2024.3392335"}],"event":{"name":"KDD '26: The 32nd ACM SIGKDD Conference on Knowledge Discovery and Data Mining","location":"Jeju Island Republic of Korea","sponsor":["SIGMOD ACM Special Interest Group on Management of Data","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data"]},"container-title":["Proceedings of the 32nd ACM SIGKDD Conference on Knowledge Discovery and Data Mining V.1"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3770854.3780206","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T12:12:28Z","timestamp":1785499948000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3770854.3780206"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,20]]},"references-count":53,"alternative-id":["10.1145\/3770854.3780206","10.1145\/3770854"],"URL":"https:\/\/doi.org\/10.1145\/3770854.3780206","relation":{},"subject":[],"published":{"date-parts":[[2026,4,20]]},"assertion":[{"value":"2026-04-20","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}