{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,25]],"date-time":"2026-08-25T20:22:57Z","timestamp":1787689377887,"version":"build-2784847793"},"publisher-location":"New York, NY, USA","reference-count":41,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,8,9]]},"DOI":"10.1145\/3770854.3783917","type":"proceedings-article","created":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T12:07:40Z","timestamp":1785499660000},"page":"2551-2561","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Optimizing Generative Ranking Relevance via Reinforcement Learning in Xiaohongshu Search"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-8773-7301","authenticated-orcid":false,"given":"Ziyang","family":"Zeng","sequence":"first","affiliation":[{"name":"Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9216-7032","authenticated-orcid":false,"given":"Heming","family":"Jing","sequence":"additional","affiliation":[{"name":"Xiaohongshu Inc., Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-2977-398X","authenticated-orcid":false,"given":"Jindong","family":"Chen","sequence":"additional","affiliation":[{"name":"Xiaohongshu Inc., Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0126-932X","authenticated-orcid":false,"given":"Xiangli","family":"Li","sequence":"additional","affiliation":[{"name":"Xiaohongshu Inc., Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-8009-5112","authenticated-orcid":false,"given":"Hongyu","family":"Liu","sequence":"additional","affiliation":[{"name":"Xiaohongshu Inc., Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-0566-0622","authenticated-orcid":false,"given":"Yixuan","family":"He","sequence":"additional","affiliation":[{"name":"Xiaohongshu Inc., Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2836-276X","authenticated-orcid":false,"given":"Zhengyu","family":"Li","sequence":"additional","affiliation":[{"name":"Xiaohongshu Inc., Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-3025-2142","authenticated-orcid":false,"given":"Yige","family":"Sun","sequence":"additional","affiliation":[{"name":"Xiaohongshu Inc., Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-7453-5781","authenticated-orcid":false,"given":"Zheyong","family":"Xie","sequence":"additional","affiliation":[{"name":"Xiaohongshu Inc., Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5333-1346","authenticated-orcid":false,"given":"Yuqing","family":"Yang","sequence":"additional","affiliation":[{"name":"Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3795-8824","authenticated-orcid":false,"given":"Shaosheng","family":"Cao","sequence":"additional","affiliation":[{"name":"Xiaohongshu Inc., Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-2127-0702","authenticated-orcid":false,"given":"Jun","family":"Fan","sequence":"additional","affiliation":[{"name":"Xiaohongshu Inc., Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-8838-2785","authenticated-orcid":false,"given":"Yi","family":"Wu","sequence":"additional","affiliation":[{"name":"Xiaohongshu Inc., Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-1274-7111","authenticated-orcid":false,"given":"Yao","family":"Hu","sequence":"additional","affiliation":[{"name":"Xiaohongshu Inc., Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,4,20]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/3477495.3531943"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/3409256.3409828"},{"key":"e_1_3_2_2_3_1","volume-title":"RL Generalizes: A Comparative Study of Foundation Model Post-training. In Forty-second International Conference on Machine Learning. https:\/\/openreview.net\/forum?id=dYur3yabMj","author":"Chu Tianzhe","year":"2025","unstructured":"Tianzhe Chu, Yuexiang Zhai, Jihan Yang, Shengbang Tong, Saining Xie, Dale Schuurmans, Quoc V Le, Sergey Levine, and Yi Ma. 2025. SFT Memorizes, RL Generalizes: A Comparative Study of Foundation Model Post-training. In Forty-second International Conference on Machine Learning. https:\/\/openreview.net\/forum?id=dYur3yabMj"},{"key":"e_1_3_2_2_4_1","volume-title":"Search engines: Information retrieval in practice","author":"Croft W Bruce","unstructured":"W Bruce Croft, Donald Metzler, and Trevor Strohman. 2010. Search engines: Information retrieval in practice. Vol. 520. Addison-Wesley Reading."},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1038\/S41586-025-09422-Z"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3701551.3703583"},{"key":"e_1_3_2_2_7_1","volume-title":"NIPS Deep Learning and Representation Learning Workshop. http:\/\/arxiv.org\/abs\/1503","author":"Hinton Geoffrey","year":"2015","unstructured":"Geoffrey Hinton, Oriol Vinyals, and Jeffrey Dean. 2015. Distilling the Knowledge in a Neural Network. In NIPS Deep Learning and Representation Learning Workshop. http:\/\/arxiv.org\/abs\/1503.02531"},{"key":"e_1_3_2_2_8_1","volume-title":"VinePPO: Refining Credit Assignment in RL Training of LLMs. In Forty-second International Conference on Machine Learning. https:\/\/openreview.net\/forum?id=Myx2kJFzAn","author":"Kazemnejad Amirhossein","year":"2025","unstructured":"Amirhossein Kazemnejad, Milad Aghajohari, Eva Portelance, Alessandro Sordoni, Siva Reddy, Aaron Courville, and Nicolas Le Roux. 2025. VinePPO: Refining Credit Assignment in RL Training of LLMs. In Forty-second International Conference on Machine Learning. https:\/\/openreview.net\/forum?id=Myx2kJFzAn"},{"key":"e_1_3_2_2_9_1","volume-title":"Sang Michael Xie, Shibani Santurkar, Surya Ganguli, Tatsunori Hashimoto, Thomas Icard, Tianyi Zhang, Vishrav Chaudhary, William Wang, Xuechen Li, Yifan Mai, Yuhui Zhang, and Yuta Koreeda.","author":"Liang Percy","year":"2023","unstructured":"Percy Liang, Rishi Bommasani, Tony Lee, Dimitris Tsipras, Dilara Soylu, Michihiro Yasunaga, Yian Zhang, Deepak Narayanan, Yuhuai Wu, Ananya Kumar, Benjamin Newman, Binhang Yuan, Bobby Yan, Ce Zhang, Christian Cosgrove, Christopher D Manning, Christopher Re, Diana Acosta-Navas, Drew A. Hudson, Eric Zelikman, Esin Durmus, Faisal Ladhak, Frieda Rong, Hongyu Ren, Huaxiu Yao, Jue WANG, Keshav Santhanam, Laurel Orr, Lucia Zheng, Mert Yuksekgonul, Mirac Suzgun, Nathan Kim, Neel Guha, Niladri S. Chatterji, Omar Khattab, Peter Henderson, Qian Huang, Ryan Andrew Chi, Sang Michael Xie, Shibani Santurkar, Surya Ganguli, Tatsunori Hashimoto, Thomas Icard, Tianyi Zhang, Vishrav Chaudhary, William Wang, Xuechen Li, Yifan Mai, Yuhui Zhang, and Yuta Koreeda. 2023. Holistic Evaluation of Language Models. Transactions on Machine Learning Research (2023). https:\/\/openreview.net\/forum?id=iO4LZibEqW Featured Certification, Expert Certification, Outstanding Certification."},{"key":"e_1_3_2_2_10_1","volume-title":"The Twelfth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=v8L0pN6EOi","author":"Lightman Hunter","year":"2024","unstructured":"Hunter Lightman, Vineet Kosaraju, Yuri Burda, Harrison Edwards, Bowen Baker, Teddy Lee, Jan Leike, John Schulman, Ilya Sutskever, and Karl Cobbe. 2024. Let's Verify Step by Step. In The Twelfth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=v8L0pN6EOi"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3657951"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.emnlp-industry.178"},{"key":"e_1_3_2_2_13_1","unstructured":"Tong Niu Shafiq Joty Ye Liu Caiming Xiong Yingbo Zhou and Semih Yavuz. 2024. JudgeRank: Leveraging Large Language Models for Reasoning-Intensive Reranking. arXiv:2411.00142 [cs.CL] https:\/\/arxiv.org\/abs\/2411.00142"},{"key":"e_1_3_2_2_14_1","unstructured":"Rodrigo Nogueira and Kyunghyun Cho. 2019. Passage Re-ranking with BERT. arXiv:1901.04085 http:\/\/arxiv.org\/abs\/1901.04085"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-naacl.97"},{"key":"e_1_3_2_2_16_1","unstructured":"Qwen: An Yang Baosong Yang Beichen Zhang Binyuan Hui Bo Zheng Bowen Yu Chengyuan Li Dayiheng Liu Fei Huang Haoran Wei Huan Lin Jian Yang Jianhong Tu Jianwei Zhang Jianxin Yang Jiaxi Yang Jingren Zhou Junyang Lin Kai Dang Keming Lu Keqin Bao Kexin Yang Le Yu Mei Li Mingfeng Xue Pei Zhang Qin Zhu Rui Men Runji Lin Tianhao Li Tianyi Tang Tingyu Xia Xingzhang Ren Xuancheng Ren Yang Fan Yang Su Yichang Zhang Yu Wan Yuqiong Liu Zeyu Cui Zhenru Zhang and Zihan Qiu. 2025. Qwen2.5 Technical Report. arXiv:2412.15115 [cs.CL] https:\/\/arxiv.org\/abs\/2412.15115"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3331184.3331296"},{"key":"e_1_3_2_2_18_1","unstructured":"John Schulman Filip Wolski Prafulla Dhariwal Alec Radford and Oleg Klimov. 2017. Proximal Policy Optimization Algorithms. arXiv:1707.06347 [cs.LG] https:\/\/arxiv.org\/abs\/1707.06347"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.2402.03300"},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3689031.3696075"},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.923"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNN.1998.712192"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3701716.3715246"},{"key":"e_1_3_2_2_24_1","volume-title":"Chain-of-Thought Prompting Elicits Reasoning in Large Language Models. In Advances in Neural Information Processing Systems 35: Annual Conference on Neural Information Processing Systems 2022","author":"Wei Jason","year":"2022","unstructured":"Jason Wei, Xuezhi Wang, Dale Schuurmans, Maarten Bosma, Brian Ichter, Fei Xia, Ed H. Chi, Quoc V. Le, and Denny Zhou. 2022. Chain-of-Thought Prompting Elicits Reasoning in Large Language Models. In Advances in Neural Information Processing Systems 35: Annual Conference on Neural Information Processing Systems 2022, NeurIPS 2022, New Orleans, LA, USA, November 28 - December 9, 2022, Sanmi Koyejo, S. Mohamed, A. Agarwal, Danielle Belgrave, K. Cho, and A. Oh (Eds.). http:\/\/papers.nips.cc\/paper_files\/paper\/2022\/hash\/9d5609613524ecf4f15af0f7b31abca4-Abstract-Conference.html"},{"key":"e_1_3_2_2_25_1","volume-title":"Second Conference on Language Modeling. https:\/\/openreview.net\/forum?id=Pg0PAvbhGv","author":"Weller Orion","year":"2025","unstructured":"Orion Weller, Kathryn Ricci, Eugene Yang, Andrew Yates, Dawn Lawrie, and Benjamin Van Durme. 2025. Rank1: Test-Time Compute for Reranking in Information Retrieval. In Second Conference on Language Modeling. https:\/\/openreview.net\/forum?id=Pg0PAvbhGv"},{"key":"e_1_3_2_2_26_1","volume-title":"CAPO: Towards Enhancing LLM Reasoning through Generative Credit Assignment. arXiv:2508.02298 [cs.LG] https:\/\/arxiv.org\/abs\/2508.02298","author":"Xie Guofu","year":"2025","unstructured":"Guofu Xie, Yunsheng Shi, Hongtao Tian, Ting Yao, and Xiao Zhang. 2025. CAPO: Towards Enhancing LLM Reasoning through Generative Credit Assignment. arXiv:2508.02298 [cs.LG] https:\/\/arxiv.org\/abs\/2508.02298"},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDE65448.2025.00073"},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.naacl-tutorials.1"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/2939672.2939677"},{"key":"e_1_3_2_2_30_1","unstructured":"Yufeng Yuan Yu Yue Ruofei Zhu Tiantian Fan and Lin Yan. 2025. What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret. arXiv:2503.01491 [cs.LG] https:\/\/arxiv.org\/abs\/2503.01491"},{"key":"e_1_3_2_2_31_1","volume-title":"2nd AI for Math Workshop @ ICML","author":"Yue Yang","year":"2025","unstructured":"Yang Yue, Zhiqi Chen, Rui Lu, Andrew Zhao, Zhaokai Wang, Yang Yue, Shiji Song, and Gao Huang. 2025. Does Reinforcement Learning Really Incentivize Reasoning Capacity in LLMs Beyond the Base Model?. In 2nd AI for Math Workshop @ ICML 2025. https:\/\/openreview.net\/forum?id=upehLVgq1b"},{"key":"e_1_3_2_2_32_1","volume-title":"A Zero-shot Explainable Doctor Ranking Framework with Large Language Models. Big Data Mining and Analytics","author":"Zeng Ziyang","year":"2025","unstructured":"Ziyang Zeng, Dongyuan Li, and Yuqing Yang. 2025. A Zero-shot Explainable Doctor Ranking Framework with Large Language Models. Big Data Mining and Analytics (2025). https:\/\/www.sciopen.com\/article\/10.26599\/BDMA.2025.9020098"},{"key":"e_1_3_2_2_33_1","volume-title":"The Thirteenth International Conference on Learning Representations, ICLR 2025","author":"Zhang Lunjun","year":"2025","unstructured":"Lunjun Zhang, Arian Hosseini, Hritik Bansal, Mehran Kazemi, Aviral Kumar, and Rishabh Agarwal. 2025a. Generative Verifiers: Reward Modeling as Next-Token Prediction. In The Thirteenth International Conference on Learning Representations, ICLR 2025, Singapore, April 24-28, 2025. OpenReview.net. https:\/\/openreview.net\/forum?id=Ccwp4tFEtE"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.emnlp-main.125"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.emnlp-industry.180"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3701716.3715222"},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3748304"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.naacl-short.31"},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3539618.3592047"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"crossref","unstructured":"Shengyao Zhuang Xueguang Ma Bevan Koopman Jimmy Lin and Guido Zuccon. 2025. Rank-R1: Enhancing Reasoning in LLM-based Document Rerankers via Reinforcement Learning. arXiv:2503.06034 [cs.IR] https:\/\/arxiv.org\/abs\/2503.06034","DOI":"10.1145\/3805712.3809961"},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3447548.3467147"}],"event":{"name":"KDD '26: The 32nd ACM SIGKDD Conference on Knowledge Discovery and Data Mining","location":"Jeju Island Republic of Korea","sponsor":["SIGMOD ACM Special Interest Group on Management of Data","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data"]},"container-title":["Proceedings of the 32nd ACM SIGKDD Conference on Knowledge Discovery and Data Mining V.1"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3770854.3783917","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,25]],"date-time":"2026-08-25T20:01:28Z","timestamp":1787688088000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3770854.3783917"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,20]]},"references-count":41,"alternative-id":["10.1145\/3770854.3783917","10.1145\/3770854"],"URL":"https:\/\/doi.org\/10.1145\/3770854.3783917","relation":{},"subject":[],"published":{"date-parts":[[2026,4,20]]},"assertion":[{"value":"2026-04-20","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}