{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T18:09:08Z","timestamp":1784138948647,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":45,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"National Natural Science Foundation of China","award":["62402470"],"award-info":[{"award-number":["62402470"]}]},{"name":"National Natural Science Foundation of China","award":["U24B20180"],"award-info":[{"award-number":["U24B20180"]}]},{"name":"National Natural Science Foundation of China","award":["62525211"],"award-info":[{"award-number":["62525211"]}]},{"name":"Fundamental Research Funds for the Central Universities of China","award":["WK2100000053"],"award-info":[{"award-number":["WK2100000053"]}]},{"name":"Anhui Provincial Natural Science Foundation","award":["2408085QF189"],"award-info":[{"award-number":["2408085QF189"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,20]]},"DOI":"10.1145\/3805712.3809535","type":"proceedings-article","created":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:06:26Z","timestamp":1784135186000},"page":"145-155","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Beyond Static Best-of-N: Bayesian List-wise Alignment for LLM-based Recommendation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-0186-9561","authenticated-orcid":false,"given":"Ruijun","family":"Chen","sequence":"first","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5187-9196","authenticated-orcid":false,"given":"Chongming","family":"Gao","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4752-2629","authenticated-orcid":false,"given":"Jiawei","family":"Chen","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5750-5515","authenticated-orcid":false,"given":"Weiqin","family":"Yang","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8472-7992","authenticated-orcid":false,"given":"Xiangnan","family":"He","sequence":"additional","affiliation":[{"name":"University of Science and Technology of China, Hefei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/3209978.3209985"},{"key":"e_1_3_2_1_2_1","volume-title":"Variational best-of-n alignment. arXiv preprint arXiv:2407.06057","author":"Amini Afra","year":"2024","unstructured":"Afra Amini, Tim Vieira, Elliott Ash, and Ryan Cotterell. 2024. Variational best-of-n alignment. arXiv preprint arXiv:2407.06057 (2024)."},{"key":"e_1_3_2_1_3_1","volume-title":"A Bi-Step Grounding Paradigm for Large Language Models in Recommendation Systems. ACM Transactions on Recommender Systems (TORS)","author":"Bao Keqin","year":"2025","unstructured":"Keqin Bao, Jizhi Zhang, Wenjie Wang, Yang Zhang, Zhengyi Yang, Yanchen Luo, Chong Chen, Fuli Feng, and Qi Tian. 2025. A Bi-Step Grounding Paradigm for Large Language Models in Recommendation Systems. ACM Transactions on Recommender Systems (TORS) (2025)."},{"key":"e_1_3_2_1_4_1","volume-title":"Decoding Matters: Addressing Amplification Bias and Homogeneity Issue for LLM-based Recommendation. EMNLP","author":"Bao Keqin","year":"2024","unstructured":"Keqin Bao, Jizhi Zhang, Yang Zhang, Xinyue Huo, Chong Chen, and Fuli Feng. 2024. Decoding Matters: Addressing Amplification Bias and Homogeneity Issue for LLM-based Recommendation. EMNLP (2024)."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3604915.3608857"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/1273496.1273513"},{"key":"e_1_3_2_1_7_1","volume-title":"Make large language model a better ranker. arXiv preprint arXiv:2403.19181","author":"Chao Wen-Shuo","year":"2024","unstructured":"Wen-Shuo Chao, Zhi Zheng, Hengshu Zhu, and Hao Liu. 2024. Make large language model a better ranker. arXiv preprint arXiv:2403.19181 (2024)."},{"key":"e_1_3_2_1_8_1","volume-title":"Hllm: Enhancing sequential recommendations via hierarchical large language models for item and user modeling. arXiv preprint arXiv:2409.12740","author":"Chen Junyi","year":"2024","unstructured":"Junyi Chen, Lu Chi, Bingyue Peng, and Zehuan Yuan. 2024a. Hllm: Enhancing sequential recommendations via hierarchical large language models for item and user modeling. arXiv preprint arXiv:2409.12740 (2024)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3289600.3290999"},{"key":"e_1_3_2_1_10_1","volume-title":"On Softmax Direct Preference Optimization for Recommendation. In The Thirty-eighth Annual Conference on Neural Information Processing Systems (NeurIPS '24)","author":"Chen Yuxin","year":"2024","unstructured":"Yuxin Chen, Junfei Tan, An Zhang, Zhengyi Yang, Leheng Sheng, Enzhi Zhang, Xiang Wang, and Tat-Seng Chua. 2024b. On Softmax Direct Preference Optimization for Recommendation. In The Thirty-eighth Annual Conference on Neural Information Processing Systems (NeurIPS '24)."},{"key":"e_1_3_2_1_11_1","volume-title":"Onepiece: Bringing context engineering and reasoning to industrial cascade ranking system. arXiv preprint arXiv:2509.18091","author":"Dai Sunhao","year":"2025","unstructured":"Sunhao Dai, Jiakai Tang, Jiahua Wu, Kun Wang, Yuxuan Zhu, Bingjun Chen, Bangyang Hong, Yu Zhao, Cong Fu, Kangle Wu, et al., 2025. Onepiece: Bringing context engineering and reasoning to industrial cascade ranking system. arXiv preprint arXiv:2509.18091 (2025)."},{"key":"e_1_3_2_1_12_1","volume-title":"Raft: Reward ranked finetuning for generative foundation model alignment. arXiv preprint arXiv:2304.06767","author":"Dong Hanze","year":"2023","unstructured":"Hanze Dong, Wei Xiong, Deepanshu Goyal, Yihan Zhang, Winnie Chow, Rui Pan, Shizhe Diao, Jipeng Zhang, Kashun Shum, and Tong Zhang. 2023. Raft: Reward ranked finetuning for generative foundation model alignment. arXiv preprint arXiv:2304.06767 (2023)."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3696410.3714524"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3523227.3546767"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3523227.3546767"},{"key":"e_1_3_2_1_16_1","unstructured":"Aaron Grattafiori Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Alex Vaughan et al. 2024. The llama 3 herd of models. arXiv preprint arXiv:2407.21783 (2024)."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.52202\/079017-0094"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/3038912.3052569"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3589334.3648158"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDM.2018.00035"},{"key":"e_1_3_2_1_21_1","volume-title":"MiniOneRec: An Open-Source Framework for Scaling Generative Recommendation. arXiv preprint arXiv:2510.24431","author":"Kong Xiaoyu","year":"2025","unstructured":"Xiaoyu Kong, Leheng Sheng, Junfei Tan, Yuxin Chen, Jiancan Wu, An Zhang, Xiang Wang, and Xiangnan He. 2025. MiniOneRec: An Open-Source Framework for Scaling Generative Recommendation. arXiv preprint arXiv:2510.24431 (2025)."},{"key":"e_1_3_2_1_22_1","volume-title":"RosePO: Aligning LLM-based Recommenders with Human Values. arXiv preprint arXiv:2410.12519","author":"Liao Jiayi","year":"2024","unstructured":"Jiayi Liao, Xiangnan He, Ruobing Xie, Jiancan Wu, Yancheng Yuan, Xingwu Sun, Zhanhui Kang, and Xiang Wang. 2024a. RosePO: Aligning LLM-based Recommenders with Human Values. arXiv preprint arXiv:2410.12519 (2024)."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3657690"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3678004"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/3726302.3730053"},{"key":"e_1_3_2_1_26_1","volume-title":"Onerec-think: In-text reasoning for generative recommendation. arXiv preprint arXiv:2510.11639","author":"Liu Zhanyu","year":"2025","unstructured":"Zhanyu Liu, Shiyao Wang, Xingmei Wang, Rongzhou Zhang, Jiaxin Deng, Honghui Bao, Jinghao Zhang, Wuchao Li, Pengfei Zheng, Xiangyu Wu, et al., 2025. Onerec-think: In-text reasoning for generative recommendation. arXiv preprint arXiv:2510.11639 (2025)."},{"key":"e_1_3_2_1_27_1","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Rafailov Rafael","year":"2024","unstructured":"Rafael Rafailov, Archit Sharma, Eric Mitchell, Christopher D Manning, Stefano Ermon, and Chelsea Finn. 2024. Direct preference optimization: Your language model is secretly a reward model. Advances in Neural Information Processing Systems, Vol. 36 (2024)."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3589334.3645458"},{"key":"e_1_3_2_1_29_1","volume-title":"Bond: Aligning llms with best-of-n distillation. arXiv preprint arXiv:2407.14622","author":"Sessa Pier Giuseppe","year":"2024","unstructured":"Pier Giuseppe Sessa, Robert Dadashi, L\u00e9onard Hussenot, Johan Ferret, Nino Vieillard, Alexandre Ram\u00e9, Bobak Shariari, Sarah Perrin, Abe Friesen, Geoffrey Cideron, et al., 2024. Bond: Aligning llms with best-of-n distillation. arXiv preprint arXiv:2407.14622 (2024)."},{"key":"e_1_3_2_1_30_1","volume-title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models. ArXiv","author":"Shao Zhihong","year":"2024","unstructured":"Zhihong Shao, Peiyi Wang, Qihao Zhu, Runxin Xu, Jun-Mei Song, Mingchuan Zhang, Y. K. Li, Yu Wu, and Daya Guo. 2024. DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models. ArXiv, Vol. abs\/2402.03300 (2024). https:\/\/api.semanticscholar.org\/CorpusID:267412607"},{"key":"e_1_3_2_1_31_1","volume-title":"Scaling llm test-time compute optimally can be more effective than scaling model parameters. arXiv preprint arXiv:2408.03314","author":"Snell Charlie","year":"2024","unstructured":"Charlie Snell, Jaehoon Lee, Kelvin Xu, and Aviral Kumar. 2024. Scaling llm test-time compute optimally can be more effective than scaling model parameters. arXiv preprint arXiv:2408.03314 (2024)."},{"key":"e_1_3_2_1_32_1","volume-title":"Learning to summarize with human feedback. Advances in neural information processing systems","author":"Stiennon Nisan","year":"2020","unstructured":"Nisan Stiennon, Long Ouyang, Jeffrey Wu, Daniel Ziegler, Ryan Lowe, Chelsea Voss, Alec Radford, Dario Amodei, and Paul F Christiano. 2020. Learning to summarize with human feedback. Advances in neural information processing systems, Vol. 33 (2020), 3008-3021."},{"key":"e_1_3_2_1_33_1","volume-title":"Sequence to sequence learning with neural networks. Advances in neural information processing systems","author":"Sutskever Ilya","year":"2014","unstructured":"Ilya Sutskever, Oriol Vinyals, and Quoc V Le. 2014. Sequence to sequence learning with neural networks. Advances in neural information processing systems, Vol. 27 (2014)."},{"key":"e_1_3_2_1_34_1","volume-title":"Reinforced Preference Optimization for Recommendation. arXiv preprint arXiv:2510.12211","author":"Tan Junfei","year":"2025","unstructured":"Junfei Tan, Yuxin Chen, An Zhang, Junguang Jiang, Bin Liu, Ziru Xu, Han Zhu, Jian Xu, Bo Zheng, and Xiang Wang. 2025. Reinforced Preference Optimization for Recommendation. arXiv preprint arXiv:2510.12211 (2025)."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/3627673.3679569"},{"key":"e_1_3_2_1_36_1","volume-title":"Denny Zhou, et al.","author":"Wei Jason","year":"2022","unstructured":"Jason Wei, Xuezhi Wang, Dale Schuurmans, Maarten Bosma, Fei Xia, Ed Chi, Quoc V Le, Denny Zhou, et al., 2022. Chain-of-thought prompting elicits reasoning in large language models. Advances in neural information processing systems, Vol. 35 (2022), 24824-24837."},{"key":"e_1_3_2_1_37_1","volume-title":"From decoding to meta-generation: Inference-time algorithms for large language models. arXiv preprint arXiv:2406.16838","author":"Welleck Sean","year":"2024","unstructured":"Sean Welleck, Amanda Bertsch, Matthew Finlayson, Hailey Schoelkopf, Alex Xie, Graham Neubig, Ilia Kulikov, and Zaid Harchaoui. 2024. From decoding to meta-generation: Inference-time algorithms for large language models. arXiv preprint arXiv:2406.16838 (2024)."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11280-024-01291-2"},{"key":"e_1_3_2_1_39_1","volume-title":"Inference scaling laws: An empirical analysis of compute-optimal inference for problem-solving with language models. arXiv preprint arXiv:2408.00724","author":"Wu Yangzhen","year":"2024","unstructured":"Yangzhen Wu, Zhiqing Sun, Shanda Li, Sean Welleck, and Yiming Yang. 2024a. Inference scaling laws: An empirical analysis of compute-optimal inference for problem-solving with language models. arXiv preprint arXiv:2408.00724 (2024)."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/3726302.3729933"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3711896.3736866"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3589334.3645537"},{"key":"e_1_3_2_1_43_1","volume-title":"From generation to consumption: Personalized list value estimation for re-ranking. arXiv preprint arXiv:2508.02242","author":"Zhang Kaike","year":"2025","unstructured":"Kaike Zhang, Xiaobei Wang, Xiaoyu Yang, Shuchang Liu, Hailan Yang, Xiang Li, Fei Sun, and Qi Cao. 2025. From generation to consumption: Personalized list value estimation for re-ranking. arXiv preprint arXiv:2508.02242 (2025)."},{"key":"e_1_3_2_1_44_1","volume-title":"Deep reinforcement learning for list-wise recommendations. arXiv preprint arXiv:1801.00209","author":"Zhao Xiangyu","year":"2017","unstructured":"Xiangyu Zhao, Liang Zhang, Long Xia, Zhuoye Ding, Dawei Yin, and Jiliang Tang. 2017. Deep reinforcement learning for list-wise recommendations. arXiv preprint arXiv:1801.00209 (2017)."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/1060745.1060754"}],"event":{"name":"SIGIR '26: The 49th International ACM SIGIR Conference on Research and Development in Information Retrieval","location":"Melbourne VIC Australia","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 49th International ACM SIGIR Conference on Research and Development in Information Retrieval"],"original-title":[],"deposited":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:27:01Z","timestamp":1784136421000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805712.3809535"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"references-count":45,"alternative-id":["10.1145\/3805712.3809535","10.1145\/3805712"],"URL":"https:\/\/doi.org\/10.1145\/3805712.3809535","relation":{},"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"2026-07-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}