{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T05:03:35Z","timestamp":1784351015599,"version":"3.55.0"},"publisher-location":"Singapore","reference-count":27,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819233908","type":"print"},{"value":"9789819233915","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-3391-5_45","type":"book-chapter","created":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T04:41:45Z","timestamp":1784349705000},"page":"548-558","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["SARE: Soft Alignment Reward for Reinforcement Learning in Generative Recommendation"],"prefix":"10.1007","author":[{"given":"Yulin","family":"Yang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zikang","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Linjing","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dajun","family":"Zeng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"45_CR1","doi-asserted-by":"publisher","first-page":"10299","DOI":"10.52202\/075280-0452","volume":"36","author":"S Rajput","year":"2023","unstructured":"Rajput, S. et al.: Others: recommender systems with generative retrieval. Adv. Neural Inf. Proces. Syst. 36, 10299\u201310315 (2023)","journal-title":"Adv. Neural Inf. Proces. Syst."},{"key":"45_CR2","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3716393","volume":"3","author":"K Bao","year":"2025","unstructured":"Bao, K. et al.: A bi-step grounding paradigm for large language models in recommendation systems. ACM Trans. Recommender Syst. 3, 1\u201327 (2025)","journal-title":"ACM Trans. Recommender Syst."},{"key":"45_CR3","doi-asserted-by":"publisher","first-page":"27463","DOI":"10.52202\/079017-0863","volume":"37","author":"Y Chen","year":"2024","unstructured":"Chen, Y. et al.: On softmax direct preference optimization for recommendation. Adv. Neural Inf. Proces. Syst. 37, 27463\u201327489 (2024)","journal-title":"Adv. Neural Inf. Proces. Syst."},{"key":"45_CR4","doi-asserted-by":"publisher","unstructured":"Gao, C., Chen, R., Yuan, S., Huang, K., Yu, Y., He, X.: SPRec: leveraging self-play to debias preference alignment for large language model-based recommendations. arXiv. (2024). https:\/\/doi.org\/10.48550\/arXiv.2412.09243","DOI":"10.48550\/arXiv.2412.09243"},{"key":"45_CR5","doi-asserted-by":"publisher","unstructured":"Shao, Z. et al.et al.: Deepseekmath: pushing the limits of mathematical reasoning in open language models. arXiv. (2024). https:\/\/doi.org\/10.48550\/arXiv.2402.03300","DOI":"10.48550\/arXiv.2402.03300"},{"key":"45_CR6","doi-asserted-by":"publisher","unstructured":"Guo, D. et al.et al.: Deepseek-r1: incentivizing reasoning capability in llms via reinforcement learning. arXiv. (2025). https:\/\/doi.org\/10.1038\/s41586-025-09422-z","DOI":"10.1038\/s41586-025-09422-z"},{"key":"45_CR7","doi-asserted-by":"publisher","unstructured":"Tan, J. et al.: Reinforced preference optimization for recommendation. arXiv. (2025). https:\/\/doi.org\/10.48550\/arXiv.2510.12211","DOI":"10.48550\/arXiv.2510.12211"},{"key":"45_CR8","doi-asserted-by":"publisher","unstructured":"Bao, K., Zhang, J., Zhang, Y., Huo, X., Chen, C., Feng, F.: Decoding matters: addressing amplification bias and homogeneity issue for llm-based recommendation. arXiv. (2024). https:\/\/doi.org\/10.48550\/arXiv.2406.14900","DOI":"10.48550\/arXiv.2406.14900"},{"key":"45_CR9","doi-asserted-by":"publisher","first-page":"197","DOI":"10.1109\/ICDM.2018.00035","volume-title":"2018 IEEE International Conference on Data Mining (ICDM)","author":"W-C Kang","year":"2018","unstructured":"Kang, W.-C., McAuley, J.: Self-attentive sequential recommendation. In: 2018 IEEE International Conference on Data Mining (ICDM), pp. 197\u2013206. IEEE (2018)"},{"key":"45_CR10","doi-asserted-by":"publisher","first-page":"P10008","DOI":"10.1088\/1742-5468\/2008\/10\/P10008","volume":"2008","author":"VD Blondel","year":"2008","unstructured":"Blondel, V.D., Guillaume, J.-L., Lambiotte, R., Lefebvre, E.: Fast unfolding of communities in large networks. J. Stat. Mech. Theory Exp. 2008, P10008 (2008)","journal-title":"J. Stat. Mech. Theory Exp."},{"key":"45_CR11","doi-asserted-by":"publisher","unstructured":"Hidasi, B., Karatzoglou, A., Baltrunas, L., Tikk, D.: Session-based recommendations with recurrent neural networks. arXiv. (2015). https:\/\/doi.org\/10.48550\/arXiv.1511.06939","DOI":"10.48550\/arXiv.1511.06939"},{"key":"45_CR12","doi-asserted-by":"publisher","first-page":"565","DOI":"10.1145\/3159652.3159656","volume-title":"Proceedings of the Eleventh ACM International Conference on Web Search and Data Mining","author":"J Tang","year":"2018","unstructured":"Tang, J., Wang, K.: Personalized top-n sequential recommendation via convolutional sequence embedding. In: Proceedings of the Eleventh ACM International Conference on Web Search and Data Mining, pp. 565\u2013573 (2018)"},{"key":"45_CR13","doi-asserted-by":"publisher","unstructured":"Team, Q., et al.: Qwen2 technical report. arXiv. (2024). https:\/\/doi.org\/10.48550\/arXiv.2407.10671","DOI":"10.48550\/arXiv.2407.10671"},{"key":"45_CR14","doi-asserted-by":"publisher","unstructured":"Team, G. et al.et al.: Gemma: open models based on gemini research and technology. arXiv. (2024). https:\/\/doi.org\/10.48550\/arXiv.2403.08295","DOI":"10.48550\/arXiv.2403.08295"},{"key":"45_CR15","doi-asserted-by":"publisher","first-page":"60","DOI":"10.1007\/s11280-024-01291-2","volume":"27","author":"L Wu","year":"2024","unstructured":"Wu, L. et al.: Others: a survey on large language models for recommendation. World Wide Web. 27, 60 (2024)","journal-title":"World Wide Web."},{"key":"45_CR16","doi-asserted-by":"publisher","first-page":"3464","DOI":"10.1145\/3589334.3645458","volume-title":"Proceedings of the ACM Web Conference 2024","author":"X Ren","year":"2024","unstructured":"Ren, X. et al.: Representation learning with large language models for recommendation. In: Proceedings of the ACM Web Conference 2024, pp. 3464\u20133475 (2024)"},{"key":"45_CR17","doi-asserted-by":"publisher","first-page":"1435","DOI":"10.1109\/ICDE60146.2024.00118","volume-title":"2024 IEEE 40th International Conference on Data Engineering (ICDE)","author":"B Zheng","year":"2024","unstructured":"Zheng, B. et al.: Adapting large language models by integrating collaborative semantics for recommendation. In: 2024 IEEE 40th International Conference on Data Engineering (ICDE), pp. 1435\u20131448. IEEE (2024)"},{"key":"45_CR18","doi-asserted-by":"publisher","first-page":"1007","DOI":"10.1145\/3604915.3608857","volume-title":"Proceedings of the 17th ACM Conference on Recommender Systems","author":"K Bao","year":"2023","unstructured":"Bao, K., Zhang, J., Zhang, Y., Wang, W., Feng, F., He, X.: Tallrec: an effective and efficient tuning framework to align large language model with recommendation. In: Proceedings of the 17th ACM Conference on Recommender Systems, pp. 1007\u20131014 (2023)"},{"key":"45_CR19","first-page":"1","volume":"43","author":"J Zhang","year":"2025","unstructured":"Zhang, J., Xie, R., Hou, Y., Zhao, X., Lin, L., Wen, J.-R.: Recommendation as instruction following: a large language model empowered recommendation approach. ACM Trans. Inf. Syst. 43, 1\u201337 (2025)","journal-title":"ACM Trans. Inf. Syst."},{"key":"45_CR20","doi-asserted-by":"publisher","first-page":"53728","DOI":"10.52202\/075280-2338","volume":"36","author":"R Rafailov","year":"2023","unstructured":"Rafailov, R., Sharma, A., Mitchell, E., Manning, C.D., Ermon, S., Finn, C.: Direct preference optimization: your language model is secretly a reward model. Adv. Neural Inf. Process. Syst. 36, 53728\u201353741 (2023)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"45_CR21","doi-asserted-by":"publisher","unstructured":"Tajwar, F. et al.: Preference fine-tuning of llms should leverage suboptimal, on-policy data. arXiv. (2024). https:\/\/doi.org\/10.48550\/arXiv.2404.14367","DOI":"10.48550\/arXiv.2404.14367"},{"key":"45_CR22","doi-asserted-by":"publisher","first-page":"27730","DOI":"10.52202\/068431-2011","volume":"35","author":"L Ouyang","year":"2022","unstructured":"Ouyang, L. et al.: Others: training language models to follow instructions with human feedback. Adv. Neural Inf. Process. Syst. 35, 27730\u201327744 (2022)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"45_CR23","doi-asserted-by":"publisher","unstructured":"Ziegler, D.M. et al.: Fine-tuning language models from human preferences. arXiv. (2019). https:\/\/doi.org\/10.48550\/arXiv.1909.08593","DOI":"10.48550\/arXiv.1909.08593"},{"key":"45_CR24","doi-asserted-by":"crossref","unstructured":"Szegedy, C., Vanhoucke, V., Ioffe, S., Shlens, J., Wojna, Z. Rethinking the inception architecture for computer vision. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. IEEE, pp. 2818\u20132826 (2016).","DOI":"10.1109\/CVPR.2016.308"},{"key":"45_CR25","first-page":"18661","volume":"33","author":"P Khosla","year":"2020","unstructured":"Khosla, P. et al.: Supervised contrastive learning. Adv. Neural Inf. Process. Syst. 33, 18661\u201318673 (2020)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"45_CR26","doi-asserted-by":"publisher","first-page":"285","DOI":"10.1145\/3331184.3331203","volume-title":"Proceedings of the 42nd International ACM SIGIR Conference on Research and Development in Information Retrieval","author":"Y Xian","year":"2019","unstructured":"Xian, Y., Fu, Z., Muthukrishnan, S., De Melo, G., Zhang, Y.: Reinforcement knowledge graph reasoning for explainable recommendation. In: Proceedings of the 42nd International ACM SIGIR Conference on Research and Development in Information Retrieval, pp. 285\u2013294 (2019)"},{"key":"45_CR27","doi-asserted-by":"publisher","unstructured":"Christakopoulou, K. et al.et al.: Reward shaping for user satisfaction in a REINFORCE recommender. arXiv. (2022). https:\/\/doi.org\/10.48550\/arXiv.2209.15166","DOI":"10.48550\/arXiv.2209.15166"}],"container-title":["Lecture Notes in Computer Science","Advanced Intelligent Computing Technology and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-3391-5_45","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T04:41:56Z","timestamp":1784349716000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-3391-5_45"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"ISBN":["9789819233908","9789819233915"],"references-count":27,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-3391-5_45","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"19 July 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICIC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Intelligent Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Toronto","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Canada","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icic2026a","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.ic-icc.cn\/2026\/index.htm","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}