{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T01:09:01Z","timestamp":1784423341927,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":52,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,2,27]],"date-time":"2023-02-27T00:00:00Z","timestamp":1677456000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/100000001","name":"NSF (National Science Foundation)","doi-asserted-by":"publisher","award":["IIS-1838615,IIS-2007492"],"award-info":[{"award-number":["IIS-1838615,IIS-2007492"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,2,27]]},"DOI":"10.1145\/3539597.3570443","type":"proceedings-article","created":{"date-parts":[[2023,2,22]],"date-time":"2023-02-22T23:27:00Z","timestamp":1677108420000},"page":"222-230","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":32,"title":["Meta Policy Learning for Cold-Start Conversational Recommendation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5707-1112","authenticated-orcid":false,"given":"Zhendong","family":"Chu","sequence":"first","affiliation":[{"name":"University of Virginia, Charlottesville, VA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6524-9195","authenticated-orcid":false,"given":"Hongning","family":"Wang","sequence":"additional","affiliation":[{"name":"University of Virginia, Charlottesville, VA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5358-1211","authenticated-orcid":false,"given":"Yun","family":"Xiao","sequence":"additional","affiliation":[{"name":"JD.COM Silicon Valley Research Center, Mountain View, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2489-200X","authenticated-orcid":false,"given":"Bo","family":"Long","sequence":"additional","affiliation":[{"name":"JD.COM, Beijing, UNK, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3660-651X","authenticated-orcid":false,"given":"Lingfei","family":"Wu","sequence":"additional","affiliation":[{"name":"JD.COM Silicon Valley Research Center, Mountain View, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2023,2,27]]},"reference":[{"key":"e_1_3_2_2_1_1","first-page":"397","article-title":"Using confidence bounds for exploitation-exploration trade-offs","volume":"3","author":"Auer Peter","year":"2002","unstructured":"Peter Auer. 2002. Using confidence bounds for exploitation-exploration trade-offs. Journal of Machine Learning Research, Vol. 3, Nov (2002), 397--422.","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_2_2_2_1","volume-title":"Neural machine translation by jointly learning to align and translate. arXiv preprint arXiv:1409.0473","author":"Bahdanau Dzmitry","year":"2014","unstructured":"Dzmitry Bahdanau, Kyunghyun Cho, and Yoshua Bengio. 2014. Neural machine translation by jointly learning to align and translate. arXiv preprint arXiv:1409.0473 (2014)."},{"key":"e_1_3_2_2_3_1","volume-title":"Proceedings of the 12th International Conference on Music Information Retrieval (ISMIR","author":"Bertin-Mahieux Thierry","year":"2011","unstructured":"Thierry Bertin-Mahieux, Daniel P.W. Ellis, Brian Whitman, and Paul Lamere. 2011. The Million Song Dataset. In Proceedings of the 12th International Conference on Music Information Retrieval (ISMIR 2011)."},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/3269206.3271782"},{"key":"e_1_3_2_2_5_1","volume-title":"Learning to structure long-term dependence for sequential recommendation. arXiv preprint arXiv:2001.11369","author":"Cai Renqin","year":"2020","unstructured":"Renqin Cai, Qinglei Wang, Chong Wang, and Xiaobing Liu. 2020. Learning to structure long-term dependence for sequential recommendation. arXiv preprint arXiv:2001.11369 (2020)."},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3404835.3462832"},{"key":"e_1_3_2_2_7_1","volume-title":"Proceedings of the 39th International Conference on Machine Learning","author":"Chen Dong","year":"2022","unstructured":"Dong Chen, Lingfei Wu, Siliang Tang, Xiao Yun, Bo Long, and Yueting Zhuang. 2022. Robust Meta-learning with Sampling Noise and Label Noise via Eigen-Reptile. Proceedings of the 39th International Conference on Machine Learning (2022)."},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/2939672.2939746"},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"crossref","unstructured":"Zhendong Chu Jing Ma and Hongning Wang. 2021. Learning from Crowds by Modeling Common Confusions.. In AAAI. 5832--5840.","DOI":"10.1609\/aaai.v35i7.16730"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3447548.3467409"},{"key":"e_1_3_2_2_11_1","volume-title":"Unified Conversational Recommendation Policy Learning via Graph-based Reinforcement Learning. arXiv preprint arXiv:2105.09710","author":"Deng Yang","year":"2021","unstructured":"Yang Deng, Yaliang Li, Fei Sun, Bolin Ding, and Wai Lam. 2021. Unified Conversational Recommendation Policy Learning via Graph-based Reinforcement Learning. arXiv preprint arXiv:2105.09710 (2021)."},{"key":"e_1_3_2_2_12_1","volume-title":"Rl2: Fast reinforcement learning via slow reinforcement learning. arXiv preprint arXiv:1611.02779","author":"Duan Yan","year":"2016","unstructured":"Yan Duan, John Schulman, Xi Chen, Peter L Bartlett, Ilya Sutskever, and Pieter Abbeel. 2016. Rl2: Fast reinforcement learning via slow reinforcement learning. arXiv preprint arXiv:1611.02779 (2016)."},{"key":"e_1_3_2_2_13_1","unstructured":"Chelsea Finn Pieter Abbeel and Sergey Levine. 2017. Model-agnostic meta-learning for fast adaptation of deep networks. In ICML. PMLR 1126--1135."},{"key":"e_1_3_2_2_14_1","volume-title":"Advances in Neural Information Processing Systems","volume":"29","author":"Garivier Aur\u00e9lien","year":"2016","unstructured":"Aur\u00e9lien Garivier, Tor Lattimore, and Emilie Kaufmann. 2016. On explore-then-commit strategies. Advances in Neural Information Processing Systems, Vol. 29 (2016)."},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3159652.3159687"},{"key":"e_1_3_2_2_16_1","first-page":"1","article-title":"The movielens datasets: History and context","volume":"5","author":"Maxwell Harper F","year":"2015","unstructured":"F Maxwell Harper and Joseph A Konstan. 2015. The movielens datasets: History and context. Acm TIIS, Vol. 5, 4 (2015), 1--19.","journal-title":"Acm TIIS"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/2872427.2883037"},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/3038912.3052569"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.79.8.2554"},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3403089"},{"key":"e_1_3_2_2_21_1","volume-title":"Yee Whye Teh, and Nicolas Heess","author":"Humplik Jan","year":"2019","unstructured":"Jan Humplik, Alexandre Galashov, Leonard Hasenclever, Pedro A Ortega, Yee Whye Teh, and Nicolas Heess. 2019. Meta reinforcement learning as task inference. arXiv preprint arXiv:1905.06424 (2019)."},{"key":"e_1_3_2_2_22_1","volume-title":"Learning adaptive exploration strategies in dynamic environments through informed policy regularization. arXiv preprint arXiv:2005.02934","author":"Kamienny Pierre-Alexandre","year":"2020","unstructured":"Pierre-Alexandre Kamienny, Matteo Pirotta, Alessandro Lazaric, Thibault Lavril, Nicolas Usunier, and Ludovic Denoyer. 2020. Learning adaptive exploration strategies in dynamic environments through informed policy regularization. arXiv preprint arXiv:2005.02934 (2020)."},{"key":"e_1_3_2_2_23_1","unstructured":"Minseok Kim Hwanjun Song Yooju Shin Dongmin Park Kijung Shin and Jae-Gil Lee. 2022. Meta-Learning for Online Update of Recommender Systems. (2022)."},{"key":"e_1_3_2_2_24_1","volume-title":"Bandit algorithms","author":"Lattimore Tor","unstructured":"Tor Lattimore and Csaba Szepesv\u00e1ri. 2020. Bandit algorithms. Cambridge University Press."},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/3292500.3330859"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3336191.3371769"},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3403258"},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/2684822.2685311"},{"key":"e_1_3_2_2_29_1","volume-title":"Proceedings of the 32nd NeurIPS Conference. 9748--9758","author":"Li Raymond","year":"2018","unstructured":"Raymond Li, Samira Kahou, Hannes Schulz, Vincent Michalski, Laurent Charlin, and Chris Pal. 2018. Towards deep conversational recommendations. In Proceedings of the 32nd NeurIPS Conference. 9748--9758."},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3446427","article-title":"Seamlessly unifying attributes and items: Conversational recommendation for cold-start users","volume":"39","author":"Li Shijun","year":"2021","unstructured":"Shijun Li, Wenqiang Lei, Qingyun Wu, Xiangnan He, Peng Jiang, and Tat-Seng Chua. 2021. Seamlessly unifying attributes and items: Conversational recommendation for cold-start users. ACM TOIS, Vol. 39, 4 (2021), 1--29.","journal-title":"ACM TOIS"},{"key":"e_1_3_2_2_31_1","unstructured":"Evan Z Liu Aditi Raghunathan Percy Liang and Chelsea Finn. 2021. Decoupling exploration and exploitation for meta-reinforcement learning without sacrifices. In ICML. PMLR 6925--6935."},{"key":"e_1_3_2_2_32_1","volume-title":"On first-order meta-learning algorithms. arXiv preprint arXiv:1803.02999","author":"Nichol Alex","year":"2018","unstructured":"Alex Nichol, Joshua Achiam, and John Schulman. 2018. On first-order meta-learning algorithms. arXiv preprint arXiv:1803.02999 (2018)."},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDM.2010.127"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/371920.372071"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/3159652.3159700"},{"key":"e_1_3_2_2_36_1","volume-title":"The 41st ACM SIGIR. 235--244.","author":"Sun Yueming","unstructured":"Yueming Sun and Yi Zhang. 2018. Conversational recommender system. In The 41st ACM SIGIR. 235--244."},{"key":"e_1_3_2_2_37_1","volume-title":"2009 42nd Hawaii International Conference on System Sciences. IEEE, 1--9.","author":"T\u00e9tard Franck","year":"2009","unstructured":"Franck T\u00e9tard and Mikael Collan. 2009. Lazy user theory: A dynamic model to understand user selection of products and services. In 2009 42nd Hawaii International Conference on System Sciences. IEEE, 1--9."},{"key":"e_1_3_2_2_38_1","unstructured":"Manasi Vartak Arvind Thiagarajan Conrado Miranda Jeshua Bratman and Hugo Larochelle. 2017. A meta-learning perspective on cold-start recommendations for items. (2017)."},{"key":"e_1_3_2_2_39_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N Gomez ?ukasz Kaiser and Illia Polosukhin. 2017. Attention is all you need. In Advances in neural information processing systems. 5998--6008."},{"key":"e_1_3_2_2_40_1","volume-title":"Multimodal model-agnostic meta-learning via task-aware modulation. arXiv preprint arXiv:1910.13616","author":"Vuorio Risto","year":"2019","unstructured":"Risto Vuorio, Shao-Hua Sun, Hexiang Hu, and Joseph J Lim. 2019. Multimodal model-agnostic meta-learning via task-aware modulation. arXiv preprint arXiv:1910.13616 (2019)."},{"key":"e_1_3_2_2_41_1","volume-title":"Learning to reinforcement learn. arXiv preprint arXiv:1611.05763","author":"Wang Jane X","year":"2016","unstructured":"Jane X Wang, Zeb Kurth-Nelson, Dhruva Tirumala, Hubert Soyer, Joel Z Leibo, Remi Munos, Charles Blundell, Dharshan Kumaran, and Matt Botvinick. 2016. Learning to reinforcement learn. arXiv preprint arXiv:1611.05763 (2016)."},{"key":"e_1_3_2_2_42_1","volume-title":"Simple statistical gradient-following algorithms for connectionist reinforcement learning. Machine learning","author":"Williams Ronald J","year":"1992","unstructured":"Ronald J Williams. 1992. Simple statistical gradient-following algorithms for connectionist reinforcement learning. Machine learning, Vol. 8, 3 (1992), 229--256."},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/3366423.3380285"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/3459637.3482328"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3437963.3441791"},{"key":"e_1_3_2_2_46_1","volume-title":"Reversible action design for combinatorial optimization with reinforcement learning. arXiv preprint arXiv:2102.07210","author":"Yao Fan","year":"2021","unstructured":"Fan Yao, Renqin Cai, and Hongning Wang. 2021. Reversible action design for combinatorial optimization with reinforcement learning. arXiv preprint arXiv:2102.07210 (2021)."},{"key":"e_1_3_2_2_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/3366423.3380148"},{"key":"e_1_3_2_2_48_1","volume-title":"Multi-Choice Questions based Multi-Interest Policy Learning for Conversational Recommendation. arXiv preprint arXiv:2112.11775","author":"Zhang Yiming","year":"2021","unstructured":"Yiming Zhang, Lingfei Wu, Qi Shen, Yitong Pang, Zhihua Wei, Fangli Xu, Bo Long, and Jian Pei. 2021. Multi-Choice Questions based Multi-Interest Policy Learning for Conversational Recommendation. arXiv preprint arXiv:2112.11775 (2021)."},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/3485447.3512152"},{"key":"e_1_3_2_2_50_1","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3219886"},{"key":"e_1_3_2_2_51_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.coling-main.365"},{"key":"e_1_3_2_2_52_1","doi-asserted-by":"publisher","DOI":"10.1145\/3397271.3401181"}],"event":{"name":"WSDM '23: The Sixteenth ACM International Conference on Web Search and Data Mining","location":"Singapore Singapore","acronym":"WSDM '23","sponsor":["SIGMOD ACM Special Interest Group on Management of Data","SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data","SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the Sixteenth ACM International Conference on Web Search and Data Mining"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3539597.3570443","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3539597.3570443","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T19:02:14Z","timestamp":1750186934000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3539597.3570443"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,2,27]]},"references-count":52,"alternative-id":["10.1145\/3539597.3570443","10.1145\/3539597"],"URL":"https:\/\/doi.org\/10.1145\/3539597.3570443","relation":{},"subject":[],"published":{"date-parts":[[2023,2,27]]},"assertion":[{"value":"2023-02-27","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}