{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,8]],"date-time":"2025-09-08T06:54:40Z","timestamp":1757314480914,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":47,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,8,24]],"date-time":"2024-08-24T00:00:00Z","timestamp":1724457600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,8,25]]},"DOI":"10.1145\/3637528.3671649","type":"proceedings-article","created":{"date-parts":[[2024,8,25]],"date-time":"2024-08-25T04:55:12Z","timestamp":1724561712000},"page":"5723-5730","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Multi-Task Neural Linear Bandit for Exploration in Recommender Systems"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-9207-1719","authenticated-orcid":false,"given":"Yi","family":"Su","sequence":"first","affiliation":[{"name":"Google Deepmind, Mountain View, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-7696-7732","authenticated-orcid":false,"given":"Haokai","family":"Lu","sequence":"additional","affiliation":[{"name":"Google Deepmind, Mountain View, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3849-5523","authenticated-orcid":false,"given":"Yuening","family":"Li","sequence":"additional","affiliation":[{"name":"Google, Mountain View, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-1248-1012","authenticated-orcid":false,"given":"Liang","family":"Liu","sequence":"additional","affiliation":[{"name":"Google, Mountain View, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-7545-4410","authenticated-orcid":false,"given":"Shuchao","family":"Bi","sequence":"additional","affiliation":[{"name":"Google, Mountain View, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3230-5338","authenticated-orcid":false,"given":"Ed H.","family":"Chi","sequence":"additional","affiliation":[{"name":"Google Deepmind, Mountain View, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7342-9022","authenticated-orcid":false,"given":"Minmin","family":"Chen","sequence":"additional","affiliation":[{"name":"Google Deepmind, Mountain View, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,8,24]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"International conference on machine learning. PMLR, 127--135","author":"Agrawal Shipra","year":"2013","unstructured":"Shipra Agrawal and Navin Goyal. 2013. Thompson sampling for contextual bandits with linear payoffs. In International conference on machine learning. PMLR, 127--135."},{"key":"e_1_3_2_1_2_1","volume-title":"International Conference on Machine Learning. PMLR, 367--376","author":"Arora Sanjeev","year":"2020","unstructured":"Sanjeev Arora, Simon Du, Sham Kakade, Yuping Luo, and Nikunj Saunshi. 2020. Provable representation learning for imitation learning via bi-level optimization. In International Conference on Machine Learning. PMLR, 367--376."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/2792838.2800196"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/3383313.3412217"},{"key":"e_1_3_2_1_5_1","volume-title":"Representation learning: A review and new perspectives","author":"Bengio Yoshua","year":"2013","unstructured":"Yoshua Bengio, Aaron Courville, and Pascal Vincent. 2013. Representation learning: A review and new perspectives. IEEE transactions on pattern analysis and machine intelligence, Vol. 35, 8 (2013), 1798--1828."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3460231.3474601"},{"key":"e_1_3_2_1_7_1","volume-title":"Values of User Exploration in Recommender Systems. In Fifteenth ACM Conference on Recommender Systems. 85--95","author":"Chen Minmin","year":"2021","unstructured":"Minmin Chen, Yuyan Wang, Can Xu, Ya Le, Mohit Sharma, Lee Richardson, Su-Lin Wu, and Ed Chi. 2021. Values of User Exploration in Recommender Systems. In Fifteenth ACM Conference on Recommender Systems. 85--95."},{"key":"e_1_3_2_1_8_1","volume-title":"Proceedings of the Fourteenth International Conference on Artificial Intelligence and Statistics. JMLR Workshop and Conference Proceedings, 208--214","author":"Chu Wei","year":"2011","unstructured":"Wei Chu, Lihong Li, Lev Reyzin, and Robert Schapire. 2011. Contextual bandits with linear payoff functions. In Proceedings of the Fourteenth International Conference on Artificial Intelligence and Statistics. JMLR Workshop and Conference Proceedings, 208--214."},{"key":"e_1_3_2_1_9_1","volume-title":"Multi-task learning with deep neural networks: A survey. arXiv preprint arXiv:2009.09796","author":"Crawshaw Michael","year":"2020","unstructured":"Michael Crawshaw. 2020. Multi-task learning with deep neural networks: A survey. arXiv preprint arXiv:2009.09796 (2020)."},{"key":"e_1_3_2_1_10_1","volume-title":"Multi-task learning for contextual bandits. Advances in neural information processing systems","author":"Deshmukh Aniket Anand","year":"2017","unstructured":"Aniket Anand Deshmukh, Urun Dogan, and Clay Scott. 2017. Multi-task learning for contextual bandits. Advances in neural information processing systems, Vol. 30 (2017)."},{"key":"e_1_3_2_1_11_1","volume-title":"The Eleventh International Conference on Learning Representations.","author":"Do Virginie","year":"2022","unstructured":"Virginie Do, Elvis Dohmatob, Matteo Pirotta, Alessandro Lazaric, and Nicolas Usunier. 2022. Contextual bandits with concave rewards, and an application to fair ranking. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_2_1_12_1","unstructured":"Audrey Durand Charis Achilleos Demetris Iacovides Katerina Strati Georgios D Mitsis and Joelle Pineau. 2018. Contextual bandits for adapting treatment in a mouse model of de novo carcinogenesis. In Machine learning for healthcare conference. PMLR 67--82."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3397271.3401431"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3397271.3401230"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3460231.3474248"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3306618.3314288"},{"key":"e_1_3_2_1_17_1","volume-title":"International Conference on Learning Representations.","author":"Joachims Thorsten","year":"2018","unstructured":"Thorsten Joachims, Adith Swaminathan, and Maarten De Rijke. 2018. Deep learning with logged bandit feedback. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/MC.2009.263"},{"key":"e_1_3_2_1_19_1","volume-title":"The epoch-greedy algorithm for multi-armed bandits with side information. Advances in neural information processing systems","author":"Langford John","year":"2007","unstructured":"John Langford and Tong Zhang. 2007. The epoch-greedy algorithm for multi-armed bandits with side information. Advances in neural information processing systems, Vol. 20 (2007)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33014189"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/1772690.1772758"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3298689.3346998"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.12028"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3240323.3240365"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3220007"},{"key":"e_1_3_2_1_26_1","volume-title":"Romil Shah, Simeng Qu, Gaurav Bang, and Brad Schumitsch.","author":"Mahajan Khushhall Chandra","year":"2023","unstructured":"Khushhall Chandra Mahajan, Amey Porobo Dharwadker, Romil Shah, Simeng Qu, Gaurav Bang, and Brad Schumitsch. 2023. PIE: Personalized Interest Exploration for Large-Scale Recommender Systems. arXiv preprint arXiv:2304.06844 (2023)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3240323.3240354"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3403374"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1287\/opre.2019.1918"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1287\/mksc.2018.1129"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3403359"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/2365952.2365962"},{"key":"e_1_3_2_1_33_1","volume-title":"Deep bayesian bandits showdown: An empirical comparison of bayesian deep networks for thompson sampling. arXiv preprint arXiv:1802.09127","author":"Riquelme Carlos","year":"2018","unstructured":"Carlos Riquelme, George Tucker, and Jasper Snoek. 2018. Deep bayesian bandits showdown: An empirical comparison of bayesian deep networks for thompson sampling. arXiv preprint arXiv:1802.09127 (2018)."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/371920.372071"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/2645710.2645751"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3159652.3159700"},{"key":"e_1_3_2_1_37_1","volume-title":"Deconvolving feedback loops in recommender systems. Advances in neural information processing systems","author":"Sinha Ayan","year":"2016","unstructured":"Ayan Sinha, David F Gleich, and Karthik Ramani. 2016. Deconvolving feedback loops in recommender systems. Advances in neural information processing systems, Vol. 29 (2016)."},{"key":"e_1_3_2_1_38_1","volume-title":"NIPS2014 workshop on transfer and multi-task learning: theory meets practice.","author":"Soare Marta","year":"2014","unstructured":"Marta Soare, Ouais Alsharif, Alessandro Lazaric, and Joelle Pineau. 2014. Multi-task linear bandits. In NIPS2014 workshop on transfer and multi-task learning: theory meets practice."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3488560.3498459"},{"key":"e_1_3_2_1_40_1","volume-title":"Liang Liu, Yuening Li, Haokai Lu, Benjamin Lipshitz, Sriraj Badam, Lukasz Heldt, Shuchao Bi, et al.","author":"Su Yi","year":"2023","unstructured":"Yi Su, Xiangyu Wang, Elaine Ya Le, Liang Liu, Yuening Li, Haokai Lu, Benjamin Lipshitz, Sriraj Badam, Lukasz Heldt, Shuchao Bi, et al. 2023. Value of Exploration: Measurements, Findings and Algorithms. arXiv preprint arXiv:2305.07764 (2023)."},{"key":"e_1_3_2_1_41_1","volume-title":"International Conference on Artificial Intelligence and Statistics. PMLR, 1673--1681","author":"Turgay Eralp","year":"2018","unstructured":"Eralp Turgay, Doruk Oner, and Cem Tekin. 2018. Multi-objective contextual bandit problem with similarity information. In International Conference on Artificial Intelligence and Statistics. PMLR, 1673--1681."},{"key":"e_1_3_2_1_42_1","volume-title":"Dropoutnet: Addressing cold start in recommender systems. Advances in neural information processing systems","author":"Volkovs Maksims","year":"2017","unstructured":"Maksims Volkovs, Guangwei Yu, and Tomi Poutanen. 2017. Dropoutnet: Addressing cold start in recommender systems. Advances in neural information processing systems, Vol. 30 (2017)."},{"key":"e_1_3_2_1_43_1","volume-title":"Impact of representation learning in linear bandits. arXiv preprint arXiv:2010.06531","author":"Yang Jiaqi","year":"2020","unstructured":"Jiaqi Yang, Wei Hu, Jason D Lee, and Simon S Du. 2020. Impact of representation learning in linear bandits. arXiv preprint arXiv:2010.06531 (2020)."},{"key":"e_1_3_2_1_44_1","volume-title":"Neural thompson sampling. arXiv preprint arXiv:2010.00827","author":"Zhang Weitong","year":"2020","unstructured":"Weitong Zhang, Dongruo Zhou, Lihong Li, and Quanquan Gu. 2020. Neural thompson sampling. arXiv preprint arXiv:2010.00827 (2020)."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3298689.3346997"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2021.11.041"},{"key":"e_1_3_2_1_47_1","volume-title":"International Conference on Machine Learning. PMLR, 11492--11502","author":"Zhou Dongruo","year":"2020","unstructured":"Dongruo Zhou, Lihong Li, and Quanquan Gu. 2020. Neural contextual bandits with ucb-based exploration. In International Conference on Machine Learning. PMLR, 11492--11502."}],"event":{"name":"KDD '24: The 30th ACM SIGKDD Conference on Knowledge Discovery and Data Mining","sponsor":["SIGMOD ACM Special Interest Group on Management of Data","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data"],"location":"Barcelona Spain","acronym":"KDD '24"},"container-title":["Proceedings of the 30th ACM SIGKDD Conference on Knowledge Discovery and Data Mining"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3637528.3671649","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3637528.3671649","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T00:06:00Z","timestamp":1750291560000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3637528.3671649"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,8,24]]},"references-count":47,"alternative-id":["10.1145\/3637528.3671649","10.1145\/3637528"],"URL":"https:\/\/doi.org\/10.1145\/3637528.3671649","relation":{},"subject":[],"published":{"date-parts":[[2024,8,24]]},"assertion":[{"value":"2024-08-24","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}