{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T22:10:30Z","timestamp":1784585430816,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":27,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,8,14]],"date-time":"2022-08-14T00:00:00Z","timestamp":1660435200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"National Natural Science Foundation of China","award":["U1936104"],"award-info":[{"award-number":["U1936104"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,8,14]]},"DOI":"10.1145\/3534678.3539120","type":"proceedings-article","created":{"date-parts":[[2022,8,12]],"date-time":"2022-08-12T19:06:41Z","timestamp":1660331201000},"page":"4215-4225","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":35,"title":["Training Large-Scale News Recommenders with Pretrained Language Models in the Loop"],"prefix":"10.1145","author":[{"given":"Shitao","family":"Xiao","sequence":"first","affiliation":[{"name":"Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zheng","family":"Liu","sequence":"additional","affiliation":[{"name":"Microsoft Research Asia, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yingxia","family":"Shao","sequence":"additional","affiliation":[{"name":"Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tao","family":"Di","sequence":"additional","affiliation":[{"name":"Microsoft, Redmond, WA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bhuvan","family":"Middha","sequence":"additional","affiliation":[{"name":"Microsoft, Redmond, WA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fangzhao","family":"Wu","sequence":"additional","affiliation":[{"name":"Microsoft Research Asia, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xing","family":"Xie","sequence":"additional","affiliation":[{"name":"Microsoft Research Asia, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2022,8,14]]},"reference":[{"key":"e_1_3_2_2_1_1","unstructured":"2020. Microsoft Recommenders. https:\/\/github.com\/microsoft\/recommenders\/"},{"key":"e_1_3_2_2_2_1","unstructured":"Mingxiao An Fangzhao Wu Chuhan Wu Kun Zhang Zheng Liu and Xing Xie. 2019. Neural news recommendation with long-and short-term user representations. In ACL. 336--345."},{"key":"e_1_3_2_2_3_1","unstructured":"Hangbo Bao Li Dong Furu Wei Wenhui Wang Nan Yang Xiaodong Liu Yu Wang Jianfeng Gao Songhao Piao Ming Zhou et al. 2020. Unilmv2: Pseudo- masked language models for unified language model pre-training. In ICML. 642--652."},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"crossref","unstructured":"Paul Covington Jay Adams and Emre Sargin. 2016. Deep neural networks for youtube recommendations. In RecSys. 191--198.","DOI":"10.1145\/2959100.2959190"},{"key":"e_1_3_2_2_5_1","volume-title":"Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805","author":"Devlin Jacob","year":"2018","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2018. Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)."},{"key":"e_1_3_2_2_6_1","volume-title":"SimCSE: Simple Contrastive Learning of Sentence Embeddings. arXiv preprint arXiv:2104.08821","author":"Gao Tianyu","year":"2021","unstructured":"Tianyu Gao, Xingcheng Yao, and Danqi Chen. 2021. SimCSE: Simple Contrastive Learning of Sentence Embeddings. arXiv preprint arXiv:2104.08821 (2021)."},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"crossref","unstructured":"Suyu Ge Chuhan Wu Fangzhao Wu Tao Qi and Yongfeng Huang. 2020. Graph Enhanced Representation Learning for News Recommendation. In WWW. 2863--2869.","DOI":"10.1145\/3366423.3380050"},{"key":"e_1_3_2_2_8_1","volume-title":"Xiaohan Wei, Xing Wang, Yuzhen Huang, Arun Kejariwal, Kannan Ramchandran, and Michael W Mahoney.","author":"Gupta Vipul","year":"2020","unstructured":"Vipul Gupta, Dhruv Choudhary, Ping Tak Peter Tang, Xiaohan Wei, Xing Wang, Yuzhen Huang, Arun Kejariwal, Kannan Ramchandran, and Michael W Mahoney. 2020. Training Recommender Systems at Scale: Communication-Efficient Model and Data Parallelism. arXiv preprint arXiv:2010.08899 (2020)."},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1007\/s41019-019-00113-0"},{"key":"e_1_3_2_2_10_1","unstructured":"Linmei Hu Siyong Xu Chen Li Cheng Yang Chuan Shi Nan Duan Xing Xie and Ming Zhou. 2020. Graph Neural News Recommendation with Unsupervised Preference Disentanglement. In ACL. 4255--4264."},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"crossref","unstructured":"Lihong Li Wei Chu John Langford and Robert E Schapire. 2010. A contextual-bandit approach to personalized news article recommendation. In WWW. 661--670.","DOI":"10.1145\/1772690.1772758"},{"key":"e_1_3_2_2_12_1","volume-title":"Roberta: A robustly optimized bert pretraining approach. arXiv preprint arXiv:1907.11692","author":"Liu Yinhan","year":"2019","unstructured":"Yinhan Liu, Myle Ott, Naman Goyal, Jingfei Du, Mandar Joshi, Danqi Chen, Omer Levy, Mike Lewis, Luke Zettlemoyer, and Veselin Stoyanov. 2019. Roberta: A robustly optimized bert pretraining approach. arXiv preprint arXiv:1907.11692 (2019)."},{"key":"e_1_3_2_2_13_1","unstructured":"Wenhao Lu Jian Jiao and Ruofei Zhang. 2020. TwinBERT: Distilling Knowledge to Twin-Structured Compressed BERT Models for Large-Scale Retrieval. In CIKM. 2645--2652."},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"crossref","unstructured":"Shumpei Okura Yukihiro Tagami Shingo Ono and Akira Tajima. 2017. Embedding-based news recommendation for millions of users. In KDD. 1933--1942.","DOI":"10.1145\/3097983.3098108"},{"key":"e_1_3_2_2_15_1","volume-title":"Efficient transformers: A survey. arXiv preprint arXiv:2009.06732","author":"Tay Yi","year":"2020","unstructured":"Yi Tay, Mostafa Dehghani, Dara Bahri, and Donald Metzler. 2020. Efficient transformers: A survey. arXiv preprint arXiv:2009.06732 (2020)."},{"key":"e_1_3_2_2_16_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N Gomez Lukasz Kaiser and Illia Polosukhin. 2017. Attention is all you need. In Advances in neural information processing systems. 5998--6008."},{"key":"e_1_3_2_2_17_1","volume-title":"Minilm: Deep self-attention distillation for task-agnostic compression of pre-trained transformers. arXiv preprint arXiv:2002.10957","author":"Wang Wenhui","year":"2020","unstructured":"Wenhui Wang, Furu Wei, Li Dong, Hangbo Bao, Nan Yang, and Ming Zhou. 2020. Minilm: Deep self-attention distillation for task-agnostic compression of pre-trained transformers. arXiv preprint arXiv:2002.10957 (2020)."},{"key":"e_1_3_2_2_18_1","volume-title":"Neural news recommendation with attentive multi-view learning. arXiv preprint arXiv:1907.05576","author":"Wu Chuhan","year":"2019","unstructured":"Chuhan Wu, Fangzhao Wu, Mingxiao An, Jianqiang Huang, Yongfeng Huang, and Xing Xie. 2019. Neural news recommendation with attentive multi-view learning. arXiv preprint arXiv:1907.05576 (2019)."},{"key":"e_1_3_2_2_19_1","unstructured":"Chuhan Wu Fangzhao Wu Mingxiao An Jianqiang Huang Yongfeng Huang and Xing Xie. 2019. NPA: neural news recommendation with personalized attention. In KDD. 2576--2584."},{"key":"e_1_3_2_2_20_1","unstructured":"Chuhan Wu Fangzhao Wu Suyu Ge Tao Qi Yongfeng Huang and Xing Xie. 2019. Neural News Recommendation with Multi-Head Self-Attention. In EMNLP. 6389--6394."},{"key":"e_1_3_2_2_21_1","unstructured":"Chuhan Wu Fangzhao Wu Yongfeng Huang and Xing Xie. 2021. User-as-Graph: User Modeling with Heterogeneous Graph Pooling for News Recommendation. In IJCAI."},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"crossref","unstructured":"Chuhan Wu Fangzhao Wu Tao Qi and Yongfeng Huang. 2021. Empowering News Recommendation with Pre-Trained Language Models. In SIGIR. 1652--1656.","DOI":"10.1145\/3404835.3463069"},{"key":"e_1_3_2_2_23_1","volume-title":"et al","author":"Wu Fangzhao","year":"2020","unstructured":"Fangzhao Wu, Ying Qiao, Jiun-Hung Chen, Chuhan Wu, Tao Qi, Jianxun Lian, Danyang Liu, Xing Xie, Jianfeng Gao, Winnie Wu, et al . 2020. Mind: A large-scale dataset for news recommendation. In ACL. 3597--3606."},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.1007\/s41019-020-00135-z"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"crossref","unstructured":"Zichao Yang Diyi Yang Chris Dyer Xiaodong He Alex Smola and Eduard Hovy. 2016. Hierarchical attention networks for document classification. In NAACL. 1480--1489.","DOI":"10.18653\/v1\/N16-1174"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","unstructured":"Rex Ying Ruining He Kaifeng Chen Pong Eksombatchai William L. Hamilton and Jure Leskovec. 2018. Graph Convolutional Neural Networks for Web- Scale Recommender Systems. In KDD. 974--983. https:\/\/doi.org\/10.1145\/3219819.3219890","DOI":"10.1145\/3219819.3219890"},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"crossref","unstructured":"Guorui Zhou Xiaoqiang Zhu Chenru Song Ying Fan Han Zhu Xiao Ma Yanghui Yan Junqi Jin Han Li and Kun Gai. 2018. Deep interest network for click-through rate prediction. In KDD. 1059--1068.","DOI":"10.1145\/3219819.3219823"}],"event":{"name":"KDD '22: The 28th ACM SIGKDD Conference on Knowledge Discovery and Data Mining","location":"Washington DC USA","acronym":"KDD '22","sponsor":["SIGMOD ACM Special Interest Group on Management of Data","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data"]},"container-title":["Proceedings of the 28th ACM SIGKDD Conference on Knowledge Discovery and Data Mining"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3534678.3539120","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3534678.3539120","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T19:02:58Z","timestamp":1750186978000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3534678.3539120"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,8,14]]},"references-count":27,"alternative-id":["10.1145\/3534678.3539120","10.1145\/3534678"],"URL":"https:\/\/doi.org\/10.1145\/3534678.3539120","relation":{},"subject":[],"published":{"date-parts":[[2022,8,14]]},"assertion":[{"value":"2022-08-14","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}