{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,20]],"date-time":"2026-06-20T03:47:55Z","timestamp":1781927275359,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":55,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,8,14]],"date-time":"2022-08-14T00:00:00Z","timestamp":1660435200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"NSFC","award":["61836013; 61825602"],"award-info":[{"award-number":["61836013; 61825602"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,8,14]]},"DOI":"10.1145\/3534678.3539210","type":"proceedings-article","created":{"date-parts":[[2022,8,12]],"date-time":"2022-08-12T19:06:41Z","timestamp":1660331201000},"page":"3418-3428","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":26,"title":["OAG-BERT: Towards a Unified Backbone Language Model for Academic Knowledge Services"],"prefix":"10.1145","author":[{"given":"Xiao","family":"Liu","sequence":"first","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Da","family":"Yin","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jingnan","family":"Zheng","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xingjian","family":"Zhang","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Peng","family":"Zhang","sequence":"additional","affiliation":[{"name":"ZHIPU AI, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hongxia","family":"Yang","sequence":"additional","affiliation":[{"name":"Alibaba Group, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuxiao","family":"Dong","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jie","family":"Tang","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2022,8,14]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.joi.2017.03.006"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"crossref","unstructured":"Waleed Ammar Dirk Groeneveld Chandra Bhagavatula Iz Beltagy Miles Crawford Doug Downey Jason Dunkelberger Ahmed Elgohary Sergey Feldman Vu Ha et al. 2018. Construction of the Literature Graph in Semantic Scholar. In NAACL. 84--91.","DOI":"10.18653\/v1\/N18-3011"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"crossref","unstructured":"Iz Beltagy Kyle Lo and Arman Cohan. 2019. SciBERT: A Pretrained Language Model for Scientific Text. In EMNLP.","DOI":"10.18653\/v1\/D19-1371"},{"key":"e_1_3_2_1_4_1","volume-title":"Supervised machine learning applied to link prediction in bipartite social networks","author":"Benchettara Nesserine","unstructured":"Nesserine Benchettara, Rushed Kanawati, and Celine Rouveirol. 2010. Supervised machine learning applied to link prediction in bipartite social networks. In ASONAM. IEEE, 326--330."},{"key":"e_1_3_2_1_5_1","unstructured":"Tom Brown Benjamin Mann Nick Ryder Melanie Subbiah Jared D Kaplan Prafulla Dhariwal Arvind Neelakantan Pranav Shyam Girish Sastry Amanda Askell et al. 2020. Language models are few-shot learners. NIPS (2020)."},{"key":"e_1_3_2_1_6_1","unstructured":"Yukuo Cen Zhenyu Hou Yan Wang Qibin Chen Yizhen Luo Xingcheng Yao Aohan Zeng Shiguang Guo Peng Zhang Guohao Dai et al. 2021. CogDL: an extensive toolkit for deep learning on graphs. arXiv preprint arXiv:2103.00959 (2021)."},{"key":"e_1_3_2_1_7_1","volume-title":"CONNA: Addressing Name Disambiguation on The Fly. TKDE","author":"Chen Bo","year":"2020","unstructured":"Bo Chen, Jing Zhang, Jie Tang, Lingfan Cai, Zhaoyu Wang, Shu Zhao, Hong Chen, and Cuiping Li. 2020. CONNA: Addressing Name Disambiguation on The Fly. TKDE (2020)."},{"key":"e_1_3_2_1_8_1","volume-title":"SPECTER: Document-level Representation Learning using Citationinformed Transformers. In ACL. 2270--2282.","author":"Cohan Arman","year":"2020","unstructured":"Arman Cohan, Sergey Feldman, Iz Beltagy, Doug Downey, and Daniel S Weld. 2020. SPECTER: Document-level Representation Learning using Citationinformed Transformers. In ACL. 2270--2282."},{"key":"e_1_3_2_1_9_1","volume-title":"Artificial intelligence is selecting grant reviewers in China. Nature 569, 7756","author":"Cyranoski David","year":"2019","unstructured":"David Cyranoski. 2019. Artificial intelligence is selecting grant reviewers in China. Nature 569, 7756 (2019)."},{"key":"e_1_3_2_1_10_1","volume-title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In NAACL.","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In NAACL."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"crossref","unstructured":"Yuxiao Dong Nitesh V Chawla and Ananthram Swami. 2017. metapath2vec: Scalable representation learning for heterogeneous networks. In SIGKDD.","DOI":"10.1145\/3097983.3098036"},{"key":"e_1_3_2_1_12_1","volume-title":"Link prediction and recommendation across heterogeneous social networks","author":"Dong Yuxiao","unstructured":"Yuxiao Dong, Jie Tang, Sen Wu, Jilei Tian, Nitesh V Chawla, Jinghai Rao, and Huanhuan Cao. 2012. Link prediction and recommendation across heterogeneous social networks. In ICDM. IEEE, 181--190."},{"key":"e_1_3_2_1_13_1","volume-title":"Polar: Attention-based cnn for one-shot personalized article recommendation","author":"Du Zhengxiao","year":"2018","unstructured":"Zhengxiao Du, Jie Tang, and Yuhui Ding. 2018. Polar: Attention-based cnn for one-shot personalized article recommendation. In ECML-PKDD. Springer."},{"key":"e_1_3_2_1_14_1","volume-title":"Active One-shot Personalized Article Recommendation. TKDE","author":"Du Zhengxiao","year":"2019","unstructured":"Zhengxiao Du, Jie Tang, and Yuhui Ding. 2019. POLAR++: Active One-shot Personalized Article Recommendation. TKDE (2019)."},{"key":"e_1_3_2_1_15_1","volume-title":"Don't Stop Pretraining: Adapt Language Models to Domains and Tasks. arXiv preprint arXiv:2004.10964","author":"Gururangan Suchin","year":"2020","unstructured":"Suchin Gururangan, Ana Marasovi?, Swabha Swayamdipta, Kyle Lo, Iz Beltagy, Doug Downey, and Noah A Smith. 2020. Don't Stop Pretraining: Adapt Language Models to Domains and Tasks. arXiv preprint arXiv:2004.10964 (2020)."},{"key":"e_1_3_2_1_16_1","unstructured":"Ziniu Hu Yuxiao Dong Kuansan Wang and Yizhou Sun. 2020. Heterogeneous graph transformer. In WWW."},{"key":"e_1_3_2_1_17_1","volume-title":"Spanbert: Improving pre-training by representing and predicting spans. TACL 8","author":"Joshi Mandar","year":"2020","unstructured":"Mandar Joshi, Danqi Chen, Yinhan Liu, Daniel S Weld, Luke Zettlemoyer, and Omer Levy. 2020. Spanbert: Improving pre-training by representing and predicting spans. TACL 8 (2020)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"crossref","unstructured":"Anshul Kanakia Zhihong Shen Darrin Eide and KuansanWang. 2019. A scalable hybrid research paper recommender system for microsoft academic. In WWW.","DOI":"10.1145\/3308558.3313700"},{"key":"e_1_3_2_1_19_1","unstructured":"Quoc Le and Tomas Mikolov. 2014. Distributed representations of sentences and documents. In ICML. PMLR 1188--1196."},{"key":"e_1_3_2_1_20_1","volume-title":"Chan Ho So, and Jaewoo Kang","author":"Lee Jinhyuk","year":"2020","unstructured":"Jinhyuk Lee, Wonjin Yoon, Sungdong Kim, Donghyeon Kim, Sunkyu Kim, Chan Ho So, and Jaewoo Kang. 2020. BioBERT: a pre-trained biomedical language representation model for biomedical text mining. Bioinformatics 36, 4 (2020)."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i03.5681"},{"key":"e_1_3_2_1_22_1","volume-title":"Self-supervised learning: Generative or contrastive","author":"Liu Xiao","year":"2021","unstructured":"Xiao Liu, Fanjin Zhang, Zhenyu Hou, Li Mian, ZhaoyuWang, Jing Zhang, and Jie Tang. 2021. Self-supervised learning: Generative or contrastive. IEEE Transactions on Knowledge and Data Engineering (2021)."},{"key":"e_1_3_2_1_23_1","volume-title":"GPT understands, too. arXiv preprint arXiv:2103.10385","author":"Liu Xiao","year":"2021","unstructured":"Xiao Liu, Yanan Zheng, Zhengxiao Du, Ming Ding, Yujie Qian, Zhilin Yang, and Jie Tang. 2021. GPT understands, too. arXiv preprint arXiv:2103.10385 (2021)."},{"key":"e_1_3_2_1_24_1","volume-title":"Roberta: A robustly optimized bert pretraining approach. arXiv preprint arXiv:1907.11692","author":"Liu Yinhan","year":"2019","unstructured":"Yinhan Liu, Myle Ott, Naman Goyal, Jingfei Du, Mandar Joshi, Danqi Chen, Omer Levy, Mike Lewis, Luke Zettlemoyer, and Veselin Stoyanov. 2019. Roberta: A robustly optimized bert pretraining approach. arXiv preprint arXiv:1907.11692 (2019)."},{"key":"e_1_3_2_1_25_1","volume-title":"Mark Neumann, Rodney Kinney, and Dan S Weld.","author":"Lo Kyle","year":"2019","unstructured":"Kyle Lo, Lucy Lu Wang, Mark Neumann, Rodney Kinney, and Dan S Weld. 2019. S2orc: The semantic scholar open research corpus. arXiv preprint arXiv:1911.02782 (2019)."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/1149941.1149949"},{"key":"e_1_3_2_1_27_1","volume-title":"andWaleed Ammar","author":"Neumann Mark","year":"2019","unstructured":"Mark Neumann, Daniel King, Iz Beltagy, andWaleed Ammar. 2019. Scispacy: Fast and robust models for biomedical natural language processing. arXiv preprint arXiv:1902.07669 (2019)."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"crossref","unstructured":"Vahed Qazvinian and Dragomir Radev. 2008. Scientific Paper Summarization Using Citation Summary Networks. In COLING. 689--696.","DOI":"10.3115\/1599081.1599168"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2018\/534"},{"key":"e_1_3_2_1_30_1","volume-title":"Language models are unsupervised multitask learners. OpenAI blog 1, 8","author":"Radford Alec","year":"2019","unstructured":"Alec Radford, Jeffrey Wu, Rewon Child, David Luan, Dario Amodei, and Ilya Sutskever. 2019. Language models are unsupervised multitask learners. OpenAI blog 1, 8 (2019)."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3406703"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"crossref","unstructured":"Nils Reimers and Iryna Gurevych. 2019. Sentence-BERT: Sentence Embeddings using Siamese BERT-Networks. In EMNLP. 3982--3992.","DOI":"10.18653\/v1\/D19-1410"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"crossref","unstructured":"Jiaming Shen Zhihong Shen Chenyan Xiong Chi Wang Kuansan Wang and Jiawei Han. 2020. TaxoExpan: Self-supervised taxonomy expansion with positionenhanced graph neural network. In WWW. 486--497.","DOI":"10.1145\/3366423.3380132"},{"key":"e_1_3_2_1_34_1","volume-title":"A Web-scale system for scientific knowledge exploration. ACL","author":"Shen Zhihong","year":"2018","unstructured":"Zhihong Shen, Hao Ma, and Kuansan Wang. 2018. A Web-scale system for scientific knowledge exploration. ACL (2018), 87."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"crossref","unstructured":"Arnab Sinha Zhihong Shen Yang Song Hao Ma Darrin Eide Bo-June Hsu and Kuansan Wang. 2015. An overview of microsoft academic service (mas) and applications. In WWW.","DOI":"10.1145\/2740908.2742839"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"crossref","unstructured":"Kazunari Sugiyama and Min-Yen Kan. 2010. Scholarly paper recommendation via user's recent research interests. In JCDL. 29--38.","DOI":"10.1145\/1816123.1816129"},{"key":"e_1_3_2_1_37_1","volume-title":"Bo Wang, and Jing Zhang.","author":"Tang Jie","year":"2011","unstructured":"Jie Tang, Alvis CM Fong, Bo Wang, and Jing Zhang. 2011. A unified probabilistic framework for name disambiguation in digital library. TKDE (2011)."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"crossref","unstructured":"Jie Tang Jing Zhang Limin Yao Juanzi Li Li Zhang and Zhong Su. 2008. Arnetminer: extraction and mining of academic social networks. In SIGKDD.","DOI":"10.1145\/1401890.1402008"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1162\/089120103321337458"},{"key":"e_1_3_2_1_40_1","volume-title":"Attention is all you need. arXiv preprint arXiv:1706.03762","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, Lukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. arXiv preprint arXiv:1706.03762 (2017)."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1162\/qss_a_00021"},{"key":"e_1_3_2_1_42_1","unstructured":"Howard D White H Cooper LV Hedges et al. 2009. Scientific communication and literature retrieval. The handbook of research synthesis and meta-analysis 2 (2009) 51--71."},{"key":"e_1_3_2_1_43_1","volume-title":"CareerMap: visualizing career trajectory. Science China Information Sciences","author":"Tang Jie","year":"2018","unstructured":"KanWu, Jie Tang, Zhou Shao, Xinyi Xu, Bo Gao, and Shu Zhao. 2018. CareerMap: visualizing career trajectory. Science China Information Sciences (2018)."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"crossref","unstructured":"Kan Wu Jie Tang and Chenhui Zhang. 2018. Where Have You Been? Inferring Career Trajectory from Academic Social Network.. In IJCAI.","DOI":"10.24963\/ijcai.2018\/499"},{"key":"e_1_3_2_1_45_1","volume-title":"Xlnet: Generalized autoregressive pretraining for language understanding. arXiv preprint arXiv:1906.08237","author":"Yang Zhilin","year":"2019","unstructured":"Zhilin Yang, Zihang Dai, Yiming Yang, Jaime Carbonell, Ruslan Salakhutdinov, and Quoc V Le. 2019. Xlnet: Generalized autoregressive pretraining for language understanding. arXiv preprint arXiv:1906.08237 (2019)."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICMLA.2012.127"},{"key":"e_1_3_2_1_47_1","volume-title":"What's your next move: User activity prediction in location-based social networks","author":"Ye Jihang","unstructured":"Jihang Ye, Zhe Zhu, and Hong Cheng. 2013. What's your next move: User activity prediction in location-based social networks. In SDM. SIAM, 171--179."},{"key":"e_1_3_2_1_48_1","volume-title":"Ming Ding, and Jie Tang.","author":"Yin Da","year":"2021","unstructured":"Da Yin, Weng Lam Tam, Ming Ding, and Jie Tang. 2021. MRT: Tracing the Evolution of Scientific Publications. TKDE (2021)."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/3292500.3330785"},{"key":"e_1_3_2_1_50_1","unstructured":"Fanjin Zhang Jie Tang Xueyi Liu Zhenyu Hou Yuxiao Dong Jing Zhang Xiao Liu Ruobing Xie Kai Zhuang Xu Zhang et al. 2021. Understanding WeChat User Preferences and ?Wow\" Diffusion. TKDE (2021)."},{"key":"e_1_3_2_1_51_1","first-page":"64","article-title":"Multi-Modal Generative Adversarial Network for Short Product Title Generation in Mobile E-Commerce","volume":"2019","author":"Zhang Jian-Guo","year":"2019","unstructured":"Jian-Guo Zhang, Pengcheng Zou, Zhao Li, Yao Wan, Xiuming Pan, Yu Gong, and Philip S Yu. 2019. Multi-Modal Generative Adversarial Network for Short Product Title Generation in Mobile E-Commerce. NAACL HLT 2019 (2019), 64--72.","journal-title":"NAACL HLT"},{"key":"e_1_3_2_1_52_1","volume-title":"Accelerating Training of Transformer- Based Language Models with Progressive Layer Dropping. arXiv preprint arXiv:2010.13369","author":"Zhang Minjia","year":"2020","unstructured":"Minjia Zhang and Yuxiong He. 2020. Accelerating Training of Transformer- Based Language Models with Progressive Layer Dropping. arXiv preprint arXiv:2010.13369 (2020)."},{"key":"e_1_3_2_1_53_1","volume-title":"MATCH: Metadata-Aware Text Classification in A Large Hierarchy. In WWW (WWW '21)","author":"Zhang Yu","year":"2021","unstructured":"Yu Zhang, Zhihong Shen, Yuxiao Dong, Kuansan Wang, and Jiawei Han. 2021. MATCH: Metadata-Aware Text Classification in A Large Hierarchy. In WWW (WWW '21). ACM, New York, NY, USA, 3246--3257."},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"crossref","unstructured":"Yutao Zhang Fanjin Zhang Peiran Yao and Jie Tang. 2018. Name Disambiguation in AMiner: Clustering Maintenance and Human in the Loop.. In SIGKDD.","DOI":"10.1145\/3219819.3219859"},{"key":"e_1_3_2_1_55_1","volume-title":"ERNIE: Enhanced language representation with informative entities. arXiv preprint arXiv:1905.07129","author":"Zhang Zhengyan","year":"2019","unstructured":"Zhengyan Zhang, Xu Han, Zhiyuan Liu, Xin Jiang, Maosong Sun, and Qun Liu. 2019. ERNIE: Enhanced language representation with informative entities. arXiv preprint arXiv:1905.07129 (2019)."}],"event":{"name":"KDD '22: The 28th ACM SIGKDD Conference on Knowledge Discovery and Data Mining","location":"Washington DC USA","acronym":"KDD '22","sponsor":["SIGMOD ACM Special Interest Group on Management of Data","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data"]},"container-title":["Proceedings of the 28th ACM SIGKDD Conference on Knowledge Discovery and Data Mining"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3534678.3539210","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3534678.3539210","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T18:59:58Z","timestamp":1750186798000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3534678.3539210"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,8,14]]},"references-count":55,"alternative-id":["10.1145\/3534678.3539210","10.1145\/3534678"],"URL":"https:\/\/doi.org\/10.1145\/3534678.3539210","relation":{},"subject":[],"published":{"date-parts":[[2022,8,14]]},"assertion":[{"value":"2022-08-14","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}