{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,3]],"date-time":"2026-03-03T17:09:33Z","timestamp":1772557773402,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":49,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,2,27]],"date-time":"2023-02-27T00:00:00Z","timestamp":1677456000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["2019897"],"award-info":[{"award-number":["2019897"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"name":"DARPA KAIROS","award":["FA8750-19-2-1004"],"award-info":[{"award-number":["FA8750-19-2-1004"]}]},{"name":"DARPA INCAS","award":["HR001121C0165"],"award-info":[{"award-number":["HR001121C0165"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,2,27]]},"DOI":"10.1145\/3539597.3570475","type":"proceedings-article","created":{"date-parts":[[2023,2,22]],"date-time":"2023-02-22T23:27:00Z","timestamp":1677108420000},"page":"429-437","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":14,"title":["Effective Seed-Guided Topic Discovery by Integrating Multiple Types of Contexts"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0540-6758","authenticated-orcid":false,"given":"Yu","family":"Zhang","sequence":"first","affiliation":[{"name":"University of Illinois Urbana-Champaign, Urbana, IL, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9790-4855","authenticated-orcid":false,"given":"Yunyi","family":"Zhang","sequence":"additional","affiliation":[{"name":"University of Illinois Urbana-Champaign, Urbana, IL, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7757-4294","authenticated-orcid":false,"given":"Martin","family":"Michalski","sequence":"additional","affiliation":[{"name":"University of Illinois Urbana-Champaign, Urbana, IL, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4129-9339","authenticated-orcid":false,"given":"Yucheng","family":"Jiang","sequence":"additional","affiliation":[{"name":"University of Illinois Urbana-Champaign, Urbana, IL, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2554-2888","authenticated-orcid":false,"given":"Yu","family":"Meng","sequence":"additional","affiliation":[{"name":"University of Illinois Urbana-Champaign, Urbana, IL, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3629-2696","authenticated-orcid":false,"given":"Jiawei","family":"Han","sequence":"additional","affiliation":[{"name":"University of Illinois Urbana-Champaign, Urbana, IL, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,2,27]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/1553374.1553378"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P16-2087"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-short.96"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.5555\/944919.944937"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v29i1.9506"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/P15-1077"},{"key":"e_1_3_2_2_7_1","volume-title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In NAACL-HLT'19. 4171--4186.","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In NAACL-HLT'19. 4171--4186."},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00325"},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1037\/h0031619"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00078"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.0307752101"},{"key":"e_1_3_2_2_12_1","volume-title":"BERTopic: Neural topic modeling with a class-based TF-IDF procedure. arXiv preprint arXiv:2203.05794","author":"Grootendorst Maarten","year":"2022","unstructured":"Maarten Grootendorst. 2022. BERTopic: Neural topic modeling with a class-based TF-IDF procedure. arXiv preprint arXiv:2203.05794 (2022)."},{"key":"e_1_3_2_2_13_1","volume-title":"Keyword Assisted Embedded Topic Model. In WSDM'22","author":"Harandizadeh Bahareh","year":"2022","unstructured":"Bahareh Harandizadeh, J Hunter Priniski, and Fred Morstatter. 2022. Keyword Assisted Embedded Topic Model. In WSDM'22. 372--380."},{"key":"e_1_3_2_2_14_1","unstructured":"Alexander Hoyle Pranav Goel Andrew Hian-Cheong Denis Peskov Jordan Boyd-Graber and Philip Resnik. 2021. Is automated topic model evaluation broken? the incoherence of coherence. In NeurIPS'21. 2018--2033."},{"key":"e_1_3_2_2_15_1","volume-title":"EACL'12","author":"Jagarlamudi Jagadeesh","year":"2012","unstructured":"Jagadeesh Jagarlamudi, Hal Daum\u00e9, and Raghavendra Udupa. 2012. Incorporating lexical priors into topic models. In EACL'12. 204--213."},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1321"},{"key":"e_1_3_2_2_17_1","volume-title":"NIPS'09","author":"Lacoste-Julien Simon","year":"2008","unstructured":"Simon Lacoste-Julien, Fei Sha, and Michael I Jordan. 2008. DiscLDA: discriminative learning for dimensionality reduction and classification. In NIPS'09. 897--904."},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/E14-1056"},{"key":"e_1_3_2_2_19_1","volume-title":"ICML'14","author":"Le Quoc","year":"2014","unstructured":"Quoc Le and Tomas Mikolov. 2014. Distributed representations of sentences and documents. In ICML'14. 1188--1196."},{"key":"e_1_3_2_2_20_1","volume-title":"TaxoCom: Topic Taxonomy Completion with Hierarchical Discovery of Novel Topic Clusters. In WWW'22","author":"Lee Dongha","year":"2022","unstructured":"Dongha Lee, Jiaming Shen, SeongKu Kang, Susik Yoon, Jiawei Han, and Hwanjo Yu. 2022. TaxoCom: Topic Taxonomy Completion with Hierarchical Discovery of Novel Topic Clusters. In WWW'22. 2819--2829."},{"key":"e_1_3_2_2_21_1","volume-title":"COLING'16","author":"Li Ximing","year":"2016","unstructured":"Ximing Li, Jinjin Chi, Changchun Li, Jihong Ouyang, and Bo Fu. 2016. Integrating topic modeling with word embeddings by mixtures of vMFs. In COLING'16. 151--160."},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v29i1.9522"},{"key":"e_1_3_2_2_23_1","volume-title":"RoBERTa: A robustly optimized bert pretraining approach. arXiv preprint arXiv:1907.11692","author":"Liu Yinhan","year":"2019","unstructured":"Yinhan Liu, Myle Ott, Naman Goyal, Jingfei Du, Mandar Joshi, Danqi Chen, Omer Levy, Mike Lewis, Luke Zettlemoyer, and Veselin Stoyanov. 2019. RoBERTa: A robustly optimized bert pretraining approach. arXiv preprint arXiv:1907.11692 (2019)."},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/P14-5010"},{"key":"e_1_3_2_2_25_1","volume-title":"NIPS'07","author":"Mcauliffe Jon","year":"2007","unstructured":"Jon Mcauliffe and David Blei. 2007. Supervised topic models. In NIPS'07. 121--128."},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.30"},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3366423.3380278"},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3269206.3271737"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3485447.3512034"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3403242"},{"key":"e_1_3_2_2_31_1","volume-title":"NIPS'13","author":"Mikolov Tomas","year":"2013","unstructured":"Tomas Mikolov, Ilya Sutskever, Kai Chen, Greg S Corrado, and Jeff Dean. 2013. Distributed representations of words and phrases and their compositionality. In NIPS'13. 3111--3119."},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00140"},{"key":"e_1_3_2_2_33_1","unstructured":"Alec Radford Jeff Wu Rewon Child David Luan Dario Amodei and Ilya Sutskever. 2019. Language Models are Unsupervised Multitask Learners."},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4471-2099-5_24"},{"key":"e_1_3_2_2_35_1","volume-title":"Exploiting Cloze-Questions for Few-Shot Text Classification and Natural Language Inference. In EACL'21","author":"Schick Timo","year":"2021","unstructured":"Timo Schick and Hinrich Sch\u00fctze. 2021. Exploiting Cloze-Questions for Few-Shot Text Classification and Natural Language Inference. In EACL'21. 255--269."},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2012.6289079"},{"key":"e_1_3_2_2_37_1","volume-title":"Neural Machine Translation of Rare Words with Subword Units. In ACL'16","author":"Sennrich Rico","year":"2016","unstructured":"Rico Sennrich, Barry Haddow, and Alexandra Birch. 2016. Neural Machine Translation of Rare Words with Subword Units. In ACL'16. 1715--1725."},{"key":"e_1_3_2_2_38_1","first-page":"1825","article-title":"Automated phrase mining from massive text corpora","volume":"30","author":"Shang Jingbo","year":"2018","unstructured":"Jingbo Shang, Jialu Liu, Meng Jiang, Xiang Ren, Clare R Voss, and Jiawei Han. 2018. Automated phrase mining from massive text corpora. IEEE TKDE, Vol. 30, 10 (2018), 1825--1837.","journal-title":"IEEE TKDE"},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.135"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/2783258.2783307"},{"key":"e_1_3_2_2_41_1","first-page":"74","article-title":"Multi-Dimensional, Phrase-Based Summarization in Text Cubes","volume":"39","author":"Tao Fangbo","year":"2016","unstructured":"Fangbo Tao, Honglei Zhuang, Chi Wang Yu, Qi Wang, Taylor Cassidy, Lance M Kaplan, Clare R Voss, and Jiawei Han. 2016. Multi-Dimensional, Phrase-Based Summarization in Text Cubes. IEEE Data Eng. Bull., Vol. 39, 3 (2016), 74--84.","journal-title":"IEEE Data Eng. Bull."},{"key":"e_1_3_2_2_42_1","volume-title":"Topic modeling with contextualized word representation clusters. arXiv preprint arXiv:2010.12626","author":"Thompson Laure","year":"2020","unstructured":"Laure Thompson and David Mimno. 2020. Topic modeling with contextualized word representation clusters. arXiv preprint arXiv:2010.12626 (2020)."},{"key":"e_1_3_2_2_43_1","volume-title":"NIPS'17","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. In NIPS'17. 5998--6008."},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/3097983.3098009"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2017\/588"},{"key":"e_1_3_2_2_46_1","volume-title":"MotifClass: Weakly Supervised Text Classification with Higher-order Metadata Information. In WSDM'22","author":"Zhang Yu","year":"2022","unstructured":"Yu Zhang, Shweta Garg, Yu Meng, Xiusi Chen, and Jiawei Han. 2022b. MotifClass: Weakly Supervised Text Classification with Higher-order Metadata Information. In WSDM'22. 1357--1367."},{"key":"e_1_3_2_2_47_1","volume-title":"Seed-Guided Topic Discovery with Out-of-Vocabulary Seeds. In NAACL'22","author":"Zhang Yu","year":"2022","unstructured":"Yu Zhang, Yu Meng, Xuan Wang, Sheng Wang, and Jiawei Han. 2022c. Seed-Guided Topic Discovery with Out-of-Vocabulary Seeds. In NAACL'22. 279--290."},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDM.2019.00098"},{"key":"e_1_3_2_2_49_1","volume-title":"An Empirical Study on Clustering with Contextual Embeddings for Topics. In NAACL'22","author":"Zhang Zihan","year":"2022","unstructured":"Zihan Zhang, Meng Fang, Ling Chen, and Mohammad-Reza Namazi-Rad. 2022a. Is Neural Topic Modelling Better than Clustering? An Empirical Study on Clustering with Contextual Embeddings for Topics. In NAACL'22. 3886--3993."}],"event":{"name":"WSDM '23: The Sixteenth ACM International Conference on Web Search and Data Mining","location":"Singapore Singapore","acronym":"WSDM '23","sponsor":["SIGMOD ACM Special Interest Group on Management of Data","SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data","SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the Sixteenth ACM International Conference on Web Search and Data Mining"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3539597.3570475","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3539597.3570475","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3539597.3570475","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T19:02:15Z","timestamp":1750186935000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3539597.3570475"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,2,27]]},"references-count":49,"alternative-id":["10.1145\/3539597.3570475","10.1145\/3539597"],"URL":"https:\/\/doi.org\/10.1145\/3539597.3570475","relation":{},"subject":[],"published":{"date-parts":[[2023,2,27]]},"assertion":[{"value":"2023-02-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}