{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,13]],"date-time":"2026-04-13T19:31:07Z","timestamp":1776108667736,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":44,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,10,17]],"date-time":"2022-10-17T00:00:00Z","timestamp":1665964800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["IIS-1838200, IIS-2145411, CNS-2124104"],"award-info":[{"award-number":["IIS-1838200, IIS-2145411, CNS-2124104"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000002","name":"National Institutes of Health","doi-asserted-by":"publisher","award":["5R01LM013323-03, 5K01LM012924-03"],"award-info":[{"award-number":["5R01LM013323-03, 5K01LM012924-03"]}],"id":[{"id":"10.13039\/100000002","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,10,17]]},"DOI":"10.1145\/3511808.3557675","type":"proceedings-article","created":{"date-parts":[[2022,10,16]],"date-time":"2022-10-16T01:22:22Z","timestamp":1665883342000},"page":"4470-4474","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["PubMed Author-assigned Keyword Extraction (PubMedAKE) Benchmark"],"prefix":"10.1145","author":[{"given":"Jiasheng","family":"Sheng","sequence":"first","affiliation":[{"name":"Carnegie Mellon University, Pittsburgh, PA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zelalem","family":"Gero","sequence":"additional","affiliation":[{"name":"Microsoft Research, Redmond, WA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Joyce C.","family":"Ho","sequence":"additional","affiliation":[{"name":"Emory University, Atlanta, GA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2022,10,17]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.21105\/joss.01979"},{"key":"e_1_3_2_2_2_1","unstructured":"Shabbir Ahmed and Farzana Mithun. 2004. Word stemming to enhance spam filtering. In CEAS.  Shabbir Ahmed and Farzana Mithun. 2004. Word stemming to enhance spam filtering. In CEAS."},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3308558.3313642"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1093\/bioinformatics\/14.7.600"},{"key":"e_1_3_2_2_5_1","volume-title":"Proceedings of the 5th International Workshop on Semantic Evaluation, 190--193","author":"Samhaa","unstructured":"Samhaa R. El-Beltagy and Ahmed Rafea. 2010. KP-miner: participation in SemEval-2 . In Proceedings of the 5th International Workshop on Semantic Evaluation, 190--193 . Samhaa R. El-Beltagy and Ahmed Rafea. 2010. KP-miner: participation in SemEval-2. In Proceedings of the 5th International Workshop on Semantic Evaluation, 190--193."},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"crossref","unstructured":"D Blake. 1994. Indexing the british heart journal: choice of keywords. British heart journal 71 3 212.  D Blake. 1994. Indexing the british heart journal: choice of keywords. British heart journal 71 3 212.","DOI":"10.1136\/hrt.71.3.212"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/2064696.2064710"},{"key":"e_1_3_2_2_8_1","volume-title":"Proceedings of COLING 2016, the 26th International Conference on Computational Linguistics: System Demonstrations, 69--73","author":"Boudin Florian","year":"2016","unstructured":"Florian Boudin . 2016 . Pke: an open source python-based keyphrase extraction toolkit . In Proceedings of COLING 2016, the 26th International Conference on Computational Linguistics: System Demonstrations, 69--73 . Florian Boudin. 2016. Pke: an open source python-based keyphrase extraction toolkit. In Proceedings of COLING 2016, the 26th International Conference on Computational Linguistics: System Demonstrations, 69--73."},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N18-2105"},{"key":"e_1_3_2_2_10_1","volume-title":"Proceedings of the Sixth International Joint Conference on Natural Language Processing, 543--551","author":"Bougouin Adrien","year":"2013","unstructured":"Adrien Bougouin , Florian Boudin , and B\u00e9atrice Daille . 2013 . TopicRank: graphbased topic ranking for keyphrase extraction . In Proceedings of the Sixth International Joint Conference on Natural Language Processing, 543--551 . Adrien Bougouin, Florian Boudin, and B\u00e9atrice Daille. 2013. TopicRank: graphbased topic ranking for keyphrase extraction. In Proceedings of the Sixth International Joint Conference on Natural Language Processing, 543--551."},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.ins.2019.09.013"},{"key":"e_1_3_2_2_12_1","first-page":"1","article-title":"Pubmed: the bibliographic database","volume":"2","author":"Canese Kathi","year":"2013","unstructured":"Kathi Canese and Sarah Weis . 2013 . Pubmed: the bibliographic database . The NCBI Handbook , 2 , 1 . Kathi Canese and Sarah Weis. 2013. Pubmed: the bibliographic database. The NCBI Handbook, 2, 1.","journal-title":"The NCBI Handbook"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/D14-1150"},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1439"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/D14-1179"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P17-1102"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3383583.3398517"},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.bionlp-1.17"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/BHI50953.2021.9508592"},{"key":"e_1_3_2_2_20_1","volume-title":"Proceedings of the 10th ACM International Conference on Bioinformatics, Computational Biology and Health Informatics (BCB '19)","author":"Gero Zelalem","unstructured":"Zelalem Gero and Joyce C. Ho . 2019. Namedkeys: unsupervised keyphrase extraction for biomedical documents . In Proceedings of the 10th ACM International Conference on Bioinformatics, Computational Biology and Health Informatics (BCB '19) . Niagara Falls, NY, USA, 328--337. Zelalem Gero and Joyce C. Ho. 2019. Namedkeys: unsupervised keyphrase extraction for biomedical documents. In Proceedings of the 10th ACM International Conference on Bioinformatics, Computational Biology and Health Informatics (BCB '19). Niagara Falls, NY, USA, 328--337."},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.5555\/1254866.1254875"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.3115\/1119355.1119383"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/BIBM.2015.7359756"},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"crossref","unstructured":"Sun Kim Lana Yeganova Donald C Comeau W John Wilbur and Zhiyong Lu. 2018. Pubmed phrases an open set of coherent phrases for searching biomedical literature. Scientific data 5 1 1--11.  Sun Kim Lana Yeganova Donald C Comeau W John Wilbur and Zhiyong Lu. 2018. Pubmed phrases an open set of coherent phrases for searching biomedical literature. Scientific data 5 1 1--11.","DOI":"10.1038\/sdata.2018.104"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1093\/bioinformatics\/btz682"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.3115\/1118108.1118117"},{"key":"e_1_3_2_2_27_1","volume-title":"Proceedings of the Ninth International Conference on Language Resources and Evaluation (LREC'14)","author":"Loza Vanessa","year":"2014","unstructured":"Vanessa Loza , Shibamouli Lahiri , Rada Mihalcea , and Po-Hsiang Lai . 2014 . Building a dataset for summarization and keyword extraction from emails . In Proceedings of the Ninth International Conference on Language Resources and Evaluation (LREC'14) , 2441--2446. Vanessa Loza, Shibamouli Lahiri, Rada Mihalcea, and Po-Hsiang Lai. 2014. Building a dataset for summarization and keyword extraction from emails. In Proceedings of the Ninth International Conference on Language Resources and Evaluation (LREC'14), 2441--2446."},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P17-1054"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P17-1054"},{"key":"e_1_3_2_2_30_1","volume-title":"Proceedings of the 2004 Conference on Empirical Methods in Natural Language Processing, 404--411","author":"Mihalcea Rada","year":"2004","unstructured":"Rada Mihalcea and Paul Tarau . 2004 . TextRank: bringing order into text . In Proceedings of the 2004 Conference on Empirical Methods in Natural Language Processing, 404--411 . Rada Mihalcea and Paul Tarau. 2004. TextRank: bringing order into text. In Proceedings of the 2004 Conference on Empirical Methods in Natural Language Processing, 404--411."},{"key":"e_1_3_2_2_32_1","volume-title":"Proceedings of the EACL Hackashop on News Media Content Analysis and Automated Report Generation, 35--44","author":"Piskorski Jakub","year":"2021","unstructured":"Jakub Piskorski , Nicolas Stefanovitch , Guillaume Jacquet , and Aldo Podavini . 2021 . Exploring linguistically-lightweight keyword extraction techniques for indexing news articles in a multilingual set-up . In Proceedings of the EACL Hackashop on News Media Content Analysis and Automated Report Generation, 35--44 . Jakub Piskorski, Nicolas Stefanovitch, Guillaume Jacquet, and Aldo Podavini. 2021. Exploring linguistically-lightweight keyword extraction techniques for indexing news articles in a multilingual set-up. In Proceedings of the EACL Hackashop on News Media Content Analysis and Automated Report Generation, 35--44."},{"key":"e_1_3_2_2_33_1","unstructured":"Martin F. Porter. 2001. Snowball: a language for stemming algorithms. (2001).  Martin F. Porter. 2001. Snowball: a language for stemming algorithms. (2001)."},{"key":"e_1_3_2_2_34_1","volume-title":"Proceedings of the first instructional conference on machine learning number 1.","volume":"242","author":"Juan","unstructured":"Juan Ramos et al. 2003. Using tf-idf to determine word relevance in document queries . In Proceedings of the first instructional conference on machine learning number 1. Vol. 242 , 29--48. Juan Ramos et al. 2003. Using tf-idf to determine word relevance in document queries. In Proceedings of the first instructional conference on machine learning number 1. Vol. 242, 29--48."},{"key":"e_1_3_2_2_35_1","first-page":"328","article-title":"Keyphrase extraction as sequence labeling using contextualized embeddings","volume":"12036","author":"Dhruva Sahrawat","year":"2020","unstructured":"Dhruva Sahrawat et al. 2020 . Keyphrase extraction as sequence labeling using contextualized embeddings . Advances in Information Retrieval , 12036 , 328 . Dhruva Sahrawat et al. 2020. Keyphrase extraction as sequence labeling using contextualized embeddings. Advances in Information Retrieval, 12036, 328.","journal-title":"Advances in Information Retrieval"},{"key":"e_1_3_2_2_36_1","first-page":"392","article-title":"Dake: document-level attention for keyphrase extraction","volume":"12036","author":"Sri Sai Santosh Tokala Yaswanth","year":"2020","unstructured":"Tokala Yaswanth Sri Sai Santosh , Debarshi Kumar Sanyal , Plaban Kumar Bhowmick , and Partha Pratim Das . 2020 . Dake: document-level attention for keyphrase extraction . Advances in Information Retrieval , 12036 , 392 . Tokala Yaswanth Sri Sai Santosh, Debarshi Kumar Sanyal, Plaban Kumar Bhowmick, and Partha Pratim Das. 2020. Dake: document-level attention for keyphrase extraction. Advances in Information Retrieval, 12036, 392.","journal-title":"Advances in Information Retrieval"},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.5120\/10565-5528"},{"key":"e_1_3_2_2_38_1","unstructured":"Alexander Thorsten Schutz etal 2008. Keyphrase extraction from single documents in the open domain exploiting linguistic and statistical methods. M. App. Sc Thesis.  Alexander Thorsten Schutz et al. 2008. Keyphrase extraction from single documents in the open domain exploiting linguistic and statistical methods. M. App. Sc Thesis."},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"crossref","unstructured":"Jasmeet Singh and Vishal Gupta. 2016. Text stemming: approaches applications and challenges. ACM Comput. Surv. 49 3 Article 45 46 pages.  Jasmeet Singh and Vishal Gupta. 2016. Text stemming: approaches applications and challenges. ACM Comput. Surv. 49 3 Article 45 46 pages.","DOI":"10.1145\/2975608"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/2740908.2742730"},{"key":"e_1_3_2_2_41_1","volume-title":"Advances in Neural Information Processing Systems 27: Annual Conference on Neural Information Processing Systems 2014","author":"Sutskever Ilya","year":"2014","unstructured":"Ilya Sutskever , Oriol Vinyals , and Quoc V. Le . 2014. Sequence to sequence learning with neural networks . In Advances in Neural Information Processing Systems 27: Annual Conference on Neural Information Processing Systems 2014 , December 8 --13 2014 , Montreal, Quebec, Canada, 3104--3112. Ilya Sutskever, Oriol Vinyals, and Quoc V. Le. 2014. Sequence to sequence learning with neural networks. In Advances in Neural Information Processing Systems 27: Annual Conference on Neural Information Processing Systems 2014, December 8--13 2014, Montreal, Quebec, Canada, 3104--3112."},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.1186\/s12874-019-0782-0"},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.3115\/1599081.1599203"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1240"},{"key":"e_1_3_2_2_45_1","volume-title":"Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing: System Demonstrations, 38--45","author":"Thomas","unstructured":"Thomas Wolf et al. 2020. Transformers: state-of-the-art natural language processing . In Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing: System Demonstrations, 38--45 . Thomas Wolf et al. 2020. Transformers: state-of-the-art natural language processing. In Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing: System Demonstrations, 38--45."}],"event":{"name":"CIKM '22: The 31st ACM International Conference on Information and Knowledge Management","location":"Atlanta GA USA","acronym":"CIKM '22","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web","SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 31st ACM International Conference on Information &amp; Knowledge Management"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3511808.3557675","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3511808.3557675","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3511808.3557675","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T17:48:49Z","timestamp":1750182529000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3511808.3557675"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,10,17]]},"references-count":44,"alternative-id":["10.1145\/3511808.3557675","10.1145\/3511808"],"URL":"https:\/\/doi.org\/10.1145\/3511808.3557675","relation":{},"subject":[],"published":{"date-parts":[[2022,10,17]]},"assertion":[{"value":"2022-10-17","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}