{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T02:41:36Z","timestamp":1783737696767,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":25,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,21]],"date-time":"2023-10-21T00:00:00Z","timestamp":1697846400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,21]]},"DOI":"10.1145\/3583780.3615207","type":"proceedings-article","created":{"date-parts":[[2023,10,21]],"date-time":"2023-10-21T07:45:42Z","timestamp":1697874342000},"page":"3865-3869","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["Learning Sparse Lexical Representations Over Specified Vocabularies for Retrieval"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9980-0320","authenticated-orcid":false,"given":"Jeffrey M.","family":"Dudek","sequence":"first","affiliation":[{"name":"Google Research, Mountain View, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-2201-0127","authenticated-orcid":false,"given":"Weize","family":"Kong","sequence":"additional","affiliation":[{"name":"Google Research, Mountain View, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0678-1357","authenticated-orcid":false,"given":"Cheng","family":"Li","sequence":"additional","affiliation":[{"name":"Google Research, Mountain View, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1685-2895","authenticated-orcid":false,"given":"Mingyang","family":"Zhang","sequence":"additional","affiliation":[{"name":"Google Research, Mountain View, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2941-6240","authenticated-orcid":false,"given":"Michael","family":"Bendersky","sequence":"additional","affiliation":[{"name":"Google Research, Mountain View, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2023,10,21]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"TensorFlow: Large-scale machine learning on heterogeneous systems","author":"Abadi Mart\u00edn","year":"2015","unstructured":"Mart\u00edn Abadi , Ashish Agarwal , Paul Barham , Eugene Brevdo , Zhifeng Chen , Craig Citro , Greg S. Corrado , Andy Davis , Jeffrey Dean , Matthieu Devin , Sanjay Ghemawat , Ian Goodfellow , Andrew Harp , Geoffrey Irving , Michael Isard , Yangqing Jia , Rafal Jozefowicz , Lukasz Kaiser , Manjunath Kudlur , Josh Levenberg , Dandelion Man\u00e9 , Rajat Monga , Sherry Moore , Derek Murray , Chris Olah , Mike Schuster , Jonathon Shlens , Benoit Steiner , Ilya Sutskever , Kunal Talwar , Paul Tucker , Vincent Vanhoucke , Vijay Vasudevan , Fernanda Vi\u00e9gas , Oriol Vinyals , Pete Warden , Martin Wattenberg , Martin Wicke , Yuan Yu , and Xiaoqiang Zheng . TensorFlow: Large-scale machine learning on heterogeneous systems , 2015 . URL https:\/\/www.tensorflow.org\/. Software available from tensorflow.org. Mart\u00edn Abadi, Ashish Agarwal, Paul Barham, Eugene Brevdo, Zhifeng Chen, Craig Citro, Greg S. Corrado, Andy Davis, Jeffrey Dean, Matthieu Devin, Sanjay Ghemawat, Ian Goodfellow, Andrew Harp, Geoffrey Irving, Michael Isard, Yangqing Jia, Rafal Jozefowicz, Lukasz Kaiser, Manjunath Kudlur, Josh Levenberg, Dandelion Man\u00e9, Rajat Monga, Sherry Moore, Derek Murray, Chris Olah, Mike Schuster, Jonathon Shlens, Benoit Steiner, Ilya Sutskever, Kunal Talwar, Paul Tucker, Vincent Vanhoucke, Vijay Vasudevan, Fernanda Vi\u00e9gas, Oriol Vinyals, Pete Warden, Martin Wattenberg, Martin Wicke, Yuan Yu, and Xiaoqiang Zheng. TensorFlow: Large-scale machine learning on heterogeneous systems, 2015. URL https:\/\/www.tensorflow.org\/. Software available from tensorflow.org."},{"key":"e_1_3_2_1_2_1","volume-title":"Sparterm: Learning term-based sparse representation for fast text retrieval. arXiv preprint arXiv:2010.00768","author":"Bai Yang","year":"2020","unstructured":"Yang Bai , Xiaoguang Li , Gang Wang , Chaoliang Zhang , Lifeng Shang , Jun Xu , Zhaowei Wang , Fangshan Wang , and Qun Liu . Sparterm: Learning term-based sparse representation for fast text retrieval. arXiv preprint arXiv:2010.00768 , 2020 . Yang Bai, Xiaoguang Li, Gang Wang, Chaoliang Zhang, Lifeng Shang, Jun Xu, Zhaowei Wang, Fangshan Wang, and Qun Liu. Sparterm: Learning term-based sparse representation for fast text retrieval. arXiv preprint arXiv:2010.00768, 2020."},{"key":"e_1_3_2_1_3_1","first-page":"17","volume-title":"SIGIR 2012 workshop on open source information retrieval","author":"Andrzej","year":"2012","unstructured":"Andrzej Bia?ecki, Robert Muir , Grant Ingersoll , and Lucid Imagination . Apache lucene 4 . In SIGIR 2012 workshop on open source information retrieval , page 17 , 2012 . Andrzej Bia?ecki, Robert Muir, Grant Ingersoll, and Lucid Imagination. Apache lucene 4. In SIGIR 2012 workshop on open source information retrieval, page 17, 2012."},{"key":"e_1_3_2_1_4_1","volume-title":"Context-aware sentence\/passage term importance estimation for first stage retrieval. arXiv preprint arXiv:1910.10687","author":"Dai Zhuyun","year":"2019","unstructured":"Zhuyun Dai and Jamie Callan . Context-aware sentence\/passage term importance estimation for first stage retrieval. arXiv preprint arXiv:1910.10687 , 2019 . Zhuyun Dai and Jamie Callan. Context-aware sentence\/passage term importance estimation for first stage retrieval. arXiv preprint arXiv:1910.10687, 2019."},{"key":"e_1_3_2_1_5_1","volume-title":"Splade v2: Sparse lexical and expansion model for information retrieval. arXiv preprint arXiv:2109.10086","author":"Formal Thibault","year":"2021","unstructured":"Thibault Formal , Carlos Lassance , Benjamin Piwowarski , and St\u00e9phane Clinchant . Splade v2: Sparse lexical and expansion model for information retrieval. arXiv preprint arXiv:2109.10086 , 2021 . Thibault Formal, Carlos Lassance, Benjamin Piwowarski, and St\u00e9phane Clinchant. Splade v2: Sparse lexical and expansion model for information retrieval. arXiv preprint arXiv:2109.10086, 2021."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3404835.3463098"},{"key":"e_1_3_2_1_7_1","volume-title":"From distillation to hard negative sampling: Making sparse neural ir models more effective. arXiv preprint arXiv:2205.04733","author":"Formal Thibault","year":"2022","unstructured":"Thibault Formal , Carlos Lassance , Benjamin Piwowarski , and St\u00e9phane Clinchant . From distillation to hard negative sampling: Making sparse neural ir models more effective. arXiv preprint arXiv:2205.04733 , 2022 . Thibault Formal, Carlos Lassance, Benjamin Piwowarski, and St\u00e9phane Clinchant. From distillation to hard negative sampling: Making sparse neural ir models more effective. arXiv preprint arXiv:2205.04733, 2022."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.203"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.naacl-main.241"},{"key":"e_1_3_2_1_10_1","volume-title":"Improving efficient neural ranking models with cross-architecture knowledge distillation. arXiv preprint arXiv:2010.02666","author":"Hofst\u00e4tter Sebastian","year":"2020","unstructured":"Sebastian Hofst\u00e4tter , Sophia Althammer , Michael Schr\u00f6der , Mete Sertkan , and Allan Hanbury . Improving efficient neural ranking models with cross-architecture knowledge distillation. arXiv preprint arXiv:2010.02666 , 2020 . Sebastian Hofst\u00e4tter, Sophia Althammer, Michael Schr\u00f6der, Mete Sertkan, and Allan Hanbury. Improving efficient neural ranking models with cross-architecture knowledge distillation. arXiv preprint arXiv:2010.02666, 2020."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.procs.2019.09.190"},{"key":"e_1_3_2_1_12_1","first-page":"752","volume-title":"Proceedings of the 2nd Conference of the Asia-Pacific Chapter of the Association for Computational Linguistics and the 12th International Joint Conference on Natural Language Processing","author":"Iida Hiroki","year":"2022","unstructured":"Hiroki Iida and Naoaki Okazaki . Unsupervised domain adaptation for sparse retrieval by filling vocabulary and word frequency gaps . In Proceedings of the 2nd Conference of the Asia-Pacific Chapter of the Association for Computational Linguistics and the 12th International Joint Conference on Natural Language Processing , pages 752 -- 765 , 2022 . Hiroki Iida and Naoaki Okazaki. Unsupervised domain adaptation for sparse retrieval by filling vocabulary and word frequency gaps. In Proceedings of the 2nd Conference of the Asia-Pacific Chapter of the Association for Computational Linguistics and the 12th International Joint Conference on Natural Language Processing, pages 752--765, 2022."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3397271.3401075"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3539618.3592065"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3477495.3531833"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3404835.3463030"},{"key":"e_1_3_2_1_17_1","unstructured":"Rodrigo Nogueira Jimmy Lin and AI Epistemic. From doc2query to doctttttquery.  Rodrigo Nogueira Jimmy Lin and AI Epistemic. From doc2query to doctttttquery."},{"key":"e_1_3_2_1_18_1","first-page":"103","volume-title":"TREC","volume":"1","author":"Ogilvie Paul","year":"2001","unstructured":"Paul Ogilvie and James P Callan . Experiments using the lemur toolkit . In TREC , volume 1 , pages 103 -- 108 , 2001 . Paul Ogilvie and James P Callan. Experiments using the lemur toolkit. In TREC, volume 1, pages 103--108, 2001."},{"key":"e_1_3_2_1_19_1","volume-title":"International Conference on Learning Representations","author":"Paria Biswajit","year":"2019","unstructured":"Biswajit Paria , Chih-Kuan Yeh , Ian EH Yen , Ning Xu , Pradeep Ravikumar , and Barnab\u00e1s P\u00f3czos . Minimizing flops to learn efficient sparse representations . In International Conference on Learning Representations , 2019 . Biswajit Paria, Chih-Kuan Yeh, Ian EH Yen, Ning Xu, Pradeep Ravikumar, and Barnab\u00e1s P\u00f3czos. Minimizing flops to learn efficient sparse representations. In International Conference on Learning Representations, 2019."},{"key":"e_1_3_2_1_20_1","volume-title":"Daxiang Dong, Hua Wu, and Haifeng Wang. Rocketqa: An optimized training approach to dense passage retrieval for open-domain question answering. arXiv preprint arXiv:2010.08191","author":"Qu Yingqi","year":"2020","unstructured":"Yingqi Qu , Yuchen Ding , Jing Liu , Kai Liu , Ruiyang Ren , Wayne Xin Zhao , Daxiang Dong, Hua Wu, and Haifeng Wang. Rocketqa: An optimized training approach to dense passage retrieval for open-domain question answering. arXiv preprint arXiv:2010.08191 , 2020 . Yingqi Qu, Yuchen Ding, Jing Liu, Kai Liu, Ruiyang Ren, Wayne Xin Zhao, Daxiang Dong, Hua Wu, and Haifeng Wang. Rocketqa: An optimized training approach to dense passage retrieval for open-domain question answering. arXiv preprint arXiv:2010.08191, 2020."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.5555\/3455716.3455856"},{"key":"e_1_3_2_1_22_1","volume-title":"Colbertv2: Effective and efficient retrieval via lightweight late interaction. arXiv preprint arXiv:2112.01488","author":"Santhanam Keshav","year":"2021","unstructured":"Keshav Santhanam , Omar Khattab , Jon Saad-Falcon , Christopher Potts , and Matei Zaharia . Colbertv2: Effective and efficient retrieval via lightweight late interaction. arXiv preprint arXiv:2112.01488 , 2021 . Keshav Santhanam, Omar Khattab, Jon Saad-Falcon, Christopher Potts, and Matei Zaharia. Colbertv2: Effective and efficient retrieval via lightweight late interaction. arXiv preprint arXiv:2112.01488, 2021."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3442381.3450129"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3269206.3271800"},{"key":"e_1_3_2_1_25_1","volume-title":"Improving semantic matching via multi-task learning in e-commerce. In eCOM@ SIGIR","author":"Zhang Hongchun","year":"2019","unstructured":"Hongchun Zhang , Tianyi Wang , Xiaonan Meng , Yi Hu , and Hao Wang . Improving semantic matching via multi-task learning in e-commerce. In eCOM@ SIGIR , 2019 . Hongchun Zhang, Tianyi Wang, Xiaonan Meng, Yi Hu, and Hao Wang. Improving semantic matching via multi-task learning in e-commerce. In eCOM@ SIGIR, 2019."}],"event":{"name":"CIKM '23: The 32nd ACM International Conference on Information and Knowledge Management","location":"Birmingham United Kingdom","acronym":"CIKM '23","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web","SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 32nd ACM International Conference on Information and Knowledge Management"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3583780.3615207","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3583780.3615207","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T16:36:58Z","timestamp":1750178218000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3583780.3615207"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,21]]},"references-count":25,"alternative-id":["10.1145\/3583780.3615207","10.1145\/3583780"],"URL":"https:\/\/doi.org\/10.1145\/3583780.3615207","relation":{},"subject":[],"published":{"date-parts":[[2023,10,21]]},"assertion":[{"value":"2023-10-21","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}