{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T18:03:45Z","timestamp":1784138625340,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":50,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,20]]},"DOI":"10.1145\/3805712.3809869","type":"proceedings-article","created":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T14:28:19Z","timestamp":1783693699000},"page":"4298-4304","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["LACONIC: Dense-Level Effectiveness for Scalable Sparse Retrieval via a Two-Phase Training Curriculum"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2370-4487","authenticated-orcid":false,"given":"Zhichao","family":"Xu","sequence":"first","affiliation":[{"name":"University of Utah, Salt Lake City, UT, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6711-0955","authenticated-orcid":false,"given":"Shengyao","family":"Zhuang","sequence":"additional","affiliation":[{"name":"The University of Queensland, Brisbane, QLD, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-0756-8110","authenticated-orcid":false,"given":"Crystina","family":"Zhang","sequence":"additional","affiliation":[{"name":"University of Waterloo, Waterloo, ON, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3430-4910","authenticated-orcid":false,"given":"Xueguang","family":"Ma","sequence":"additional","affiliation":[{"name":"University of Waterloo, Waterloo, ON, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2795-6080","authenticated-orcid":false,"given":"Yijun","family":"Tian","sequence":"additional","affiliation":[{"name":"University of Notre Dame, IN, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-5358-7486","authenticated-orcid":false,"given":"Maitrey","family":"Mehta","sequence":"additional","affiliation":[{"name":"University of Utah, Salt Lake City, UT, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0661-7189","authenticated-orcid":false,"given":"Jimmy","family":"Lin","sequence":"additional","affiliation":[{"name":"University of Waterloo, Waterloo, ON, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0419-6568","authenticated-orcid":false,"given":"Vivek","family":"Srikumar","sequence":"additional","affiliation":[{"name":"University of Utah, Salt Lake City, Utah, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Yang Bai Xiaoguang Li Gang Wang Chaoliang Zhang Lifeng Shang Jun Xu Zhaowei Wang Fangshan Wang and Qun Liu. 2020. SparTerm: Learning term-based sparse representation for fast text retrieval. arXiv:2010.00768 [cs.IR] https:\/\/arxiv.org\/abs\/2010.00768"},{"key":"e_1_3_2_1_2_1","volume-title":"First Conference on Language Modeling. https:\/\/openreview.net\/forum?id=IW1PR7vEBf","author":"BehnamGhader Parishad","year":"2024","unstructured":"Parishad BehnamGhader, Vaibhav Adlakha, Marius Mosbach, Dzmitry Bahdanau, Nicolas Chapados, and Siva Reddy. 2024. LLM2Vec: Large Language Models Are Secretly Powerful Text Encoders. In First Conference on Language Modeling. https:\/\/openreview.net\/forum?id=IW1PR7vEBf"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3657769"},{"key":"e_1_3_2_1_4_1","volume-title":"Cosimo Rulli, and Rossano Venturini.","author":"Bruch Sebastian","year":"2025","unstructured":"Sebastian Bruch, Franco Maria Nardini, Cosimo Rulli, and Rossano Venturini. 2025. Efficient Sketching and Nearest Neighbor Search Algorithms for Sparse Vector Sets. arXiv:2509.24815 [cs.DS] https:\/\/arxiv.org\/abs\/2509.24815"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3331184.3331303"},{"key":"e_1_3_2_1_6_1","unstructured":"Tri Dao. 2023. FlashAttention-2: Faster Attention with Better Parallelism and Work Partitioning. arXiv:2307.08691 [cs.LG] https:\/\/arxiv.org\/abs\/2307.08691"},{"key":"e_1_3_2_1_7_1","volume-title":"Luxical: High-Speed Lexical-Dense Text Embeddings. arXiv:2512.09015 [cs.CL] https:\/\/arxiv.org\/abs\/2512.09015","author":"Luke Merrick AI","year":"2025","unstructured":"DatologyAI, Luke Merrick, Alex Fang, Aldo Carranza, Alvin Deng, Amro Abbas, Brett Larsen, Cody Blakeney, Darren Teh, David Schwab, Fan Pan, Haakon Mongstad, Haoli Yin, Jack Urbanek, Jason Lee, Jason Telanoff, Josh Wills, Kaleigh Mentzer, Paul Burstein, Parth Doshi, Paul Burnstein, Pratyush Maini, Ricardo Monti, Rishabh Adiga, Scott Loftin, Siddharth Joshi, Spandan Das, Tony Jiang, Vineeth Dorna, Zhengping Wang, Bogdan Gaza, Ari Morcos, and Matthew Leavitt. 2025. Luxical: High-Speed Lexical-Dense Text Embeddings. arXiv:2512.09015 [cs.CL] https:\/\/arxiv.org\/abs\/2512.09015"},{"key":"e_1_3_2_1_8_1","unstructured":"Meet Doshi Vishwajeet Kumar Rudra Murthy Vignesh P and Jaydeep Sen. 2024. Mistral-SPLADE: LLMs for better Learned Sparse Retrieval. arXiv:2408.11119 [cs.IR] https:\/\/arxiv.org\/abs\/2408.11119"},{"key":"e_1_3_2_1_9_1","unstructured":"Thibault Formal Carlos Lassance Benjamin Piwowarski and St\u00e9phane Clinchant. 2021a. SPLADE v2: Sparse Lexical and Expansion Model for Information Retrieval. arXiv:2109.10086 [cs.IR] https:\/\/arxiv.org\/abs\/2109.10086"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3477495.3531857"},{"key":"e_1_3_2_1_11_1","unstructured":"Thibault Formal Maxime Louis Herv\u00e9 Dejean and St\u00e9phane Clinchant. 2026. Learning Retrieval Models with Sparse Autoencoders. arXiv:2603.13277 [cs.LG] https:\/\/arxiv.org\/abs\/2603.13277"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3404835.3463098"},{"key":"e_1_3_2_1_13_1","volume-title":"Tevatron: An efficient and flexible toolkit for dense retrieval. arXiv:2203.05765 [cs.IR] https:\/\/arxiv.org\/abs\/2203.05765","author":"Gao Luyu","year":"2022","unstructured":"Luyu Gao, Xueguang Ma, Jimmy Lin, and Jamie Callan. 2022. Tevatron: An efficient and flexible toolkit for dense retrieval. arXiv:2203.05765 [cs.IR] https:\/\/arxiv.org\/abs\/2203.05765"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.repl4nlp-1.31"},{"key":"e_1_3_2_1_15_1","unstructured":"Zhichao Geng Yiwen Wang Dongyu Ru and Yang Yang. 2025. Towards Competitive Search Relevance For Inference-Free Learned Sparse Retrievers. arXiv:2411.04403 [cs.IR] https:\/\/arxiv.org\/abs\/2411.04403"},{"key":"e_1_3_2_1_16_1","unstructured":"Aaron Grattafiori Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Alex Vaughan et al. 2024. The Llama 3 herd of models. arXiv:2407.21783 [cs.CL] https:\/\/arxiv.org\/abs\/2407.21783"},{"key":"e_1_3_2_1_17_1","volume-title":"Susana Guzman, Georgios Mastrapas, Saba Sturua, Bo Wang, Maximilian Werk, Nan Wang, and Han Xiao.","author":"G\u00fcnther Michael","year":"2024","unstructured":"Michael G\u00fcnther, Jackmin Ong, Isabelle Mohr, Alaeddine Abdessalem, Tanguy Abel, Mohammad Kalim Akram, Susana Guzman, Georgios Mastrapas, Saba Sturua, Bo Wang, Maximilian Werk, Nan Wang, and Han Xiao. 2024. Jina Embeddings 2: 8192-Token General-Purpose Text Embeddings for Long Documents. arXiv:2310.19923 [cs.CL] https:\/\/arxiv.org\/abs\/2310.19923"},{"key":"e_1_3_2_1_18_1","unstructured":"Edward J. Hu Yelong Shen Phillip Wallis Zeyuan Allen-Zhu Yuanzhi Li Shean Wang Lu Wang and Weizhu Chen. 2021. LoRA: Low-Rank Adaptation of Large Language Models. arXiv:2106.09685 [cs.CL] https:\/\/arxiv.org\/abs\/2106.09685"},{"key":"e_1_3_2_1_19_1","unstructured":"Samuel Humeau Kurt Shuster Marie-Anne Lachaux and Jason Weston. 2020. Poly-encoders: Architectures and Pre-training Strategies for Fast and Accurate Multi-sentence Scoring. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=SkxgnnNFvH"},{"key":"e_1_3_2_1_20_1","volume-title":"Unsupervised Dense Information Retrieval with Contrastive Learning. Transactions on Machine Learning Research","author":"Izacard Gautier","year":"2022","unstructured":"Gautier Izacard, Mathilde Caron, Lucas Hosseini, Sebastian Riedel, Piotr Bojanowski, Armand Joulin, and Edouard Grave. 2022. Unsupervised Dense Information Retrieval with Contrastive Learning. Transactions on Machine Learning Research (2022). https:\/\/openreview.net\/forum?id=jKN1pXi7b0"},{"key":"e_1_3_2_1_21_1","unstructured":"Jeff Johnson Matthijs Douze and Herv\u00e9 J\u00e9gou. 2017. Billion-scale similarity search with GPUs. arXiv:1702.08734 [cs.CV] https:\/\/arxiv.org\/abs\/1702.08734"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.550"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3477495.3531833"},{"key":"e_1_3_2_1_24_1","unstructured":"Carlos Lassance Herv\u00e9 D\u00e9jean Thibault Formal and St\u00e9phane Clinchant. 2024. SPLADE-v3: New baselines for SPLADE. arXiv:2403.06789 [cs.IR] https:\/\/arxiv.org\/abs\/2403.06789"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1612"},{"key":"e_1_3_2_1_26_1","volume-title":"Making Text Embedders Few-Shot Learners. In The Thirteenth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=wfLuiDjQ0u","author":"Li Chaofan","year":"2025","unstructured":"Chaofan Li, Minghao Qin, Shitao Xiao, Jianlyu Chen, Kun Luo, Defu Lian, Yingxia Shao, and Zheng Liu. 2025. Making Text Embedders Few-Shot Learners. In The Thirteenth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=wfLuiDjQ0u"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-02181-7"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3578337.3605129"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.eacl-main.148"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3726302.3730185"},{"key":"e_1_3_2_1_31_1","volume-title":"Milco: Learned Sparse Retrieval Across Languages via a Multilingual Connector. arXiv:2510.00671 [cs.IR] https:\/\/arxiv.org\/abs\/2510.00671","author":"Nguyen Thong","year":"2025","unstructured":"Thong Nguyen, Yibin Lei, Jia-Huei Ju, Eugene Yang, and Andrew Yates. 2025. Milco: Learned Sparse Retrieval Across Languages via a Multilingual Connector. arXiv:2510.00671 [cs.IR] https:\/\/arxiv.org\/abs\/2510.00671"},{"key":"e_1_3_2_1_32_1","volume-title":"Nomic Embed: Training a Reproducible Long Context Text Embedder. arXiv:2402.01613 [cs.CL] https:\/\/arxiv.org\/abs\/2402.01613","author":"Nussbaum Zach","year":"2025","unstructured":"Zach Nussbaum, John X. Morris, Brandon Duderstadt, and Andriy Mulyar. 2025. Nomic Embed: Training a Reproducible Long Context Text Embedder. arXiv:2402.01613 [cs.CL] https:\/\/arxiv.org\/abs\/2402.01613"},{"key":"e_1_3_2_1_33_1","unstructured":"Aaron van den Oord Yazhe Li and Oriol Vinyals. 2018. Representation learning with contrastive predictive coding. arXiv:1807.03748 [cs.LG] https:\/\/arxiv.org\/abs\/1807.03748"},{"key":"e_1_3_2_1_34_1","volume-title":"International Conference on Learning Representations.","author":"Paria Biswajit","year":"2020","unstructured":"Biswajit Paria, Chih-Kuan Yeh, Ian EH Yen, Ning Xu, Pradeep Ravikumar, and Barnab\u00e1s P\u00f3czos. 2020. Minimizing FLOPs to Learn Efficient Sparse Representations. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_35_1","volume-title":"International Workshop on Knowledge-Enhanced Information Retrieval. Springer, 19-35","author":"Qiao Jingfen","year":"2025","unstructured":"Jingfen Qiao, Thong Nguyen, Evangelos Kanoulas, and Andrew Yates. 2025. Leveraging decoder architectures for learned sparse retrieval. In International Workshop on Knowledge-Enhanced Information Retrieval. Springer, 19-35."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1410"},{"key":"e_1_3_2_1_37_1","first-page":"109","article-title":"Okapi at TREC-3","volume":"109","author":"Robertson Stephen E","year":"1995","unstructured":"Stephen E Robertson, Steve Walker, Susan Jones, Micheline M Hancock-Beaulieu, Mike Gatford, et al., 1995. Okapi at TREC-3. NIST Special Publication Sp, Vol. 109 (1995), 109.","journal-title":"NIST Special Publication Sp"},{"key":"e_1_3_2_1_38_1","unstructured":"Jacob Mitchell Springer Suhas Kotha Daniel Fried Graham Neubig and Aditi Raghunathan. 2024. Repetition improves language model embeddings. arXiv:2402.15449 [cs.CL] https:\/\/arxiv.org\/abs\/2402.15449"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-emnlp.481"},{"key":"e_1_3_2_1_40_1","unstructured":"Liang Wang Nan Yang Xiaolong Huang Binxing Jiao Linjun Yang Daxin Jiang Rangan Majumder and Furu Wei. 2022. Text embeddings by weakly-supervised contrastive pre-training. arXiv:2212.03533 [cs.CL] https:\/\/arxiv.org\/abs\/2212.03533"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.ijcnlp-long.7"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-ijcnlp.33"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.901"},{"key":"e_1_3_2_1_44_1","volume-title":"Jimmy Lin, and Vivek Srikumar.","author":"Xu Zhichao","year":"2026","unstructured":"Zhichao Xu, Fengran Mo, Zhiqi Huang, Crystina Zhang, Puxuan Yu, Bei Wang Phillips, Jimmy Lin, and Vivek Srikumar. 2026. A Survey of Model Architectures in Information Retrieval. Transactions on Machine Learning Research (2026). https:\/\/openreview.net\/forum?id=xAIbTbHRrX Survey Certification."},{"key":"e_1_3_2_1_45_1","unstructured":"Puxuan Yu Luke Merrick Gaurav Nuti and Daniel Campos. 2024. Arctic-Embed 2.0: Multilingual Retrieval Without Compromise. arXiv:2412.04506 [cs.CL] https:\/\/arxiv.org\/abs\/2412.04506"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/3269206.3271800"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/3726302.3730225"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.naacl-main.47"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"crossref","unstructured":"Yanli Zhao Andrew Gu Rohan Varma Liang Luo Chien-Chin Huang Min Xu Less Wright Hamid Shojanazeri Myle Ott Sam Shleifer et al. 2023. PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel. arXiv:2304.11277 [cs.DC] https:\/\/arxiv.org\/abs\/2304.11277","DOI":"10.14778\/3611540.3611569"},{"key":"e_1_3_2_1_50_1","unstructured":"Yutao Zhu Huaying Yuan Shuting Wang Jiongnan Liu Wenhan Liu Chenlong Deng Haonan Chen Zheng Liu Zhicheng Dou and Ji-Rong Wen. 2023. Large language models for information retrieval: A survey. arXiv:2308.07107 [cs.IR] https:\/\/arxiv.org\/abs\/2308.07107"}],"event":{"name":"SIGIR '26: The 49th International ACM SIGIR Conference on Research and Development in Information Retrieval","location":"Melbourne VIC Australia","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 49th International ACM SIGIR Conference on Research and Development in Information Retrieval"],"original-title":[],"deposited":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:11:18Z","timestamp":1784135478000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805712.3809869"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"references-count":50,"alternative-id":["10.1145\/3805712.3809869","10.1145\/3805712"],"URL":"https:\/\/doi.org\/10.1145\/3805712.3809869","relation":{},"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"2026-07-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}