{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,24]],"date-time":"2026-03-24T14:00:01Z","timestamp":1774360801212,"version":"3.50.1"},"publisher-location":"Cham","reference-count":56,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032212993","type":"print"},{"value":"9783032213006","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-21300-6_7","type":"book-chapter","created":{"date-parts":[[2026,3,24]],"date-time":"2026-03-24T13:08:23Z","timestamp":1774357703000},"page":"104-121","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Contrastive Learning Falls Short: Improving Dense Retrieval with\u00a0Cross-Encoder Listwise Distillation and\u00a0Synthetic Data"],"prefix":"10.1007","author":[{"given":"Manveer Singh","family":"Tamber","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Suleman","family":"Kazi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Vivek","family":"Sourabh","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jimmy","family":"Lin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,3,25]]},"reference":[{"key":"7_CR1","unstructured":"Achiam, J., et\u00a0al.: GPT-4 Technical Report. arXiv:2303.08774 (2023)"},{"issue":"4","key":"7_CR2","doi-asserted-by":"publisher","first-page":"365","DOI":"10.1007\/s10791-022-09411-0","volume":"25","author":"N Arabzadeh","year":"2022","unstructured":"Arabzadeh, N., Vtyurina, A., Yan, X., Clarke, C.L.A.: Shallow pooling for sparse labels. Inf. Retr. 25(4), 365\u2013385 (2022)","journal-title":"Inf. Retr."},{"key":"7_CR3","unstructured":"Bajaj, P., et al.: MS MARCO: A Human Generated MAchine Reading COmprehension Dataset. arXiv:1611.09268v3 (2016)"},{"key":"7_CR4","doi-asserted-by":"publisher","first-page":"384","DOI":"10.1007\/978-3-030-58219-7_26","volume-title":"Experimental IR Meets Multilinguality, Multimodality, and Interaction","author":"A Bondarenko","year":"2020","unstructured":"Bondarenko, A., et al.: Overview of Touch\u00e9 2020: argument retrieval. In: Arampatzis, A., et al. (eds.) Experimental IR Meets Multilinguality, Multimodality, and Interaction, pp. 384\u2013395. Springer International Publishing, Cham (2020)"},{"key":"7_CR5","doi-asserted-by":"crossref","unstructured":"Bonifacio, L., Abonizio, H., Fadaee, M., Nogueira, R.: InPars: Data Augmentation for Information Retrieval using Large Language Models. arXiv:2202.05144 (2022)","DOI":"10.1145\/3477495.3531863"},{"key":"7_CR6","doi-asserted-by":"publisher","first-page":"716","DOI":"10.1007\/978-3-319-30671-1_58","volume-title":"Advances in Information Retrieval","author":"V Boteva","year":"2016","unstructured":"Boteva, V., Gholipour, D., Sokolov, A., Riezler, S.: A full-text learning to rank dataset for medical information retrieval. In: Ferro, N., et al. (eds.) Advances in Information Retrieval, pp. 716\u2013722. Springer International Publishing, Cham (2016)"},{"key":"7_CR7","unstructured":"Boytsov, L., et al.: InPars-light: cost-effective unsupervised training of efficient rankers. Trans. Mach. Learn. Res. (2024)"},{"key":"7_CR8","unstructured":"Brown, T., et al.: Language models are few-shot learners. In: Larochelle, H., Ranzato, M., Hadsell, R., Balcan, M., Lin, H. (eds.) Advances in Neural Information Processing Systems, vol.\u00a033, pp. 1877\u20131901. Curran Associates, Inc. (2020)"},{"key":"7_CR9","unstructured":"Choi, C., et al.: Linq-Embed-Mistral Technical Report. arXiv:2412.03223 (2024)"},{"issue":"70","key":"7_CR10","first-page":"1","volume":"25","author":"HW Chung","year":"2024","unstructured":"Chung, H.W., et al.: Scaling instruction-finetuned language models. J. Mach. Learn. Res. 25(70), 1\u201353 (2024)","journal-title":"J. Mach. Learn. Res."},{"key":"7_CR11","doi-asserted-by":"crossref","unstructured":"Cohan, A., Feldman, S., Beltagy, I., Downey, D., Weld, D.: SPECTER: document-level representation learning using citation-informed transformers. In: Jurafsky, D., Chai, J., Schluter, N., Tetreault, J. (eds.) Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics, pp. 2270\u20132282. Association for Computational Linguistics, Online, July 2020","DOI":"10.18653\/v1\/2020.acl-main.207"},{"key":"7_CR12","doi-asserted-by":"crossref","unstructured":"Craswell, N., Mitra, B., Yilmaz, E., Campos, D.: Overview of the TREC 2020 deep learning track. In: Proceedings of the Twenty-Ninth Text REtrieval Conference Proceedings (TREC 2020), Gaithersburg, Maryland (2020)","DOI":"10.6028\/NIST.SP.1266.deep-overview"},{"key":"7_CR13","doi-asserted-by":"crossref","unstructured":"Craswell, N., Mitra, B., Yilmaz, E., Campos, D., Voorhees, E.M.: Overview of the TREC 2019 deep learning track. In: Proceedings of the Twenty-Eighth Text REtrieval Conference Proceedings (TREC 2019), Gaithersburg, Maryland (2019)","DOI":"10.6028\/NIST.SP.1266.deep-overview"},{"key":"7_CR14","unstructured":"Dai, Z., et al.: Promptagator: few-shot dense retrieval from 8 examples. In: The Eleventh International Conference on Learning Representations (2023)"},{"key":"7_CR15","unstructured":"Diggelmann, T., Boyd-Graber, J., Bulian, J., Ciaramita, M., Leippold, M.: CLIMATE-FEVER: a dataset for verification of real-world climate claims. arXiv:2012.00614 (2020)"},{"key":"7_CR16","unstructured":"Dubey, A., et\u00a0al.: The Llama 3 Herd of Models. arXiv:2407.21783 (2024)"},{"key":"7_CR17","doi-asserted-by":"crossref","unstructured":"Gao, L., Zhang, Y., Han, J., Callan, J.: Scaling deep contrastive learning batch size under memory limited setup. In: Proceedings of the 6th Workshop on Representation Learning for NLP (2021)","DOI":"10.18653\/v1\/2021.repl4nlp-1.31"},{"key":"7_CR18","doi-asserted-by":"crossref","unstructured":"Hasibi, F., et al.: DBpedia-entity v2: a test collection for entity search. In: Proceedings of the 40th International ACM SIGIR Conference on Research and Development in Information Retrieval, SIGIR 2017, pp. 1265\u20131268. Association for Computing Machinery, New York, NY, USA (2017)","DOI":"10.1145\/3077136.3080751"},{"key":"7_CR19","unstructured":"Hofst\u00e4tter, S., Althammer, S., Schr\u00f6der, M., Sertkan, M., Hanbury, A.: Improving efficient neural ranking models with cross-architecture knowledge distillation. arXiv:2010.02666 (2020)"},{"key":"7_CR20","unstructured":"Jeronymo, V., et al.: InPars-v2: Large Language Models as Efficient Dataset Generators for Information Retrieval. arXiv:2301.01820 (2023)"},{"key":"7_CR21","doi-asserted-by":"crossref","unstructured":"Jin, Q., Dhingra, B., Liu, Z., Cohen, W., Lu, X.: PubMedQA: a dataset for biomedical research question answering. In: Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP), pp. 2567\u20132577 (2019)","DOI":"10.18653\/v1\/D19-1259"},{"key":"7_CR22","doi-asserted-by":"crossref","unstructured":"Khattab, O., Zaharia, M.: ColBERT: efficient and effective passage search via contextualized late interaction over BERT. In: Proceedings of the 43rd International ACM SIGIR Conference on Research and Development in Information Retrieval, SIGIR 2020, pp. 39\u201348. Association for Computing Machinery, New York, NY, USA (2020)","DOI":"10.1145\/3397271.3401075"},{"key":"7_CR23","first-page":"452","volume":"7","author":"T Kwiatkowski","year":"2019","unstructured":"Kwiatkowski, T., et al.: Natural questions: a benchmark for question answering research. Trans. Assoc. Comput. Linguistics 7, 452\u2013466 (2019)","journal-title":"Trans. Assoc. Comput. Linguistics"},{"key":"7_CR24","doi-asserted-by":"crossref","unstructured":"Kwon, W., et al.: Efficient memory management for large language model serving with PagedAttention. In: Proceedings of the ACM SIGOPS 29th Symposium on Operating Systems Principles (2023)","DOI":"10.1145\/3600006.3613165"},{"key":"7_CR25","unstructured":"Lee, C., et al.: NV-embed: improved techniques for training LLMs as generalist embedding models. arXiv:2405.17428 (2024)"},{"key":"7_CR26","unstructured":"Li, C., et al.: Making text embedders few-shot learners. In: The Thirteenth International Conference on Learning Representations (2025)"},{"key":"7_CR27","unstructured":"Li, Z., Zhang, X., Zhang, Y., Long, D., Xie, P., Zhang, M.: Towards general text embeddings with multi-stage contrastive learning. arXiv:2308.03281 (2023)"},{"key":"7_CR28","unstructured":"Liang, D., et al.: Embedding-based zero-shot retrieval through query generation. arXiv:2009.10270 (2020)"},{"key":"7_CR29","doi-asserted-by":"crossref","unstructured":"Lin, S.C., et al.: How to train your dragon: diverse augmentation towards generalizable dense retrieval. In: Bouamor, H., Pino, J., Bali, K. (eds.) Findings of the Association for Computational Linguistics: EMNLP 2023, pp. 6385\u20136400. Association for Computational Linguistics, Singapore, December 2023","DOI":"10.18653\/v1\/2023.findings-emnlp.423"},{"key":"7_CR30","unstructured":"Lin, S.C., Yang, J.H., Lin, J.: Distilling dense representations for ranking using tightly-coupled teachers. arXiv:2010.11386 (2020)"},{"key":"7_CR31","doi-asserted-by":"crossref","unstructured":"Ma, J., Korotkov, I., Yang, Y., Hall, K., McDonald, R.: Zero-shot neural passage retrieval via domain-targeted synthetic question generation. In: Merlo, P., Tiedemann, J., Tsarfaty, R. (eds.) Proceedings of the 16th Conference of the European Chapter of the Association for Computational Linguistics: Main Volume, pp. 1075\u20131088. Association for Computational Linguistics, Online, April 2021","DOI":"10.18653\/v1\/2021.eacl-main.92"},{"key":"7_CR32","doi-asserted-by":"crossref","unstructured":"Maia, M., et al.: WWW\u201918 open challenge: financial opinion mining and question answering. In: Companion Proceedings of the The Web Conference 2018, WWW 2018, pp. 1941\u20131942. International World Wide Web Conferences Steering Committee, Republic and Canton of Geneva, CHE (2018)","DOI":"10.1145\/3184558.3192301"},{"key":"7_CR33","unstructured":"Menon, A., Jayasumana, S., Rawat, A.S., Kim, S., Reddi, S., Kumar, S.: In defense of dual-encoders for neural ranking. In: International Conference on Machine Learning, pp. 15376\u201315400. PMLR (2022)"},{"key":"7_CR34","unstructured":"Merrick, L., Xu, D., Nuti, G., Campos, D.: Arctic-embed: scalable, efficient, and accurate text embedding models. arXiv:2405.05374 (2024)"},{"key":"7_CR35","doi-asserted-by":"crossref","unstructured":"Moreira, G.D.S.P., Osmulski, R., Xu, M., Ak, R., Schifferer, B., Oldridge, E.: NV-retriever: improving text embedding models with effective hard-negative mining. arXiv:2407.15831 (2025)","DOI":"10.1145\/3746252.3761254"},{"key":"7_CR36","doi-asserted-by":"crossref","unstructured":"Muennighoff, N., Tazi, N., Magne, L., Reimers, N.: MTEB: massive text embedding benchmark. In: Vlachos, A., Augenstein, I. (eds.) Proceedings of the 17th Conference of the European Chapter of the Association for Computational Linguistics, pp. 2014\u20132037. Association for Computational Linguistics, Dubrovnik, Croatia, May 2023","DOI":"10.18653\/v1\/2023.eacl-main.148"},{"key":"7_CR37","unstructured":"Oord, A.V.D., Li, Y., Vinyals, O.: Representation Learning with Contrastive Predictive Coding. arXiv:1807.03748 (2018)"},{"key":"7_CR38","doi-asserted-by":"crossref","unstructured":"Pradeep, R., et al.: Ragnar\u00f6k: a reusable RAG framework and baselines for TREC 2024 retrieval-augmented generation track. arXiv:2406.16828 (2024)","DOI":"10.1007\/978-3-031-88708-6_9"},{"key":"7_CR39","doi-asserted-by":"crossref","unstructured":"Qin, Z., et\u00a0al.: Large language models are effective text rankers with pairwise ranking prompting. arXiv:2306.17563 (2023)","DOI":"10.18653\/v1\/2024.findings-naacl.97"},{"key":"7_CR40","doi-asserted-by":"crossref","unstructured":"Qu, Y., et al.: RocketQA: an optimized training approach to dense passage retrieval for open-domain question answering. arXiv:2010.08191 (2021)","DOI":"10.18653\/v1\/2021.naacl-main.466"},{"key":"7_CR41","doi-asserted-by":"crossref","unstructured":"Ren, R., et al.: RocketQAv2: a joint training method for dense passage retrieval and passage re-ranking. In: Moens, M.F., Huang, X., Specia, L., Yih, S.W.T. (eds.) Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing, pp. 2825\u20132835. Association for Computational Linguistics, Online and Punta Cana, Dominican Republic, November 2021","DOI":"10.18653\/v1\/2021.emnlp-main.224"},{"key":"7_CR42","doi-asserted-by":"crossref","unstructured":"Saad-Falcon, J., et al.: UDAPDR: unsupervised domain adaptation via LLM prompting and distillation of rerankers. In: Bouamor, H., Pino, J., Bali, K. (eds.) Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, pp. 11265\u201311279. Association for Computational Linguistics, Singapore, December 2023","DOI":"10.18653\/v1\/2023.emnlp-main.693"},{"key":"7_CR43","doi-asserted-by":"crossref","unstructured":"Santhanam, K., Khattab, O., Saad-Falcon, J., Potts, C., Zaharia, M.: ColBERTv2: effective and efficient retrieval via lightweight late interaction. In: Carpuat, M., de\u00a0Marneffe, M.C., Meza\u00a0Ruiz, I.V. (eds.) Proceedings of the 2022 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, pp. 3715\u20133734. Association for Computational Linguistics, Seattle, United States, July 2022","DOI":"10.18653\/v1\/2022.naacl-main.272"},{"key":"7_CR44","unstructured":"Tamber, M.S., Pradeep, R., Lin, J.: Scaling down, LiTting up: efficient zero-shot listwise reranking with Seq2seq encoder-decoder models. arXiv:2312.16098 (2023)"},{"key":"7_CR45","unstructured":"Thakur, N., Reimers, N., R\u00fcckl\u00e9, A., Srivastava, A., Gurevych, I.: BEIR: a heterogenous benchmark for zero-shot evaluation of information retrieval models. arXiv:2104.08663 (2021)"},{"key":"7_CR46","doi-asserted-by":"crossref","unstructured":"Thorne, J., Vlachos, A., Christodoulopoulos, C., Mittal, A.: FEVER: a large-scale dataset for fact extraction and VERification. In: Walker, M., Ji, H., Stent, A. (eds.) Proceedings of the 2018 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long Papers), pp. 809\u2013819. Association for Computational Linguistics, New Orleans, Louisiana, June 2018","DOI":"10.18653\/v1\/N18-1074"},{"key":"7_CR47","doi-asserted-by":"crossref","unstructured":"Voorhees, E., et al.: TREC-COVID: constructing a pandemic information retrieval test collection. In: ACM SIGIR Forum, vol.\u00a054, pp. 1\u201312. ACM, New York, NY, USA (2021)","DOI":"10.1145\/3451964.3451965"},{"key":"7_CR48","doi-asserted-by":"crossref","unstructured":"Wachsmuth, H., Syed, S., Stein, B.: Retrieval of the best counterargument without prior topic knowledge. In: Gurevych, I., Miyao, Y. (eds.) Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 241\u2013251. Association for Computational Linguistics, Melbourne, Australia, July 2018","DOI":"10.18653\/v1\/P18-1023"},{"key":"7_CR49","doi-asserted-by":"crossref","unstructured":"Wadden, D., et al.: Fact or fiction: verifying scientific claims. In: Webber, B., Cohn, T., He, Y., Liu, Y. (eds.) Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP), pp. 7534\u20137550. Association for Computational Linguistics, Online, November 2020","DOI":"10.18653\/v1\/2020.emnlp-main.609"},{"key":"7_CR50","doi-asserted-by":"crossref","unstructured":"Wang, K., Thakur, N., Reimers, N., Gurevych, I.: GPL: generative pseudo labeling for unsupervised domain adaptation of dense retrieval. In: Carpuat, M., de\u00a0Marneffe, M.C., Meza\u00a0Ruiz, I.V. (eds.) Proceedings of the 2022 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, pp. 2345\u20132360. Association for Computational Linguistics, Seattle, United States, July 2022","DOI":"10.18653\/v1\/2022.naacl-main.168"},{"key":"7_CR51","unstructured":"Wang, L., et al.: Text embeddings by weakly-supervised contrastive pre-training. arXiv:2212.03533 (2022)"},{"key":"7_CR52","doi-asserted-by":"crossref","unstructured":"Xiao, S., Liu, Z., Zhang, P., Muennighoff, N., Lian, D., Nie, J.Y.: C-pack: packed resources for general chinese embeddings. In: Proceedings of the 47th International ACM SIGIR Conference on Research and Development in Information Retrieval, SIGIR 2024, pp. 641\u2013649. Association for Computing Machinery, New York, NY, USA (2024)","DOI":"10.1145\/3626772.3657878"},{"key":"7_CR53","unstructured":"Yang, S., Seo, M.: Is Retriever Merely an Approximator of Reader? arXiv:2010.10999 (2020)"},{"key":"7_CR54","doi-asserted-by":"crossref","unstructured":"Yang, Z., et al.: HotpotQA: a dataset for diverse, explainable multi-hop question answering. In: Riloff, E., Chiang, D., Hockenmaier, J., Tsujii, J. (eds.) Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing, pp. 2369\u20132380. Association for Computational Linguistics, Brussels, Belgium, October\u2013November 2018","DOI":"10.18653\/v1\/D18-1259"},{"key":"7_CR55","doi-asserted-by":"crossref","unstructured":"Yoon, S., Choi, E., Kim, J., Yun, H., Kim, Y., Hwang, S.W.: ListT5: listwise reranking with fusion-in-decoder improves zero-shot retrieval. In: Ku, L.W., Martins, A., Srikumar, V. (eds.) Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 2287\u20132308. Association for Computational Linguistics, Bangkok, Thailand, August 2024","DOI":"10.18653\/v1\/2024.acl-long.125"},{"key":"7_CR56","doi-asserted-by":"crossref","unstructured":"Zhuang, H., et al.: RankT5: fine-tuning T5 for text ranking with ranking losses. In: Proceedings of the 46th International ACM SIGIR Conference on Research and Development in Information Retrieval, pp. 2308\u20132313 (2023)","DOI":"10.1145\/3539618.3592047"}],"container-title":["Lecture Notes in Computer Science","Advances in Information Retrieval"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-21300-6_7","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,24]],"date-time":"2026-03-24T13:08:53Z","timestamp":1774357733000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-21300-6_7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9783032212993","9783032213006"],"references-count":56,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-21300-6_7","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"25 March 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"Some authors were employed by Vectara during this work (including internship employment). The authors declare no other competing interests relevant to this work. Funding is listed in the Acknowledgements.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"ECIR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Information Retrieval","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Delft","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"The Netherlands","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 March 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2 April 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"48","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ecir2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ecir2026.eu\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}