{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T02:27:26Z","timestamp":1783736846786,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":32,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,7,10]],"date-time":"2024-07-10T00:00:00Z","timestamp":1720569600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,7,10]]},"DOI":"10.1145\/3626772.3657854","type":"proceedings-article","created":{"date-parts":[[2024,7,11]],"date-time":"2024-07-11T12:40:05Z","timestamp":1720701605000},"page":"575-584","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":7,"title":["Negative Sampling Techniques for Dense Passage Retrieval in a Multilingual Setting"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5482-664X","authenticated-orcid":false,"given":"Thilina Chaturanga","family":"Rajapakse","sequence":"first","affiliation":[{"name":"University of Amsterdam, Amsterdam, Netherlands"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5970-880X","authenticated-orcid":false,"given":"Andrew","family":"Yates","sequence":"additional","affiliation":[{"name":"University of Amsterdam, Amsterdam, Netherlands"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1086-0202","authenticated-orcid":false,"given":"Maarten","family":"de Rijke","sequence":"additional","affiliation":[{"name":"University of Amsterdam, Amsterdam, Netherlands"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,7,11]]},"reference":[{"key":"e_1_3_2_1_1_1","first-page":"7547","article-title":"One Question Answering Model for Many Languages with Cross-lingual Dense Passage Retrieval","volume":"34","author":"Asai Akari","year":"2021","unstructured":"Akari Asai, Xinyan Yu, Jungo Kasai, and Hanna Hajishirzi. 2021. One Question Answering Model for Many Languages with Cross-lingual Dense Passage Retrieval. Advances in Neural Information Processing Systems , Vol. 34 (2021), 7547--7560.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_2_1","volume-title":"Israel Campiotti, Marzieh Fadaee, Roberto Lotufo, and Rodrigo Nogueira.","author":"Bonifacio Luiz","year":"2021","unstructured":"Luiz Bonifacio, Vitor Jeronymo, Hugo Queiroz Abonizio, Israel Campiotti, Marzieh Fadaee, Roberto Lotufo, and Rodrigo Nogueira. 2021. mMarco: A Multilingual Version of the MS MARCO Passage Ranking Dataset. arXiv preprint arXiv:2108.13897 (2021)."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00317"},{"key":"e_1_3_2_1_4_1","volume-title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. arXiv preprint arXiv:1810.04805","author":"Devlin Jacob","year":"2018","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2018. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. arXiv preprint arXiv:1810.04805 (2018)."},{"key":"e_1_3_2_1_5_1","volume-title":"Cross-language Information Retrieval. arXiv preprint arXiv:2111.05988","author":"\u00e1kov\u00e1 Petra Galuvs","year":"2021","unstructured":"Petra Galuvs vc \u00e1kov\u00e1 , Douglas W Oard, and Suraj Nair. 2021. Cross-language Information Retrieval. arXiv preprint arXiv:2111.05988 (2021)."},{"key":"e_1_3_2_1_6_1","volume-title":"Proceedings of the 44th International ACM SIGIR Conference on Research and Development in Information Retrieval. 113--122","author":"Sebastian","year":"2021","unstructured":"Sebastian Hofst\"atter, Sheng-Chieh Lin, Jheng-Hong Yang, Jimmy Lin, and Allan Hanbury. 2021. Efficiently Teaching an Effective Dense Retriever with Balanced Topic Aware Sampling. In Proceedings of the 44th International ACM SIGIR Conference on Research and Development in Information Retrieval. 113--122."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/TBDATA.2019.2921572"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.550"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3397271.3401075"},{"key":"e_1_3_2_1_10_1","volume-title":"The Tale of Two MS MARCO -- And their Unfair Comparisons. arXiv preprint arXiv:2304.12904","author":"Lassance Carlos","year":"2023","unstructured":"Carlos Lassance and St\u00e9phane Clinchant. 2023. The Tale of Two MS MARCO -- And their Unfair Comparisons. arXiv preprint arXiv:2304.12904 (2023)."},{"key":"e_1_3_2_1_11_1","volume-title":"Bhavani Iyer, Young-Suk Lee, and Avirup Sil.","author":"Li Yulong","year":"2021","unstructured":"Yulong Li, Martin Franz, Md Arafat Sultan, Bhavani Iyer, Young-Suk Lee, and Avirup Sil. 2021. Learning Cross-Lingual IR from an English Retriever. arXiv preprint arXiv:2112.08185 (2021)."},{"key":"e_1_3_2_1_12_1","first-page":"4134","article-title":"Efficient Training of Retrieval Models Using Negative Cache","volume":"34","author":"Lindgren Erik","year":"2021","unstructured":"Erik Lindgren, Sashank Reddi, Ruiqi Guo, and Sanjiv Kumar. 2021. Efficient Training of Retrieval Models Using Negative Cache. Advances in Neural Information Processing Systems , Vol. 34 (2021), 4134--4146.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-45442-5_31"},{"key":"e_1_3_2_1_14_1","volume-title":"MS MARCO: A Human Generated Machine Reading Comprehension Dataset. In CoCo@ NIPs.","author":"Nguyen Tri","year":"2016","unstructured":"Tri Nguyen, Mir Rosenberg, Xia Song, Jianfeng Gao, Saurabh Tiwary, Rangan Majumder, and Li Deng. 2016. MS MARCO: A Human Generated Machine Reading Comprehension Dataset. In CoCo@ NIPs."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1007\/3--540--45368--7_3"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1561\/1500000019"},{"key":"e_1_3_2_1_17_1","unstructured":"Guilherme Rosa Luiz Bonifacio Vitor Jeronymo Hugo Abonizio Marzieh Fadaee Roberto Lotufo and Rodrigo Nogueira. 2022a. In Defense of Cross-Encoders for Zero-Shot Retrieval. arXiv preprint arXiv:2212.06121 (2022)."},{"key":"e_1_3_2_1_18_1","volume-title":"No Parameter Left Behind: How Distillation and Model Size Affect Zero-shot Retrieval. arXiv preprint arXiv:2206.02873","author":"Rosa Guilherme Moraes","year":"2022","unstructured":"Guilherme Moraes Rosa, Luiz Bonifacio, Vitor Jeronymo, Hugo Abonizio, Marzieh Fadaee, Roberto Lotufo, and Rodrigo Nogueira. 2022b. No Parameter Left Behind: How Distillation and Model Size Affect Zero-shot Retrieval. arXiv preprint arXiv:2206.02873 (2022)."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.mrl-1.24"},{"key":"e_1_3_2_1_20_1","volume-title":"BEIR: A Heterogenous Benchmark for Zero-shot Evaluation of Information Retrieval Models. arXiv preprint arXiv:2104.08663","author":"Thakur Nandan","year":"2021","unstructured":"Nandan Thakur, Nils Reimers, Andreas R\u00fcckl\u00e9, Abhishek Srivastava, and Iryna Gurevych. 2021. BEIR: A Heterogenous Benchmark for Zero-shot Evaluation of Information Retrieval Models. arXiv preprint arXiv:2104.08663 (2021)."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N. Gomez Lukasz Kaiser and Illia Polosukhin. 2017. Attention Is All You Need. https:\/\/doi.org\/10.48550\/ARXIV.1706.03762","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"e_1_3_2_1_22_1","volume-title":"Proceedings of TREC-8. 77--82","author":"Voorhees Ellen M.","year":"1999","unstructured":"Ellen M. Voorhees. 1999. The TREC-8 Question Answering Track Report. In Proceedings of TREC-8. 77--82."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3539618.3591915"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-demos.6"},{"key":"e_1_3_2_1_25_1","volume-title":"Approximate Nearest Neighbor Negative Contrastive Learning for Dense Text Retrieval. arXiv preprint arXiv:2007.00808","author":"Xiong Lee","year":"2020","unstructured":"Lee Xiong, Chenyan Xiong, Ye Li, Kwok-Fung Tang, Jialin Liu, Paul Bennett, Junaid Ahmed, and Arnold Overwijk. 2020. Approximate Nearest Neighbor Negative Contrastive Learning for Dense Text Retrieval. arXiv preprint arXiv:2007.00808 (2020)."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.findings-acl.281"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3404835.3462880"},{"key":"e_1_3_2_1_28_1","volume-title":"Evaluating Extrapolation Performance of Dense Retrieval. arXiv preprint arXiv:2204.11447","author":"Zhan Jingtao","year":"2022","unstructured":"Jingtao Zhan, Xiaohui Xie, Jiaxin Mao, Yiqun Liu, Min Zhang, and Shaoping Ma. 2022. Evaluating Extrapolation Performance of Dense Retrieval. arXiv preprint arXiv:2204.11447 (2022)."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-57884-8_3"},{"key":"e_1_3_2_1_30_1","volume-title":"TyDi: A Multi-lingual Benchmark for Dense Retrieval. arXiv preprint arXiv:2108.08787","author":"Zhang Xinyu","year":"2021","unstructured":"Xinyu Zhang, Xueguang Ma, Peng Shi, and Jimmy Lin. 2021. Mr. TyDi: A Multi-lingual Benchmark for Dense Retrieval. arXiv preprint arXiv:2108.08787 (2021)."},{"key":"e_1_3_2_1_31_1","volume-title":"Towards Best Practices for Training Multilingual Dense Retrieval Models. arXiv preprint arXiv:2204.02363","author":"Zhang Xinyu","year":"2022","unstructured":"Xinyu Zhang, Kelechi Ogueji, Xueguang Ma, and Jimmy Lin. 2022a. Towards Best Practices for Training Multilingual Dense Retrieval Models. arXiv preprint arXiv:2204.02363 (2022)."},{"key":"e_1_3_2_1_32_1","volume-title":"Making a MIRACL: Multilingual Information Retrieval Across a Continuum of Languages. arXiv preprint arXiv:2210.09984","author":"Zhang Xinyu","year":"2022","unstructured":"Xinyu Zhang, Nandan Thakur, Odunayo Ogundepo, Ehsan Kamalloo, David Alfonso-Hermelo, Xiaoguang Li, Qun Liu, Mehdi Rezagholizadeh, and Jimmy Lin. 2022b. Making a MIRACL: Multilingual Information Retrieval Across a Continuum of Languages. arXiv preprint arXiv:2210.09984 (2022)."}],"event":{"name":"SIGIR 2024: The 47th International ACM SIGIR Conference on Research and Development in Information Retrieval","location":"Washington DC USA","acronym":"SIGIR 2024","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 47th International ACM SIGIR Conference on Research and Development in Information Retrieval"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3626772.3657854","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3626772.3657854","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T05:28:12Z","timestamp":1755840492000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3626772.3657854"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,7,10]]},"references-count":32,"alternative-id":["10.1145\/3626772.3657854","10.1145\/3626772"],"URL":"https:\/\/doi.org\/10.1145\/3626772.3657854","relation":{},"subject":[],"published":{"date-parts":[[2024,7,10]]},"assertion":[{"value":"2024-07-11","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}