{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T05:11:52Z","timestamp":1784178712015,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":39,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,7,13]]},"DOI":"10.1145\/3726302.3730201","type":"proceedings-article","created":{"date-parts":[[2025,7,14]],"date-time":"2025-07-14T01:38:52Z","timestamp":1752457132000},"page":"2926-2930","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Generate-Distill: Training Cross-Language IR Models with Synthetically-Generated Data"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7347-7086","authenticated-orcid":false,"given":"Dawn","family":"Lawrie","sequence":"first","affiliation":[{"name":"Johns Hopkins HLTCOE, Baltimore, MD, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2357-9807","authenticated-orcid":false,"given":"Efsun","family":"Kayi","sequence":"additional","affiliation":[{"name":"Johns Hopkins HLTCOE, Baltimore, MD, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0051-1535","authenticated-orcid":false,"given":"Eugene","family":"Yang","sequence":"additional","affiliation":[{"name":"Johns Hopkins HLTCOE, Baltimore, MD, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3866-3013","authenticated-orcid":false,"given":"James","family":"Mayfield","sequence":"additional","affiliation":[{"name":"Johns Hopkins HLTCOE, Baltimore, MD, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1696-0407","authenticated-orcid":false,"given":"Douglas W.","family":"Oard","sequence":"additional","affiliation":[{"name":"University of Maryland, College Park, MD, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-3345-6346","authenticated-orcid":false,"given":"Scott","family":"Miller","sequence":"additional","affiliation":[{"name":"USC Information Sciences Institute, Boston, MA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,7,13]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/3583780.3615221"},{"key":"e_1_3_2_1_2_1","volume-title":"DUQGen: Effective Unsupervised Domain Adaptation of Neural Rankers by Diversifying Synthetic Query Generation. arXiv preprint arXiv:2404.02489","author":"Chandradevan Ramraj","year":"2024","unstructured":"Ramraj Chandradevan, Kaustubh D Dhole, and Eugene Agichtein. 2024. DUQGen: Effective Unsupervised Domain Adaptation of Neural Rankers by Diversifying Synthetic Query Generation. arXiv preprint arXiv:2404.02489 (2024)."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3626772.3657765"},{"key":"e_1_3_2_1_4_1","unstructured":"David R. Cheriton. 2019. From doc2query to docTTTTTquery. https:\/\/api.semanticscholar.org\/CorpusID:208612557"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.747"},{"key":"e_1_3_2_1_6_1","volume-title":"Proceedings of the 44th European Conference on Information Retrieval (ECIR).","author":"Costello Cash","year":"2022","unstructured":"Cash Costello, Eugene Yang, Dawn Lawrie, and James Mayfield. 2022. Patapsco: A Python Framework for Cross-Language Information Retrieval Experiments. In Proceedings of the 44th European Conference on Information Retrieval (ECIR)."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3331184.3331303"},{"key":"e_1_3_2_1_8_1","volume-title":"Promptagator: Few-shot Dense Retrieval From 8 Examples. arxiv:2209.11755 [cs.CL] https:\/\/arxiv.org\/abs\/2209.11755","author":"Dai Zhuyun","year":"2022","unstructured":"Zhuyun Dai, Vincent Y. Zhao, Ji Ma, Yi Luan, Jianmo Ni, Jing Lu, Anton Bakalov, Kelvin Guu, Keith B. Hall, and Ming-Wei Chang. 2022. Promptagator: Few-shot Dense Retrieval From 8 Examples. arxiv:2209.11755 [cs.CL] https:\/\/arxiv.org\/abs\/2209.11755"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3583780.3614923"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.120"},{"key":"e_1_3_2_1_11_1","first-page":"736","article-title":"Unsupervised Multilingual Dense Retrieval via Generative Pseudo Labeling","volume":"2024","author":"Huang Chao-Wei","year":"2024","unstructured":"Chao-Wei Huang, Chen-An Li, Tsu-Yuan Hsu, Chen-Yu Hsu, and Yun-Nung Chen. 2024. Unsupervised Multilingual Dense Retrieval via Generative Pseudo Labeling. In Findings of the Association for Computational Linguistics: EACL 2024. 736-746.","journal-title":"Findings of the Association for Computational Linguistics: EACL"},{"key":"e_1_3_2_1_12_1","volume-title":"Unsupervised Dense Information Retrieval with Contrastive Learning. Transactions on Machine Learning Research","author":"Izacard Gautier","year":"2022","unstructured":"Gautier Izacard, Mathilde Caron, Lucas Hosseini, Sebastian Riedel, Piotr Bojanowski, Armand Joulin, and Edouard Grave. 2022. Unsupervised Dense Information Retrieval with Contrastive Learning. Transactions on Machine Learning Research (2022). https:\/\/openreview.net\/forum?id=jKN1pXi7b0"},{"key":"e_1_3_2_1_13_1","volume-title":"NeuralMind-UNICAMP at 2022 TREC NeuCLIR: Large Boring Rerankers for Cross-lingual Retrieval. arXiv preprint arXiv:2303.16145","author":"Jeronymo Vitor","year":"2023","unstructured":"Vitor Jeronymo, Roberto Lotufo, and Rodrigo Nogueira. 2023. NeuralMind-UNICAMP at 2022 TREC NeuCLIR: Large Boring Rerankers for Cross-lingual Retrieval. arXiv preprint arXiv:2303.16145 (2023)."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.550"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3397271.3401075"},{"key":"e_1_3_2_1_16_1","first-page":"22199","volume-title":"Oh (Eds.)","volume":"35","author":"Kojima Takeshi","year":"2022","unstructured":"Takeshi Kojima, Shixiang (Shane) Gu, Machel Reid, Yutaka Matsuo, and Yusuke Iwasawa. 2022. Large Language Models are Zero-Shot Reasoners. In Advances in Neural Information Processing Systems, S. Koyejo, S. Mohamed, A. Agarwal, D. Belgrave, K. Cho, and A. Oh (Eds.), Vol. 35. Curran Associates, Inc., 22199-22213. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2022\/file\/8bb0d291acd4acf06ef112099c16f326-Paper-Conference.pdf"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"crossref","unstructured":"Thiago Laitz Konstantinos Papakostas Roberto Lotufo and Rodrigo Nogueira. 2024. InRanker: Distilled Rankers for Zero-shot Information Retrieval. arxiv:2401.06910 [cs.IR]","DOI":"10.1007\/978-3-031-79032-4_10"},{"key":"e_1_3_2_1_18_1","volume-title":"Overview of the TREC 2022 NeuCLIR Track. In The Thirty-First Text REtrieval Conference (TREC 2022) Proceedings.","author":"Lawrie Dawn","year":"2023","unstructured":"Dawn Lawrie, Sean MacAvaney, James Mayfield, Paul McNamee, Douglas W. Oard, Luca Soldanini, and Eugene Yang. 2023a. Overview of the TREC 2022 NeuCLIR Track. In The Thirty-First Text REtrieval Conference (TREC 2022) Proceedings."},{"key":"e_1_3_2_1_19_1","volume-title":"Overview of the TREC 2023 NeuCLIR Track. In The Thirty-Second Text REtrieval Conference (TREC 2023) Proceedings.","author":"Lawrie Dawn","year":"2024","unstructured":"Dawn Lawrie, Sean MacAvaney, James Mayfield, Paul McNamee, Douglas W. Oard, Luca Soldanini, and Eugene Yang. 2024. Overview of the TREC 2023 NeuCLIR Track. In The Thirty-Second Text REtrieval Conference (TREC 2023) Proceedings."},{"key":"e_1_3_2_1_20_1","volume-title":"Overview of the TREC 2024 NeuCLIR Track. In The Thirty-Third Text REtrieval Conference (TREC 2024) Proceedings.","author":"Lawrie Dawn","year":"2025","unstructured":"Dawn Lawrie, Sean MacAvaney, James Mayfield, Paul McNamee, Douglas W. Oard, Luca Soldanini, and Eugene Yang. 2025. Overview of the TREC 2024 NeuCLIR Track. In The Thirty-Third Text REtrieval Conference (TREC 2024) Proceedings."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3539618.3591893"},{"key":"e_1_3_2_1_22_1","volume-title":"Proceedings of the 29th International Conference on Computational Linguistics. International Committee on Computational Linguistics, Gyeongju, Republic of Korea, 3917-3923","author":"Li Yudong","year":"2022","unstructured":"Yudong Li, Yuqing Zhang, Zhe Zhao, Linlin Shen, Weijie Liu, Weiquan Mao, and Hui Zhang. 2022. CSL: A Large-scale Chinese Scientific Literature Dataset. In Proceedings of the 29th International Conference on Computational Linguistics. International Committee on Computational Linguistics, Gyeongju, Republic of Korea, 3917-3923. https:\/\/aclanthology.org\/2022.coling-1.344"},{"key":"e_1_3_2_1_23_1","volume-title":"Zero-Shot Neural Passage Retrieval via Domain-Targeted Synthetic Question Generation. arXiv preprint arXiv:2004.14503","author":"Ma Ji","year":"2020","unstructured":"Ji Ma, Ivan Korotkov, Yinfei Yang, Keith Hall, and Ryan McDonald. 2020. Zero-Shot Neural Passage Retrieval via Domain-Targeted Synthetic Question Generation. arXiv preprint arXiv:2004.14503 (2020)."},{"key":"e_1_3_2_1_24_1","volume-title":"Synthetic Cross-Language Information Retrieval Training Data. arXiv preprint arXiv:2305.00331","author":"Mayfield James","year":"2023","unstructured":"James Mayfield, Eugene Yang, Dawn Lawrie, Samuel Barham, Orion Weller, Marc Mason, Suraj Nair, and Scott Miller. 2023. Synthetic Cross-Language Information Retrieval Training Data. arXiv preprint arXiv:2305.00331 (2023)."},{"key":"e_1_3_2_1_25_1","volume-title":"DCAI24 workshop at CIKM2024","author":"Meng Rui","year":"2024","unstructured":"Rui Meng, Ye Liu, Semih Yavuz, Divyansh Agarwal, Lifu Tu, Ning Yu, Jianguo Zhang, Meghana Bhat, and Yingbo Zhou. 2024. AugTriever: Unsupervised Dense Retrieval by Scalable Data Augmentation. In DCAI24 workshop at CIKM2024."},{"key":"e_1_3_2_1_26_1","volume-title":"Oard","author":"Nair Suraj","year":"2022","unstructured":"Suraj Nair, Eugene Yang, Dawn Lawrie, Kevin Duh, Paul McNamee, Kenton Murray, James Mayfield, and Douglas W. Oard. 2022. Transfer Learning Approaches for Building Cross-Language Dense Retrieval Models. In Advances in Information Retrieval: 44th European Conference on IR Research, ECIR 2022, Stavanger, Norway, April 10-14, 2022, Proceedings, Part I (Stavanger, Norway). Springer-Verlag, Berlin, Heidelberg, 382-396."},{"key":"e_1_3_2_1_27_1","volume-title":"MS MARCO: A Human Generated MAchine Reading COmprehension Dataset. arXiv preprint arXiv:1611.09268","author":"Nguyen Tri","year":"2016","unstructured":"Tri Nguyen, Mir Rosenberg, Xia Song, Jianfeng Gao, Saurabh Tiwary, Rangan Majumder, and Li Deng. 2016. MS MARCO: A Human Generated MAchine Reading COmprehension Dataset. arXiv preprint arXiv:1611.09268 (2016). arXiv:1611.09268 http:\/\/arxiv.org\/abs\/1611.09268"},{"key":"e_1_3_2_1_28_1","unstructured":"Rodrigo Nogueira Wei Yang Jimmy Lin and Kyunghyun Cho. 2019. Document Expansion by Query Prediction. arxiv:1904.08375 [cs.IR] https:\/\/arxiv.org\/abs\/1904.08375"},{"key":"e_1_3_2_1_29_1","volume-title":"Proceedings of the 29th International Conference on Computational Linguistics. 1065-1070","author":"Reddy Revanth Gangi","year":"2022","unstructured":"Revanth Gangi Reddy, Vikas Yadav, Md Arafat Sultan, Martin Franz, Vittorio Castelli, Heng Ji, and Avirup Sil. 2022. Towards Robust Neural Retrieval with Source Domain Synthetic Pre-Finetuning. In Proceedings of the 29th International Conference on Computational Linguistics. 1065-1070."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.203"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3511808.3557325"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.naacl-long.426"},{"key":"e_1_3_2_1_33_1","volume-title":"Quoc Le, and Denny Zhou.","author":"Wei Jason","year":"2022","unstructured":"Jason Wei, Xuezhi Wang, Dale Schuurmans, Maarten Bosma, Ed H. Chi, Quoc Le, and Denny Zhou. 2022. Chain of Thought Prompting Elicits Reasoning in Large Language Models. CoRR, Vol. abs\/2201.11903 (2022). arXiv:2201.11903 https:\/\/arxiv.org\/abs\/2201.11903"},{"key":"e_1_3_2_1_34_1","volume-title":"Proceedings of the Thirteenth International Conference on Learning Representations","author":"Weller Orion","year":"2025","unstructured":"Orion Weller, Benjamin Van Durme, Dawn Lawrie, Ashwin Paranjape, Yuhao Zhang, and Jack Hessel. 2025 a. Promptriever: Instruction-Trained Retrievers Can Be Prompted Like Language Models. In Proceedings of the Thirteenth International Conference on Learning Representations. Singapore. https:\/\/arxiv.org\/abs\/2409.11136"},{"key":"e_1_3_2_1_35_1","unstructured":"Orion Weller Kathryn Ricci Eugene Yang Andrew Yates Dawn Lawrie and Benjamin Van Durme. 2025 b. Rank1: Test-Time Compute for Reranking in Information Retrieval. arxiv:2502.18418 [cs.IR] https:\/\/arxiv.org\/abs\/2502.18418"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.naacl-main.41"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-56060-6_4"},{"key":"e_1_3_2_1_38_1","volume-title":"DCAI24 workshop at CIKM2024","author":"Zeng Qiuhai","unstructured":"Qiuhai Zeng, Zimeng Qiu, Dae Yon Hwang, Xin He, and William M. Campbell. 2024. Unsupervised Text Representation Learning via Instruction-Tuning for Zero-Shot Dense Retrieval. In DCAI24 workshop at CIKM2024."},{"key":"e_1_3_2_1_39_1","volume-title":"Proceedings of the 46th International ACM SIGIR Conference on Research and Development in Information Retrieval. 1827-1832","author":"Zhuang Shengyao","year":"2023","unstructured":"Shengyao Zhuang, Linjun Shou, and Guido Zuccon. 2023. Augmenting Passage Representations with Guery Generation for Enhanced Cross-Lingual Dense Retrieval. In Proceedings of the 46th International ACM SIGIR Conference on Research and Development in Information Retrieval. 1827-1832."}],"event":{"name":"SIGIR '25: The 48th International ACM SIGIR Conference on Research and Development in Information Retrieval","location":"Padua Italy","acronym":"SIGIR '25","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 48th International ACM SIGIR Conference on Research and Development in Information Retrieval"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3726302.3730201","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T12:10:03Z","timestamp":1755864603000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3726302.3730201"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7,13]]},"references-count":39,"alternative-id":["10.1145\/3726302.3730201","10.1145\/3726302"],"URL":"https:\/\/doi.org\/10.1145\/3726302.3730201","relation":{},"subject":[],"published":{"date-parts":[[2025,7,13]]},"assertion":[{"value":"2025-07-13","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}