{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,28]],"date-time":"2026-07-28T20:10:00Z","timestamp":1785269400619,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":48,"publisher":"ACM","license":[{"start":{"date-parts":[[2021,7,11]],"date-time":"2021-07-11T00:00:00Z","timestamp":1625961600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/100014409","name":"Center for Intelligent Information Retrieval, University of Massachusetts Amherst","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100014409","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2021,7,11]]},"DOI":"10.1145\/3404835.3462987","type":"proceedings-article","created":{"date-parts":[[2021,7,12]],"date-time":"2021-07-12T02:41:54Z","timestamp":1626057714000},"page":"1753-1757","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":20,"title":["Passage Retrieval for Outside-Knowledge Visual Question Answering"],"prefix":"10.1145","author":[{"given":"Chen","family":"Qu","sequence":"first","affiliation":[{"name":"University of Massachusetts Amherst, Amherst, MA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hamed","family":"Zamani","sequence":"additional","affiliation":[{"name":"University of Massachusetts Amherst, Amherst, MA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Liu","family":"Yang","sequence":"additional","affiliation":[{"name":"University of Massachusetts Amherst, Amherst, MA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"W. Bruce","family":"Croft","sequence":"additional","affiliation":[{"name":"University of Massachusetts Amherst, Amherst, MA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Erik","family":"Learned-Miller","sequence":"additional","affiliation":[{"name":"University of Massachusetts Amherst, Amherst, MA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2021,7,11]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-016-0966-6"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00636"},{"key":"e_1_3_2_1_3_1","volume-title":"ICCV","author":"TH.","year":"2017","unstructured":", Cord, and Thome]Benyounes2017MUTANMTH. Ben-younes, R. Cad\u00e8ne, M. Cord, and N. Thome. MUTAN: Multimodal Tucker Fusion for Visual Question Answering. In ICCV, 2017."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/3166072.3166084"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P17-1171"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/1571941.1572114"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3404835.3462806"},{"key":"e_1_3_2_1_8_1","volume-title":"NAACL-HLT","author":"Devlin J.","year":"2019","unstructured":"J. Devlin, M.-W. Chang, K. Lee, and K. Toutanova. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In NAACL-HLT, 2019."},{"key":"e_1_3_2_1_9_1","volume-title":"TREC","author":"Fox E. A.","year":"1993","unstructured":"E. A. Fox and J. A. Shaw. Combination of Multiple Searches. In TREC, 1993."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D16-1044"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.findings-emnlp.44"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.670"},{"key":"e_1_3_2_1_13_1","volume-title":"REALM: Retrieval-Augmented Language Model Pre-Training. ArXiv, abs\/2002.08909","author":"Guu K.","year":"2020","unstructured":"K. Guu, K. Lee, Z. Tung, P. Pasupat, and M.-W. Chang. REALM: Retrieval-Augmented Language Model Pre-Training. ArXiv, abs\/2002.08909, 2020."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2018\/570"},{"key":"e_1_3_2_1_15_1","volume-title":"ArXiv","author":"Johnson J.","year":"2017","unstructured":"u]faissJ. Johnson, M. Douze, and H. J\u00e9gou. Billion-scale similarity search with GPUs. ArXiv, 2017."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.550"},{"key":"e_1_3_2_1_17_1","volume-title":"NeurIPS","author":"Kim J.","year":"2018","unstructured":"J. Kim, J. Jun, and B. Zhang. Bilinear Attention Networks. In NeurIPS, 2018."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-016-0981-7"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/258525.258587"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1612"},{"key":"e_1_3_2_1_21_1","volume-title":"Incorporating external knowledge to answer open-domain visual questions with dynamic memory networks. ArXiv, abs\/1712.00733","author":"Li G.","year":"2017","unstructured":"G. Li, H. Su, and W. Zhu. Incorporating external knowledge to answer open-domain visual questions with dynamic memory networks. ArXiv, abs\/1712.00733, 2017."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3397271.3401244"},{"key":"e_1_3_2_1_23_1","volume-title":"NIPS","author":"Lu J.","year":"2016","unstructured":"J. Lu, J. Yang, D. Batra, and D. Parikh. Hierarchical Question-Image Co-Attention for Visual Question Answering. In NIPS, 2016."},{"key":"e_1_3_2_1_24_1","volume-title":"Sparse, Dense, and Attentional Representations for Text Retrieval. ArXiv, abs\/2005.00181","author":"Luan Y.","year":"2020","unstructured":"Y. Luan, J. Eisenstein, K. Toutanova, and M. Collins. Sparse, Dense, and Attentional Representations for Text Retrieval. ArXiv, abs\/2005.00181, 2020."},{"key":"e_1_3_2_1_25_1","volume-title":"NIPS","author":"Malinowski M.","year":"2014","unstructured":"M. Malinowski and M. Fritz. A Multi-World Approach to Question Answering about Real-World Scenes based on Uncertain Input. In NIPS, 2014."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.9"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00331"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01237-3_28"},{"key":"e_1_3_2_1_29_1","volume-title":"NeurIPS","author":"Narasimhan M.","year":"2018","unstructured":"M. Narasimhan, S. Lazebnik, and A. G. Schwing. Out of the Box: Reasoning with Graph Convolution Nets for Factual Visual Question Answering. In NeurIPS, 2018."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3397271.3401110"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-72113-8_35"},{"key":"e_1_3_2_1_32_1","volume-title":"RocketQA: An Optimized Training Approach to Dense Passage Retrieval for Open-Domain Question Answering. ArXiv, abs\/2010.08191","author":"Qu OY.","year":"2020","unstructured":"Qu, Ding, Liu, Liu, Ren, Zhao, Dong, Wu, and Wang]Qu2020RocketQAAOY. Qu, Y. Ding, J. Liu, K. Liu, R. Ren, X. Zhao, D. Dong, H. Wu, and H. Wang. RocketQA: An Optimized Training Approach to Dense Passage Retrieval for Open-Domain Question Answering. ArXiv, abs\/2010.08191, 2020 b ."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D16-1264"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2016.2577031"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1514"},{"key":"e_1_3_2_1_36_1","volume-title":"NIPS","author":"Vaswani A.","year":"2017","unstructured":"A. Vaswani, N. Shazeer, N. Parmar, J. Uszkoreit, L. Jones, A. N. Gomez, L. Kaiser, and I. Polosukhin. Attention Is All You Need. In NIPS, 2017."},{"key":"e_1_3_2_1_37_1","volume-title":"TREC","author":"Voorhees E. M.","year":"1999","unstructured":"E. M. Voorhees and D. M. Tice. The TREC-8 Question Answering Track Evaluation. In TREC, 1999."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2017\/179"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2017.2754246"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.500"},{"key":"e_1_3_2_1_41_1","volume-title":"ICML","author":"Xiong C.","year":"2016","unstructured":"C. Xiong, S. Merity, and R. Socher. Dynamic Memory Networks for Visual and Textual Question Answering. In ICML, 2016."},{"key":"e_1_3_2_1_42_1","volume-title":"Approximate Nearest Neighbor Negative Contrastive Learning for Dense Text Retrieval. ArXiv, abs\/2007.00808","author":"Xiong NL.","year":"2020","unstructured":"Xiong, Xiong, Li, Tang, Liu, Bennett, Ahmed, and Overwijk]Xiong2020ApproximateNNL. Xiong, C. Xiong, Y. Li, K.-F. Tang, J. Liu, P. Bennett, J. Ahmed, and A. Overwijk. Approximate Nearest Neighbor Negative Contrastive Learning for Dense Text Retrieval. ArXiv, abs\/2007.00808, 2020 a ."},{"key":"e_1_3_2_1_43_1","unstructured":"Xiong Li Iyer Du Lewis Wang Mehdad tau Yih Riedel Kiela and Ouguz]Xiong2020AnsweringCOW. Xiong X. Li S. Iyer J. Du P. Lewis W. Y. Wang Y. Mehdad W. tau Yih S. Riedel D. Kiela and B. Ouguz. Answering Complex Open-Domain Questions with Multi-Hop Dense Retrieval. ArXiv abs\/2009.12756 2020 b ."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/3366423.3380011"},{"key":"e_1_3_2_1_45_1","first-page":"107563","volume":"108","author":"Yu J.","year":"2020","unstructured":"J. Yu, Z. Zhu, Y. Wang, W. Zhang, Y. Hu, and J. Tan. Cross-modal Knowledge Reasoning for Knowledge-based Visual Question Answering. Pattern Recognition, 108: 107563, 2020.","journal-title":"Cross-modal Knowledge Reasoning for Knowledge-based Visual Question Answering. Pattern Recognition"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.283"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.540"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2020\/153"}],"event":{"name":"SIGIR '21: The 44th International ACM SIGIR Conference on Research and Development in Information Retrieval","location":"Virtual Event Canada","acronym":"SIGIR '21","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 44th International ACM SIGIR Conference on Research and Development in Information Retrieval"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3404835.3462987","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3404835.3462987","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T20:18:20Z","timestamp":1750191500000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3404835.3462987"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,7,11]]},"references-count":48,"alternative-id":["10.1145\/3404835.3462987","10.1145\/3404835"],"URL":"https:\/\/doi.org\/10.1145\/3404835.3462987","relation":{},"subject":[],"published":{"date-parts":[[2021,7,11]]},"assertion":[{"value":"2021-07-11","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}