{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T18:03:38Z","timestamp":1784138618345,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":26,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"Australian Research Council","award":["IE230100119"],"award-info":[{"award-number":["IE230100119"]}]},{"name":"Australian Research Council","award":["LP230200821"],"award-info":[{"award-number":["LP230200821"]}]},{"name":"Commonwealth Scientific and Industrial Research Organisation, Data61","award":["PhD Top-up Scholarship"],"award-info":[{"award-number":["PhD Top-up Scholarship"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,20]]},"DOI":"10.1145\/3805712.3809956","type":"proceedings-article","created":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:06:26Z","timestamp":1784135186000},"page":"4216-4220","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["PeaCap: Patch-Level Retrieval for Lightweight Retrieval-Augmented Image Captioning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0282-2481","authenticated-orcid":false,"given":"Robin","family":"Viltoriano","sequence":"first","affiliation":[{"name":"Adelaide University, Adelaide, SA, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0406-5974","authenticated-orcid":false,"given":"Wei Emma","family":"Zhang","sequence":"additional","affiliation":[{"name":"Adelaide University, Adelaide, SA, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1725-873X","authenticated-orcid":false,"given":"Hu","family":"Wang","sequence":"additional","affiliation":[{"name":"Mohamed bin Zayed University of Artificial Intelligence, Abu Dhabi, United Arab Emirates"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9590-3082","authenticated-orcid":false,"given":"Mong Yuan","family":"Sim","sequence":"additional","affiliation":[{"name":"Adelaide University, Adelaide, Australia and CSIRO Data61, Sydney, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5196-1341","authenticated-orcid":false,"given":"Yanjun","family":"Shu","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology, Harbin, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2142"},{"key":"e_1_3_2_1_2_1","first-page":"4171","volume-title":"Proc. of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (NAACL-HLT","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In Proc. of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (NAACL-HLT 2019). 4171-4186."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01422"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01855"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i2.16220"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00550"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02238"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/TBDATA.2019.2921572"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"e_1_3_2_1_10_1","volume-title":"Tuna: Instruction Tuning Using Feedback from Large Language Models. arXiv preprint arXiv:2310.13385","author":"Li Haoran","year":"2023","unstructured":"Haoran Li, Yiran Liu, Xingxing Zhang, Wei Lu, and Furu Wei. 2023b. Tuna: Instruction Tuning Using Feedback from Large Language Models. arXiv preprint arXiv:2310.13385 (2023)."},{"key":"e_1_3_2_1_11_1","first-page":"0","volume-title":"Proc. of the International Conference on Machine Learning (ICML 2023","author":"Li Junnan","year":"2023","unstructured":"Junnan Li, Dongxu Li, Silvio Savarese, and Steven Hoi. 2023a. BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models. In Proc. of the International Conference on Machine Learning (ICML 2023). 19730-19742."},{"key":"e_1_3_2_1_12_1","first-page":"12888","volume-title":"Proc. of the International Conference on Machine Learning (ICML","author":"Li Junnan","year":"2022","unstructured":"Junnan Li, Dongxu Li, Caiming Xiong, and Steven Hoi. 2022. BLIP: Bootstrapping Language-Image Pre-Training for Unified Vision-Language Understanding and Generation. In Proc. of the International Conference on Machine Learning (ICML 2022). 12888-12900."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01303"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.503"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58577-8_8"},{"key":"e_1_3_2_1_16_1","first-page":"740","volume-title":"Proc. of the European Conference on Computer Vision (ECCV","author":"Lin Tsung-Yi","year":"2014","unstructured":"Tsung-Yi Lin, Michael Maire, Serge Belongie, James Hays, Pietro Perona, Deva Ramanan, Piotr Doll\u00e1r, and C. Lawrence Zitnick. 2014. Microsoft COCO: Common Objects in Context. In Proc. of the European Conference on Computer Vision (ECCV 2014). 740-755."},{"key":"e_1_3_2_1_17_1","volume-title":"Bermano","author":"Mokady Ron","year":"2021","unstructured":"Ron Mokady, Amir Hertz, and Amit H. Bermano. 2021. ClipCap: CLIP Prefix for Image Captioning. arXiv preprint arXiv:2111.09734 (2021)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00278"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3726302.3731957"},{"key":"e_1_3_2_1_21_1","volume-title":"Vicuna: An Open-Source Chatbot Impressing GPT-4. https:\/\/lmsys.org\/blog\/2023-03-30-vicuna\/ (accessed","author":"Team The Vicuna","year":"2023","unstructured":"The Vicuna Team. 2023. Vicuna: An Open-Source Chatbot Impressing GPT-4. https:\/\/lmsys.org\/blog\/2023-03-30-vicuna\/ (accessed 12 Feb 2026)."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01747"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.52202\/079017-3860"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-emnlp.793"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01337"},{"key":"e_1_3_2_1_26_1","volume-title":"Proc. of the Twelfth International Conference on Learning Representations (ICLR","author":"Zhu Deyao","year":"2024","unstructured":"Deyao Zhu, Jun Chen, Xiaoqian Shen, Xiang Li, and Mohamed Elhoseiny. 2024. MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models. In Proc. of the Twelfth International Conference on Learning Representations (ICLR 2024)."}],"event":{"name":"SIGIR '26: The 49th International ACM SIGIR Conference on Research and Development in Information Retrieval","location":"Melbourne VIC Australia","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 49th International ACM SIGIR Conference on Research and Development in Information Retrieval"],"original-title":[],"deposited":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:08:39Z","timestamp":1784135319000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805712.3809956"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"references-count":26,"alternative-id":["10.1145\/3805712.3809956","10.1145\/3805712"],"URL":"https:\/\/doi.org\/10.1145\/3805712.3809956","relation":{},"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"2026-07-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}