{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T02:43:43Z","timestamp":1783737823634,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":49,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,7,18]],"date-time":"2023-07-18T00:00:00Z","timestamp":1689638400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"the Defence Science and Technology Agency"},{"name":"the National Natural Sci- ence Foundation of China","award":["62006142"],"award-info":[{"award-number":["62006142"]}]},{"name":"the Special Fund for distinguished professors of Shandong Jianzhu University"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,7,19]]},"DOI":"10.1145\/3539618.3591712","type":"proceedings-article","created":{"date-parts":[[2023,7,19]],"date-time":"2023-07-19T00:22:59Z","timestamp":1689726179000},"page":"1252-1261","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":22,"title":["Learnable Pillar-based Re-ranking for Image-Text Retrieval"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-6555-3834","authenticated-orcid":false,"given":"Leigang","family":"Qu","sequence":"first","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1582-5764","authenticated-orcid":false,"given":"Meng","family":"Liu","sequence":"additional","affiliation":[{"name":"Shandong Jianzhu University, Jinan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5199-1428","authenticated-orcid":false,"given":"Wenjie","family":"Wang","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2434-9050","authenticated-orcid":false,"given":"Zhedong","family":"Zheng","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1476-0273","authenticated-orcid":false,"given":"Liqiang","family":"Nie","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology (Shenzhen), Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6097-7807","authenticated-orcid":false,"given":"Tat-Seng","family":"Chua","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2023,7,18]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Bottom-Up and Top-Down Attention for Image Captioning and Visual Question Answering","author":"Anderson Peter","unstructured":"Peter Anderson, Xiaodong He, Chris Buehler, Damien Teney, Mark Johnson, Stephen Gould, and Lei Zhang. 2018. Bottom-Up and Top-Down Attention for Image Captioning and Visual Question Answering. In CVPR. IEEE, 6077--6086."},{"key":"e_1_3_2_1_2_1","volume-title":"Three Things Everyone should Know to Improve Object Retrieval","author":"Arandjelovi\u0107 Relja","unstructured":"Relja Arandjelovi\u0107 and Andrew Zisserman. 2012. Three Things Everyone should Know to Improve Object Retrieval. In CVPR. IEEE, 2911--2918."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2016.2514498"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"crossref","unstructured":"Fei Cai Shangsong Liang and Maarten De Rijke. 2014. Personalized Document Re-ranking based on Bayesian Probabilistic Matrix Factorization. In SIGIR. ACM 835--838.","DOI":"10.1145\/2600428.2609453"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"crossref","unstructured":"Micael Carvalho R\u00e9mi Cad\u00e8ne David Picard Laure Soulier Nicolas Thome and Matthieu Cord. 2018. Cross-modal Retrieval in the Cooking Context: Learning Semantic Text-Image Embeddings. In SIGIR. ACM 35--44.","DOI":"10.1145\/3209978.3210036"},{"key":"e_1_3_2_1_6_1","volume-title":"IMRAM: Iterative Matching with Recurrent Attention Memory for Cross-Modal Image-Text Retrieval","author":"Chen Hui","year":"2020","unstructured":"Hui Chen, Guiguang Ding, Xudong Liu, Zijia Lin, Ji Liu, and Jungong Han. 2020. IMRAM: Iterative Matching with Recurrent Attention Memory for Cross-Modal Image-Text Retrieval. In CVPR. IEEE, 12655--12663."},{"key":"e_1_3_2_1_7_1","volume-title":"Learning the Best Pooling Strategy for Visual Semantic Embedding","author":"Chen Jiacheng","unstructured":"Jiacheng Chen, Hexiang Hu, Hao Wu, Yuning Jiang, and Changhu Wang. 2021. Learning the Best Pooling Strategy for Visual Semantic Embedding. In CVPR. IEEE, 15789--15798."},{"key":"e_1_3_2_1_8_1","volume-title":"Total Recall: Automatic Query Expansion with a Generative Feature Model for Object Retrieval","author":"Chum Ondrej","year":"2007","unstructured":"Ondrej Chum, James Philbin, Josef Sivic, Michael Isard, and Andrew Zisserman. 2007. Total Recall: Automatic Query Expansion with a Generative Feature Model for Object Retrieval. In ICCV. IEEE, 1--8."},{"key":"e_1_3_2_1_9_1","volume-title":"Similarity Reasoning and Filtration for Image-Text Matching","author":"Diao Haiwen","unstructured":"Haiwen Diao, Ying Zhang, Lin Ma, and Huchuan Lu. 2021. Similarity Reasoning and Filtration for Image-Text Matching. In AAAI. AAAI Press, 1218--1226."},{"key":"e_1_3_2_1_10_1","volume-title":"Jamie Ryan Kiros, and Sanja Fidler","author":"Faghri Fartash","year":"2018","unstructured":"Fartash Faghri, David J Fleet, Jamie Ryan Kiros, and Sanja Fidler. 2018. VSE: Improving Visual-Semantic Embeddings with Hard Negatives. In BMVC. BMVA Press, 1--13."},{"key":"e_1_3_2_1_11_1","volume-title":"NeurIPS. Curran Associates","author":"Frome Andrea","unstructured":"Andrea Frome, Greg S Corrado, Jon Shlens, Samy Bengio, Jeff Dean, Marc'Aurelio Ranzato, and Tomas Mikolov. 2013. DeViSE: A Deep Visual-Semantic Embedding Model. In NeurIPS. Curran Associates, Inc., 2121--2129."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-017-1016-8"},{"key":"e_1_3_2_1_13_1","volume-title":"Attention-based Query Expansion Learning","author":"Gordo Albert","unstructured":"Albert Gordo, Filip Radenovic, and Tamara Berg. 2020. Attention-based Query Expansion Learning. In ECCV. Springer, 172--188."},{"key":"e_1_3_2_1_14_1","volume-title":"Dimensionality Reduction by Learning an Invariant Mapping","author":"Hadsell Raia","unstructured":"Raia Hadsell, Sumit Chopra, and Yann LeCun. 2006. Dimensionality Reduction by Learning an Invariant Mapping. In CVPR. IEEE, 1735--1742."},{"key":"e_1_3_2_1_15_1","volume-title":"Fast Spectral Ranking for Similarity Search","author":"Iscen Ahmet","unstructured":"Ahmet Iscen, Yannis Avrithis, Giorgos Tolias, Teddy Furon, and Ondvr ej Chum. 2018. Fast Spectral Ranking for Similarity Search. In CVPR. IEEE, 7632--7641."},{"key":"e_1_3_2_1_16_1","volume-title":"Efficient Diffusion on Region Manifolds: Recovering Small Objects with Compact CNN Representations","author":"Iscen Ahmet","year":"2077","unstructured":"Ahmet Iscen, Giorgos Tolias, Yannis Avrithis, Teddy Furon, and Ondrej Chum. 2017. Efficient Diffusion on Region Manifolds: Recovering Small Objects with Compact CNN Representations. In CVPR. IEEE, 2077--2086."},{"key":"e_1_3_2_1_17_1","volume-title":"Stacked Cross Attention for Image-Text Matching","author":"Lee Kuang-Huei","unstructured":"Kuang-Huei Lee, Xi Chen, Gang Hua, Houdong Hu, and Xiaodong He. 2018. Stacked Cross Attention for Image-Text Matching. In ECCV. Springer, 201--216."},{"key":"e_1_3_2_1_18_1","volume-title":"Reducing Background Induced Domain Shift for Adaptive Person Re-Identification. TII","author":"Lei Jianjun","year":"2022","unstructured":"Jianjun Lei, Tianyi Qin, Bo Peng, Wanqing Li, Zhaoqing Pan, Haifeng Shen, and Sam Kwong. 2022. Reducing Background Induced Domain Shift for Adaptive Person Re-Identification. TII (2022), 1--12."},{"key":"e_1_3_2_1_19_1","volume-title":"Visual Semantic Reasoning for Image-Text Matching","author":"Li Kunpeng","unstructured":"Kunpeng Li, Yulun Zhang, Kai Li, Yuanyuan Li, and Yun Fu. 2019. Visual Semantic Reasoning for Image-Text Matching. In ICCV. IEEE, 4654--4662."},{"key":"e_1_3_2_1_20_1","volume-title":"Microsoft COCO: Common Objects in Context","author":"Lin Tsung-Yi","unstructured":"Tsung-Yi Lin, Michael Maire, Serge Belongie, James Hays, Pietro Perona, Deva Ramanan, Piotr Doll\u00e1r, and C Lawrence Zitnick. 2014. Microsoft COCO: Common Objects in Context. In ECCV. Springer, 740--755."},{"key":"e_1_3_2_1_21_1","volume-title":"Graph Structured Network for Image-Text Matching","author":"Liu Chunxiao","unstructured":"Chunxiao Liu, Zhendong Mao, Tianzhu Zhang, Hongtao Xie, Bin Wang, and Yongdong Zhang. 2020. Graph Structured Network for Image-Text Matching. In CVPR. IEEE, 10921--10930."},{"key":"e_1_3_2_1_22_1","volume-title":"NeurIPS. Curran Associates","author":"Liu Chundi","unstructured":"Chundi Liu, Guangwei Yu, Maksims Volkovs, Cheng Chang, Himanshu Rai, Junwei Ma, and Satya Krishna Gorti. 2019. Guided Similarity Separation for Image Retrieval. In NeurIPS. Curran Associates, Inc., 1--12."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"crossref","unstructured":"Meng Liu Xiang Wang Liqiang Nie Xiangnan He Baoquan Chen and Tat-Seng Chua. 2018. Attentive Moment Retrieval in Videos. In SIGIR. ACM 15--24.","DOI":"10.1145\/3209978.3210003"},{"key":"e_1_3_2_1_24_1","volume-title":"Raffaele Perego, Nicola Tonellotto, Nazli Goharian, and Ophir Frieder.","author":"MacAvaney Sean","year":"2020","unstructured":"Sean MacAvaney, Franco Maria Nardini, Raffaele Perego, Nicola Tonellotto, Nazli Goharian, and Ophir Frieder. 2020a. Efficient Document Re-ranking for Transformers by Precomputing Term Representations. In SIGIR. ACM, 49--58."},{"key":"e_1_3_2_1_25_1","volume-title":"Raffaele Perego, Nicola Tonellotto, Nazli Goharian, and Ophir Frieder.","author":"MacAvaney Sean","year":"2020","unstructured":"Sean MacAvaney, Franco Maria Nardini, Raffaele Perego, Nicola Tonellotto, Nazli Goharian, and Ophir Frieder. 2020b. Training curricula for open domain answer re-ranking. In SIGIR. ACM, 529--538."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"crossref","unstructured":"Sean MacAvaney Andrew Yates Kai Hui and Ophir Frieder. 2019. Content-based Weak Supervision for Ad-hoc Re-ranking. In SIGIR. ACM 993--996.","DOI":"10.1145\/3331184.3331316"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"crossref","unstructured":"Yoshitomo Matsubara Thuy Vu and Alessandro Moschitti. 2020. Reranking for Efficient Transformer-based Answer Selection. In SIGIR. ACM 1577--1580.","DOI":"10.1145\/3397271.3401266"},{"key":"e_1_3_2_1_28_1","volume-title":"NeurIPS. Curran Associates","author":"Ouyang Jianbo","unstructured":"Jianbo Ouyang, Hui Wu, Min Wang, Wengang Zhou, and Houqiang Li. 2021. Contextual Similarity Aggregation with Self-attention for Visual Re-ranking. In NeurIPS. Curran Associates, Inc., 3135--3148."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3397271.3401104"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2018.2876833"},{"key":"e_1_3_2_1_31_1","volume-title":"Lost in Quantization: Improving Particular Object Retrieval in Large Scale Image Databases","author":"Philbin James","unstructured":"James Philbin, Ondrej Chum, Michael Isard, Josef Sivic, and Andrew Zisserman. 2008. Lost in Quantization: Improving Particular Object Retrieval in Large Scale Image Databases. In CVPR. IEEE, 1--8."},{"key":"e_1_3_2_1_32_1","volume-title":"Hello Neighbor: Accurate Object Retrieval with K-reciprocal Nearest Neighbors","author":"Qin Danfeng","year":"2011","unstructured":"Danfeng Qin, Stephan Gammeter, Lukas Bossard, Till Quack, and Luc Van Gool. 2011. Hello Neighbor: Accurate Object Retrieval with K-reciprocal Nearest Neighbors. In CVPR. IEEE, 777--784."},{"key":"e_1_3_2_1_33_1","unstructured":"Leigang Qu Meng Liu Da Cao Liqiang Nie and Qi Tian. 2020. Context-Aware Multi-View Summarization Network for Image-Text Matching. In MM. ACM 1047--1055."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"crossref","unstructured":"Leigang Qu Meng Liu Jianlong Wu Zan Gao and Liqiang Nie. 2021. Dynamic Modality Interaction Modeling for Image-Text Retrieval. In SIGIR. ACM 1104--1113.","DOI":"10.1145\/3404835.3462829"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2018.2846566"},{"key":"e_1_3_2_1_36_1","volume-title":"Learning with Average Precision: Training Image Retrieval with a Listwise Loss","author":"Revaud Jerome","unstructured":"Jerome Revaud, Jon Almaz\u00e1n, Rafael S Rezende, and Cesar Roberto de Souza. 2019. Learning with Average Precision: Training Image Retrieval with a Listwise Loss. In ICCV. IEEE, 5107--5116."},{"key":"e_1_3_2_1_37_1","volume-title":"Facenet: A Unified Embedding for Face Recognition and Clustering","author":"Schroff Florian","year":"2015","unstructured":"Florian Schroff, Dmitry Kalenichenko, and James Philbin. 2015. Facenet: A Unified Embedding for Face Recognition and Clustering. In CVPR. IEEE, 815--823."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/3343031.3350875"},{"key":"e_1_3_2_1_39_1","volume-title":"Multi-Modality Cross Attention Network for Image and Sentence Matching","author":"Wei Xi","unstructured":"Xi Wei, Tianzhu Zhang, Yan Li, Yongdong Zhang, and Feng Wu. 2020. Multi-Modality Cross Attention Network for Image and Sentence Matching. In CVPR. IEEE, 10941--10950."},{"key":"e_1_3_2_1_40_1","unstructured":"Haokun Wen Xuemeng Song Xin Yang Yibing Zhan and Liqiang Nie. 2021. Comprehensive Linguistic-visual Composition Network for Image Retrieval. In SIGIR. ACM 1369--1378."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"crossref","unstructured":"Yunjia Xi Weiwen Liu Jieming Zhu Xilong Zhao Xinyi Dai Ruiming Tang Weinan Zhang Rui Zhang and Yong Yu. 2022. Multi-Level Interaction Reranking with User Behavior History. In SIGIR. ACM 1336--1346.","DOI":"10.1145\/3477495.3532026"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00166"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"crossref","unstructured":"George Zerveas Navid Rekabsaz Daniel Cohen and Carsten Eickhoff. 2022. Mitigating Bias in Search Results Through Contextual Document Reranking and Neutrality Regularization. In SIGIR. ACM 2532--2538.","DOI":"10.1145\/3477495.3531891"},{"key":"e_1_3_2_1_44_1","volume-title":"Understanding Image Retrieval Re-ranking: A Graph Neural Network Perspective. arXiv:2012.07620","author":"Zhang Xuanmeng","year":"2020","unstructured":"Xuanmeng Zhang, Minyue Jiang, Zhedong Zheng, Xiao Tan, Errui Ding, and Yi Yang. 2020. Understanding Image Retrieval Re-ranking: A Graph Neural Network Perspective. arXiv:2012.07620 (2020)."},{"key":"e_1_3_2_1_45_1","volume-title":"Deep Cross-Modal Projection Learning for Image-Text Matching","author":"Zhang Ying","unstructured":"Ying Zhang and Huchuan Lu. 2018. Deep Cross-Modal Projection Learning for Image-Text Matching. In ECCV. Springer, 686--701."},{"key":"e_1_3_2_1_46_1","first-page":"1520","article-title":"VehicleNet","volume":"23","author":"Zheng Zhedong","year":"2020","unstructured":"Zhedong Zheng, Tao Ruan, Yunchao Wei, Yi Yang, and Tao Mei. 2020a. VehicleNet: Learning Robust Visual Representation for Vehicle Re-identification. TMM, Vol. 23, 1520--9210 (2020), 2683--2693.","journal-title":"Learning Robust Visual Representation for Vehicle Re-identification. TMM"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/3383184"},{"key":"e_1_3_2_1_48_1","volume-title":"Re-ranking Person Re-identification with K-reciprocal Encoding","author":"Zhong Zhun","unstructured":"Zhun Zhong, Liang Zheng, Donglin Cao, and Shaozi Li. 2017. Re-ranking Person Re-identification with K-reciprocal Encoding. In CVPR. IEEE, 1318--1327."},{"key":"e_1_3_2_1_49_1","volume-title":"TILDE: Term Independent Likelihood Model for Passage Re-ranking. In SIGIR. ACM, 1483--1492.","author":"Zhuang Shengyao","year":"2021","unstructured":"Shengyao Zhuang and Guido Zuccon. 2021. TILDE: Term Independent Likelihood Model for Passage Re-ranking. In SIGIR. ACM, 1483--1492."}],"event":{"name":"SIGIR '23: The 46th International ACM SIGIR Conference on Research and Development in Information Retrieval","location":"Taipei Taiwan","acronym":"SIGIR '23","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 46th International ACM SIGIR Conference on Research and Development in Information Retrieval"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3539618.3591712","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3539618.3591712","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T16:47:00Z","timestamp":1750178820000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3539618.3591712"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,7,18]]},"references-count":49,"alternative-id":["10.1145\/3539618.3591712","10.1145\/3539618"],"URL":"https:\/\/doi.org\/10.1145\/3539618.3591712","relation":{},"subject":[],"published":{"date-parts":[[2023,7,18]]},"assertion":[{"value":"2023-07-18","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}