{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T15:51:10Z","timestamp":1782834670264,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":53,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,26]],"date-time":"2023-10-26T00:00:00Z","timestamp":1698278400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62220106008, U20B2063, 62102070"],"award-info":[{"award-number":["62220106008, U20B2063, 62102070"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Sichuan Science and Technology Program","award":["2023NSFSC1392"],"award-info":[{"award-number":["2023NSFSC1392"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,26]]},"DOI":"10.1145\/3581783.3612101","type":"proceedings-article","created":{"date-parts":[[2023,10,27]],"date-time":"2023-10-27T07:26:54Z","timestamp":1698391614000},"page":"924-934","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":37,"title":["Your Negative May not Be True Negative: Boosting Image-Text Matching with False Negative Elimination"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-1355-6256","authenticated-orcid":false,"given":"HaoXuan","family":"Li","sequence":"first","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9714-8738","authenticated-orcid":false,"given":"Yi","family":"Bin","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-9180-1867","authenticated-orcid":false,"given":"Junrong","family":"Liao","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5070-4511","authenticated-orcid":false,"given":"Yang","family":"Yang","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2999-2088","authenticated-orcid":false,"given":"Heng Tao","family":"Shen","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2023,10,27]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"crossref","unstructured":"Peter Anderson Xiaodong He Chris Buehler Damien Teney Mark Johnson Stephen Gould and Lei Zhang. 2018. Bottom-up and top-down attention for image captioning and visual question answering. In CVPR. 6077--6086.","DOI":"10.1109\/CVPR.2018.00636"},{"key":"e_1_3_2_2_2_1","volume-title":"Introduction to probability","author":"Bertsekas Dimitri","unstructured":"Dimitri Bertsekas and John N Tsitsiklis. 2008. Introduction to probability. Vol. 1. Athena Scientific. 179--180 pages."},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"crossref","unstructured":"Yi Bin Xindi Shang Bo Peng Yujuan Ding and Tat-Seng Chua. 2021. Multi-perspective video captioning. In ACM Multimedia. 5110--5118.","DOI":"10.1145\/3474085.3475173"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"crossref","unstructured":"Yi Bin Wenhao Shi Jipeng Zhang Yujuan Ding Yang Yang and Heng Tao Shen. 2022. Non-Autoregressive Cross-Modal Coherence Modelling. In ACM Multimedia. 3253--3261.","DOI":"10.1145\/3503161.3548184"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"crossref","unstructured":"Yi Bin Yang Yang Jie Zhou Zi Huang and Heng Tao Shen. 2017. Adaptively attending to visual attributes and linguistic knowledge for captioning. In ACM Multimedia. 1345--1353.","DOI":"10.1145\/3123266.3123391"},{"key":"e_1_3_2_2_6_1","volume-title":"IMRAM: Iterative matching with recurrent attention memory for cross-modal image-text retrieval. In CVPR. 12655--12663.","author":"Chen Hui","year":"2020","unstructured":"Hui Chen, Guiguang Ding, Xudong Liu, Zijia Lin, Ji Liu, and Jungong Han. 2020b. IMRAM: Iterative matching with recurrent attention memory for cross-modal image-text retrieval. In CVPR. 12655--12663."},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"crossref","unstructured":"Jiacheng Chen Hexiang Hu Hao Wu Yuning Jiang and Changhu Wang. 2021. Learning the best pooling strategy for visual semantic embedding. In CVPR. 15789--15798.","DOI":"10.1109\/CVPR46437.2021.01553"},{"key":"e_1_3_2_2_8_1","volume-title":"Adaptive offline quintuplet loss for image-text matching","author":"Chen Tianlang","unstructured":"Tianlang Chen, Jiajun Deng, and Jiebo Luo. 2020a. Adaptive offline quintuplet loss for image-text matching. In ECCV. Springer, 549--565."},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"crossref","unstructured":"Weihua Chen Xiaotang Chen Jianguo Zhang and Kaiqi Huang. 2017. Beyond triplet loss: a deep quadruplet network for person re-identification. In CVPR. 403--412.","DOI":"10.1109\/CVPR.2017.145"},{"key":"e_1_3_2_2_10_1","volume-title":"Microsoft coco captions: Data collection and evaluation server. arXiv preprint arXiv:1504.00325","author":"Chen Xinlei","year":"2015","unstructured":"Xinlei Chen, Hao Fang, Tsung-Yi Lin, Ramakrishna Vedantam, Saurabh Gupta, Piotr Doll\u00e1r, and C Lawrence Zitnick. 2015. Microsoft coco captions: Data collection and evaluation server. arXiv preprint arXiv:1504.00325 (2015)."},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"crossref","unstructured":"Chaorui Deng Qi Wu Qingyao Wu Fuyuan Hu Fan Lyu and Mingkui Tan. 2018. Visual grounding via accumulated attention. In CVPR. 7746--7755.","DOI":"10.1109\/CVPR.2018.00808"},{"key":"e_1_3_2_2_12_1","volume-title":"BERT: Pre-training of deep bidirectional transformers for language understanding. In NAACL.","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. BERT: Pre-training of deep bidirectional transformers for language understanding. In NAACL."},{"key":"e_1_3_2_2_13_1","volume-title":"Wai Keung Wong, and Tat-Seng Chua","author":"Ding Yujuan","year":"2021","unstructured":"Yujuan Ding, Yunshan Ma, Wai Keung Wong, and Tat-Seng Chua. 2021. Leveraging two types of global graph for sequential fashion recommendation. In ICMR. 73--81."},{"key":"e_1_3_2_2_14_1","first-page":"103434","article-title":"Personalized fashion outfit generation with user coordination preference learning","volume":"60","author":"Ding Yujuan","year":"2023","unstructured":"Yujuan Ding, PY Mok, Yunshan Ma, and Yi Bin. 2023. Personalized fashion outfit generation with user coordination preference learning. IPM, Vol. 60, 5 (2023), 103434.","journal-title":"IPM"},{"key":"e_1_3_2_2_15_1","volume-title":"Words: Transformers for Image Recognition at Scale. In ICLR.","author":"Dosovitskiy Alexey","year":"2020","unstructured":"Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, Xiaohua Zhai, Thomas Unterthiner, Mostafa Dehghani, Matthias Minderer, Georg Heigold, Sylvain Gelly, et al. 2020. An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. In ICLR."},{"key":"e_1_3_2_2_16_1","volume-title":"Jamie Ryan Kiros, and Sanja Fidler","author":"Faghri Fartash","year":"2018","unstructured":"Fartash Faghri, David J Fleet, Jamie Ryan Kiros, and Sanja Fidler. 2018. Vse: Improving visual-semantic embeddings with hard negatives. In BMVC."},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"crossref","unstructured":"Yang Feng Lin Ma Wei Liu and Jiebo Luo. 2019. Unsupervised image captioning. In CVPR. 4125--4134.","DOI":"10.1109\/CVPR.2019.00425"},{"key":"e_1_3_2_2_18_1","volume-title":"Devise: A deep visual-semantic embedding model. In NeurIPS. 2121--2129.","author":"Frome Andrea","year":"2013","unstructured":"Andrea Frome, Greg Corrado, Jonathon Shlens, Samy Bengio, Jeffrey Dean, Marc'Aurelio Ranzato, and Tomas Mikolov. 2013. Devise: A deep visual-semantic embedding model. In NeurIPS. 2121--2129."},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"crossref","unstructured":"Xuri Ge Fuhai Chen Songpei Xu Fuxiang Tao and Joemon M Jose. 2023. Cross-modal Semantic Enhanced Interaction for Image-Sentence Retrieval. In WACV. 1022--1031.","DOI":"10.1109\/WACV56688.2023.00108"},{"key":"e_1_3_2_2_20_1","unstructured":"Kaiming He Haoqi Fan Yuxin Wu Saining Xie and Ross Girshick. 2020. Momentum contrast for unsupervised visual representation learning. In CVPR. 9729--9738."},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"crossref","unstructured":"Andrej Karpathy and Li Fei-Fei. 2015. Deep visual-semantic alignments for generating image descriptions. In CVPR. 3128--3137.","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"e_1_3_2_2_22_1","volume-title":"Adam: A method for stochastic optimization. In ICLR.","author":"Kingma Diederik P","year":"2015","unstructured":"Diederik P Kingma and Jimmy Ba. 2015. Adam: A method for stochastic optimization. In ICLR."},{"key":"e_1_3_2_2_23_1","volume-title":"Unifying visual-semantic embeddings with multimodal neural language models. arXiv preprint arXiv:1411.2539","author":"Kiros Ryan","year":"2014","unstructured":"Ryan Kiros, Ruslan Salakhutdinov, and Richard S Zemel. 2014. Unifying visual-semantic embeddings with multimodal neural language models. arXiv preprint arXiv:1411.2539 (2014)."},{"key":"e_1_3_2_2_24_1","unstructured":"Kuang-Huei Lee Xi Chen Gang Hua Houdong Hu and Xiaodong He. 2018. Stacked cross attention for image-text matching. In ECCV. 201--216."},{"key":"e_1_3_2_2_25_1","unstructured":"Kunpeng Li Yulun Zhang Kai Li Yuanyuan Li and Yun Fu. 2019. Visual semantic reasoning for image-text matching. In ICCV. 4654--4662."},{"key":"e_1_3_2_2_26_1","unstructured":"Zheng Li Caili Guo Zerun Feng Jenq-Neng Hwang and Xijun Xue. 2022. Multi-View Visual Semantic Embedding. IJCAI."},{"key":"e_1_3_2_2_27_1","volume-title":"Selectively Hard Negative Mining for Alleviating Gradient Vanishing in Image-Text Matching. arXiv preprint arXiv:2303.00181","author":"Li Zheng","year":"2023","unstructured":"Zheng Li, Caili Guo, Xin Wang, Zerun Feng, and Zhongtian Du. 2023. Selectively Hard Negative Mining for Alleviating Gradient Vanishing in Image-Text Matching. arXiv preprint arXiv:2303.00181 (2023)."},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"crossref","unstructured":"Chunxiao Liu Zhendong Mao An-An Liu Tianzhu Zhang Bin Wang and Yongdong Zhang. 2019. Focus your attention: A bidirectional focal attention network for image-text matching. In ACM Multimedia. 3--11.","DOI":"10.1145\/3343031.3350869"},{"key":"e_1_3_2_2_29_1","unstructured":"Chunxiao Liu Zhendong Mao Tianzhu Zhang Hongtao Xie Bin Wang and Yongdong Zhang. 2020. Graph structured network for image-text matching. In CVPR. 10921--10930."},{"key":"e_1_3_2_2_30_1","unstructured":"Tomas Mikolov Kai Chen Greg Corrado and Jeffrey Dean. 2013. Efficient estimation of word representations in vector space. In ICLR. 1--12."},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"crossref","unstructured":"Ishan Misra and Laurens van der Maaten. 2020. Self-supervised learning of pretext-invariant representations. In CVPR. 6707--6717.","DOI":"10.1109\/CVPR42600.2020.00674"},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"crossref","unstructured":"Hyun Oh Song Yu Xiang Stefanie Jegelka and Silvio Savarese. 2016. Deep metric learning via lifted structured feature embedding. In CVPR. 4004--4012.","DOI":"10.1109\/CVPR.2016.434"},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"crossref","unstructured":"Liang Peng Shuangji Yang Yi Bin and Guoqing Wang. 2021. Progressive graph attention network for video question answering. In ACM Multimedia. 2871--2879.","DOI":"10.1145\/3474085.3475193"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"crossref","unstructured":"Leigang Qu Meng Liu Da Cao Liqiang Nie and Qi Tian. 2020. Context-aware multi-view summarization network for image-text matching. In ACM Multimedia. 1047--1055.","DOI":"10.1145\/3394171.3413961"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"crossref","unstructured":"Leigang Qu Meng Liu Jianlong Wu Zan Gao and Liqiang Nie. 2021. Dynamic modality interaction modeling for image-text retrieval. In SIGIR. 1104--1113.","DOI":"10.1145\/3404835.3462829"},{"key":"e_1_3_2_2_36_1","unstructured":"Shaoqing Ren Kaiming He Ross Girshick and Jian Sun. 2015. Faster r-cnn: Towards real-time object detection with region proposal networks. In NeurIPS."},{"key":"e_1_3_2_2_37_1","volume-title":"Facenet: A unified embedding for face recognition and clustering. In CVPR. 815--823.","author":"Schroff Florian","year":"2015","unstructured":"Florian Schroff, Dmitry Kalenichenko, and James Philbin. 2015. Facenet: A unified embedding for face recognition and clustering. In CVPR. 815--823."},{"key":"e_1_3_2_2_38_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N Gomez \u0141ukasz Kaiser and Illia Polosukhin. 2017. Attention is all you need. In NeurIPS. 5998--6008."},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"crossref","unstructured":"Zheng Wang Zhenwei Gao Kangshuai Guo Yang Yang Xiaoming Wang and Heng Tao Shen. 2023. Multilateral Semantic Relations Modeling for Image Text Retrieval. In CVPR. 2830--2839.","DOI":"10.1109\/CVPR52729.2023.00277"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"crossref","unstructured":"Zheng Wang Zhenwei Gao Xing Xu Yadan Luo Yang Yang and Heng Tao Shen. 2022. Point to rectangle matching for image text retrieval. In ACM Multimedia. 4977--4986.","DOI":"10.1145\/3503161.3548237"},{"key":"e_1_3_2_2_41_1","volume-title":"Camp: Cross-modal adaptive message passing for text-image retrieval. In ICCV. 5764--5773.","author":"Wang Zihao","year":"2019","unstructured":"Zihao Wang, Xihui Liu, Hongsheng Li, Lu Sheng, Junjie Yan, Xiaogang Wang, and Jing Shao. 2019. Camp: Cross-modal adaptive message passing for text-image retrieval. In ICCV. 5764--5773."},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3088863"},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"crossref","unstructured":"Xi Wei Tianzhu Zhang Yan Li Yongdong Zhang and Feng Wu. 2020. Multi-modality cross attention network for image and sentence matching. In CVPR. 10941--10950.","DOI":"10.1109\/CVPR42600.2020.01095"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"crossref","unstructured":"Yiling Wu Shuhui Wang Guoli Song and Qingming Huang. 2019. Learning fragment self-attention embeddings for image-text matching. In ACM Multimedia. 2088--2096.","DOI":"10.1145\/3343031.3350940"},{"key":"e_1_3_2_2_45_1","volume-title":"Multi-Modal Transformer with Global-Local Alignment for Composed Query Image Retrieval. TMM","author":"Xu Yahui","year":"2023","unstructured":"Yahui Xu, Yi Bin, Jiwei Wei, Yang Yang, Guoqing Wang, and Heng Tao Shen. 2023. Multi-Modal Transformer with Global-Local Alignment for Composed Query Image Retrieval. TMM (2023)."},{"key":"e_1_3_2_2_46_1","volume-title":"Hard negative examples are hard, but useful","author":"Xuan Hong","unstructured":"Hong Xuan, Abby Stylianou, Xiaotong Liu, and Robert Pless. 2020b. Hard negative examples are hard, but useful. In ECCV. Springer, 126--142."},{"key":"e_1_3_2_2_47_1","doi-asserted-by":"crossref","unstructured":"Hong Xuan Abby Stylianou and Robert Pless. 2020a. Improved embeddings with easy positive triplet mining. In WACV. 2474--2482.","DOI":"10.1109\/WACV45572.2020.9093432"},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00166"},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"crossref","unstructured":"Baosheng Yu Tongliang Liu Mingming Gong Changxing Ding and Dacheng Tao. 2018. Correcting the triplet selection bias for triplet loss. In ECCV. 71--87.","DOI":"10.1007\/978-3-030-01231-1_5"},{"key":"e_1_3_2_2_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2022.3141603"},{"key":"e_1_3_2_2_51_1","doi-asserted-by":"crossref","unstructured":"Kun Zhang Zhendong Mao Quan Wang and Yongdong Zhang. 2022b. Negative-aware attention framework for image-text matching. In CVPR. 15661--15670.","DOI":"10.1109\/CVPR52688.2022.01521"},{"key":"e_1_3_2_2_52_1","doi-asserted-by":"crossref","unstructured":"Qi Zhang Zhen Lei Zhaoxiang Zhang and Stan Z Li. 2020. Context-aware attention network for image-text retrieval. In CVPR. 3536--3545.","DOI":"10.1109\/CVPR42600.2020.00359"},{"key":"e_1_3_2_2_53_1","volume-title":"USER: Unified Semantic Enhancement with Momentum Contrast for Image-Text Retrieval. arXiv preprint arXiv:2301.06844","author":"Zhang Yan","year":"2023","unstructured":"Yan Zhang, Zhong Ji, Di Wang, Yanwei Pang, and Xuelong Li. 2023. USER: Unified Semantic Enhancement with Momentum Contrast for Image-Text Retrieval. arXiv preprint arXiv:2301.06844 (2023)."}],"event":{"name":"MM '23: The 31st ACM International Conference on Multimedia","location":"Ottawa ON Canada","acronym":"MM '23","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 31st ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612101","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3581783.3612101","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T00:00:13Z","timestamp":1755820813000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612101"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,26]]},"references-count":53,"alternative-id":["10.1145\/3581783.3612101","10.1145\/3581783"],"URL":"https:\/\/doi.org\/10.1145\/3581783.3612101","relation":{},"subject":[],"published":{"date-parts":[[2023,10,26]]},"assertion":[{"value":"2023-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}