{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T15:43:23Z","timestamp":1783439003950,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":75,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,26]],"date-time":"2023-10-26T00:00:00Z","timestamp":1698278400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Sichuan Science and Technology Program","award":["2023NSFSC1392"],"award-info":[{"award-number":["2023NSFSC1392"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62102070, 62220106008, U20B2063"],"award-info":[{"award-number":["62102070, 62220106008, U20B2063"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,26]]},"DOI":"10.1145\/3581783.3612427","type":"proceedings-article","created":{"date-parts":[[2023,10,27]],"date-time":"2023-10-27T07:27:40Z","timestamp":1698391660000},"page":"3041-3050","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":30,"title":["Unifying Two-Stream Encoders with Transformers for Cross-Modal Retrieval"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9714-8738","authenticated-orcid":false,"given":"Yi","family":"Bin","sequence":"first","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-1355-6256","authenticated-orcid":false,"given":"Haoxuan","family":"Li","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1123-6129","authenticated-orcid":false,"given":"Yahui","family":"Xu","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5685-3123","authenticated-orcid":false,"given":"Xing","family":"Xu","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5070-4511","authenticated-orcid":false,"given":"Yang","family":"Yang","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2999-2088","authenticated-orcid":false,"given":"Heng Tao","family":"Shen","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2023,10,27]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"Vqa: Visual question answering. In ICCV. 2425--2433.","author":"Antol Stanislaw","year":"2015","unstructured":"Stanislaw Antol, Aishwarya Agrawal, Jiasen Lu, Margaret Mitchell, Dhruv Batra, C Lawrence Zitnick, and Devi Parikh. 2015. Vqa: Visual question answering. In ICCV. 2425--2433."},{"key":"e_1_3_2_2_2_1","unstructured":"Dzmitry Bahdanau Kyunghyun Cho and Yoshua Bengio. 2015. Neural machine translation by jointly learning to align and translate. In ICLR."},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"crossref","unstructured":"Yi Bin Xindi Shang Bo Peng Yujuan Ding and Tat-Seng Chua. 2021. Multi-Perspective Video Captioning. In ACM Multimedia. 5110--5118.","DOI":"10.1145\/3474085.3475173"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"crossref","unstructured":"Yi Bin Yang Yang Jie Zhou Zi Huang and Heng Tao Shen. 2017. Adaptively attending to visual attributes and linguistic knowledge for captioning. In ACM Multimedia. 1345--1353.","DOI":"10.1145\/3123266.3123391"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"crossref","unstructured":"David M Blei and Michael I Jordan. 2003. Modeling annotated data. In SIGIR. 127--134.","DOI":"10.1145\/860435.860460"},{"key":"e_1_3_2_2_6_1","unstructured":"Tom Brown Benjamin Mann Nick Ryder Melanie Subbiah Jared D Kaplan Prafulla Dhariwal Arvind Neelakantan Pranav Shyam Girish Sastry Amanda Askell et al. 2020. Language models are few-shot learners. In NeurIPS."},{"key":"e_1_3_2_2_7_1","volume-title":"Inter-Intra Modal Representation Augmentation with DCT-Transformer Adversarial Network for Image-Text Matching. TMM","author":"Chen Chen","year":"2023","unstructured":"Chen Chen, Dan Wang, Bin Song, and Hao Tan. 2023. Inter-Intra Modal Representation Augmentation with DCT-Transformer Adversarial Network for Image-Text Matching. TMM (2023)."},{"key":"e_1_3_2_2_8_1","volume-title":"IMRAM: Iterative matching with recurrent attention memory for cross-modal image-text retrieval. In CVPR. 12655--12663.","author":"Chen Hui","year":"2020","unstructured":"Hui Chen, Guiguang Ding, Xudong Liu, Zijia Lin, Ji Liu, and Jungong Han. 2020a. IMRAM: Iterative matching with recurrent attention memory for cross-modal image-text retrieval. In CVPR. 12655--12663."},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"crossref","unstructured":"Jiacheng Chen Hexiang Hu Hao Wu Yuning Jiang and Changhu Wang. 2021. Learning the best pooling strategy for visual semantic embedding. In CVPR. 15789--15798.","DOI":"10.1109\/CVPR46437.2021.01553"},{"key":"e_1_3_2_2_10_1","unstructured":"Mark Chen Alec Radford Rewon Child Jeffrey Wu Heewoo Jun David Luan and Ilya Sutskever. 2020c. Generative pretraining from pixels. In ICML. PMLR 1691--1703."},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"crossref","unstructured":"Shizhe Chen Bei Liu Jianlong Fu Ruihua Song Qin Jin Pingping Lin Xiaoyu Qi Chunting Wang and Jin Zhou. 2019. Neural storyboard artist: Visualizing stories with coherent image sequences. In ACM Multimedia. 2236--2244.","DOI":"10.1145\/3343031.3350571"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6631"},{"key":"e_1_3_2_2_13_1","volume-title":"Microsoft coco captions: Data collection and evaluation server. arXiv preprint arXiv:1504.00325","author":"Chen Xinlei","year":"2015","unstructured":"Xinlei Chen, Hao Fang, Tsung-Yi Lin, Ramakrishna Vedantam, Saurabh Gupta, Piotr Doll\u00e1r, and C Lawrence Zitnick. 2015. Microsoft coco captions: Data collection and evaluation server. arXiv preprint arXiv:1504.00325 (2015)."},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"crossref","unstructured":"Yanbei Chen Shaogang Gong and Loris Bazzani. 2020b. Image search with text feedback by visiolinguistic attention learning. In CVPR. 3001--3011.","DOI":"10.1109\/CVPR42600.2020.00307"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"crossref","unstructured":"Wietse de Vries Andreas van Cranenburgh and Malvina Nissim. 2020. What's so special about BERT's layers? A closer look at the NLP pipeline in monolingual and multilingual models. In Findings of EMNLP. 4339--4350.","DOI":"10.18653\/v1\/2020.findings-emnlp.389"},{"key":"e_1_3_2_2_16_1","volume-title":"Words: Transformers for Image Recognition at Scale. In ICLR.","author":"Dosovitskiy Alexey","year":"2020","unstructured":"Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, Xiaohua Zhai, Thomas Unterthiner, Mostafa Dehghani, Matthias Minderer, Georg Heigold, Sylvain Gelly, et al. 2020. An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. In ICLR."},{"key":"e_1_3_2_2_17_1","volume-title":"Jamie Ryan Kiros, and Sanja Fidler","author":"Faghri Fartash","year":"2018","unstructured":"Fartash Faghri, David J Fleet, Jamie Ryan Kiros, and Sanja Fidler. 2018. Vse: Improving visual-semantic embeddings with hard negatives. In BMVC."},{"key":"e_1_3_2_2_18_1","volume-title":"Devise: A deep visual-semantic embedding model. In NeurIPS. 2121--2129.","author":"Frome Andrea","year":"2013","unstructured":"Andrea Frome, Greg Corrado, Jonathon Shlens, Samy Bengio, Jeffrey Dean, Marc'Aurelio Ranzato, and Tomas Mikolov. 2013. Devise: A deep visual-semantic embedding model. In NeurIPS. 2121--2129."},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"crossref","unstructured":"Ross Girshick Jeff Donahue Trevor Darrell and Jitendra Malik. 2014. Rich feature hierarchies for accurate object detection and semantic segmentation. In CVPR. 580--587.","DOI":"10.1109\/CVPR.2014.81"},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-013-0658-4"},{"key":"e_1_3_2_2_21_1","unstructured":"Kaiming He Xinlei Chen Saining Xie Yanghao Li Piotr Doll\u00e1r and Ross Girshick. 2022. Masked autoencoders are scalable vision learners. In CVPR. 16000--16009."},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"crossref","unstructured":"Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Deep residual learning for image recognition. In CVPR. 770--778.","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"crossref","unstructured":"Jeremy Howard and Sebastian Ruder. 2018. Universal Language Model Fine-tuning for Text Classification. In ACL. 328--339.","DOI":"10.18653\/v1\/P18-1031"},{"key":"e_1_3_2_2_24_1","volume-title":"Learning cross-modality similarity for multinomial data","author":"Jia Yangqing","unstructured":"Yangqing Jia, Mathieu Salzmann, and Trevor Darrell. 2011. Learning cross-modality similarity for multinomial data. In ICCV. IEEE, 2407--2414."},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"crossref","unstructured":"Andrej Karpathy and Li Fei-Fei. 2015. Deep visual-semantic alignments for generating image descriptions. In CVPR. 3128--3137.","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"e_1_3_2_2_26_1","volume-title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In NAACL-HLT. 4171--4186.","author":"Ming-Wei Chang Jacob Devlin","year":"2019","unstructured":"Jacob Devlin Ming-Wei Chang Kenton and Lee Kristina Toutanova. 2019. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In NAACL-HLT. 4171--4186."},{"key":"e_1_3_2_2_27_1","volume-title":"Adam: A method for stochastic optimization. In ICLR.","author":"Kingma Diederik P","year":"2015","unstructured":"Diederik P Kingma and Jimmy Ba. 2015. Adam: A method for stochastic optimization. In ICLR."},{"key":"e_1_3_2_2_28_1","volume-title":"Unifying visual-semantic embeddings with multimodal neural language models. arXiv preprint arXiv:1411.2539","author":"Kiros Ryan","year":"2014","unstructured":"Ryan Kiros, Ruslan Salakhutdinov, and Richard S Zemel. 2014. Unifying visual-semantic embeddings with multimodal neural language models. arXiv preprint arXiv:1411.2539 (2014)."},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"crossref","unstructured":"Benjamin Klein Guy Lev Gil Sadeh and Lior Wolf. 2015. Associating neural word embeddings with deep image representations using fisher vectors. In CVPR. 4437--4446.","DOI":"10.1109\/CVPR.2015.7299073"},{"key":"e_1_3_2_2_30_1","volume-title":"ALBERT: A Lite BERT for Self-supervised Learning of Language Representations. In ICLR.","author":"Lan Zhenzhong","year":"2019","unstructured":"Zhenzhong Lan, Mingda Chen, Sebastian Goodman, Kevin Gimpel, Piyush Sharma, and Radu Soricut. 2019. ALBERT: A Lite BERT for Self-supervised Learning of Language Representations. In ICLR."},{"key":"e_1_3_2_2_31_1","unstructured":"Kuang-Huei Lee Xi Chen Gang Hua Houdong Hu and Xiaodong He. 2018. Stacked cross attention for image-text matching. In ECCV. 201--216."},{"key":"e_1_3_2_2_32_1","volume-title":"BART: Denoising Sequence-to-Sequence Pre-training for Natural Language Generation, Translation, and Comprehension. In ACL. 7871--7880.","author":"Lewis Mike","year":"2020","unstructured":"Mike Lewis, Yinhan Liu, Naman Goyal, Marjan Ghazvininejad, Abdelrahman Mohamed, Omer Levy, Veselin Stoyanov, and Luke Zettlemoyer. 2020. BART: Denoising Sequence-to-Sequence Pre-training for Natural Language Generation, Translation, and Comprehension. In ACL. 7871--7880."},{"key":"e_1_3_2_2_33_1","unstructured":"Haoxuan Li Yi Bin Junrong Liao Yang Yang and Heng Tao Shen. 2023. Your Negative May not Be True Negative: Boosting Image-Text Matching with False Negative Elimination. In ACM Multimedia."},{"key":"e_1_3_2_2_34_1","unstructured":"Kunpeng Li Yulun Zhang Kai Li Yuanyuan Li and Yun Fu. 2019. Visual semantic reasoning for image-text matching. In ICCV. 4654--4662."},{"key":"e_1_3_2_2_35_1","volume-title":"UNIMO: Towards Unified-Modal Understanding and Generation via Cross-Modal Contrastive Learning. In ACL-IJCNLP. 2592--2607.","author":"Li Wei","year":"2021","unstructured":"Wei Li, Can Gao, Guocheng Niu, Xinyan Xiao, Hao Liu, Jiachen Liu, Hua Wu, and Haifeng Wang. 2021. UNIMO: Towards Unified-Modal Understanding and Generation via Cross-Modal Contrastive Learning. In ACL-IJCNLP. 2592--2607."},{"key":"e_1_3_2_2_36_1","volume-title":"Oscar: Object-semantics aligned pre-training for vision-language tasks","author":"Li Xiujun","year":"2020","unstructured":"Xiujun Li, Xi Yin, Chunyuan Li, Pengchuan Zhang, Xiaowei Hu, Lei Zhang, Lijuan Wang, Houdong Hu, Li Dong, Furu Wei, et al. 2020. Oscar: Object-semantics aligned pre-training for vision-language tasks. In ECCV. Springer, 121--137."},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"crossref","unstructured":"Chunxiao Liu Zhendong Mao An-An Liu Tianzhu Zhang Bin Wang and Yongdong Zhang. 2019. Focus your attention: A bidirectional focal attention network for image-text matching. In ACM Multimedia. 3--11.","DOI":"10.1145\/3343031.3350869"},{"key":"e_1_3_2_2_38_1","unstructured":"Chunxiao Liu Zhendong Mao Tianzhu Zhang Hongtao Xie Bin Wang and Yongdong Zhang. 2020. Graph structured network for image-text matching. In CVPR. 10921--10930."},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"crossref","unstructured":"Ze Liu Yutong Lin Yue Cao Han Hu Yixuan Wei Zheng Zhang Stephen Lin and Baining Guo. 2021. Swin transformer: Hierarchical vision transformer using shifted windows. In ICCV. 10012--10022.","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-016-0911-8"},{"key":"e_1_3_2_2_41_1","unstructured":"Tomas Mikolov Kai Chen Greg Corrado and Jeffrey Dean. 2013. Efficient estimation of word representations in vector space. In ICLR. 1--12."},{"key":"e_1_3_2_2_42_1","unstructured":"Zhenxing Niu Mo Zhou Le Wang Xinbo Gao and Gang Hua. 2017. Hierarchical multimodal lstm for dense visual-semantic embedding. In ICCV. 1881--1889."},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"crossref","unstructured":"Liang Peng Shuangji Yang Yi Bin and Guoqing Wang. 2021. Progressive graph attention network for video question answering. In ACM Multimedia. 2871--2879.","DOI":"10.1145\/3474085.3475193"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2013.142"},{"key":"e_1_3_2_2_45_1","volume-title":"Topic regression multi-modal latent dirichlet allocation for image annotation","author":"Putthividhy Duangmanee","unstructured":"Duangmanee Putthividhy, Hagai T Attias, and Srikantan S Nagarajan. 2010. Topic regression multi-modal latent dirichlet allocation for image annotation. In CVPR. IEEE, 3408--3415."},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"crossref","unstructured":"Leigang Qu Meng Liu Da Cao Liqiang Nie and Qi Tian. 2020. Context-aware multi-view summarization network for image-text matching. In ACM Multimedia. 1047--1055.","DOI":"10.1145\/3394171.3413961"},{"key":"e_1_3_2_2_47_1","doi-asserted-by":"crossref","unstructured":"Leigang Qu Meng Liu Jianlong Wu Zan Gao and Liqiang Nie. 2021. Dynamic modality interaction modeling for image-text retrieval. In SIGIR. 1104--1113.","DOI":"10.1145\/3404835.3462829"},{"key":"e_1_3_2_2_48_1","first-page":"1","article-title":"Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer","volume":"21","author":"Raffel Colin","year":"2020","unstructured":"Colin Raffel, Noam Shazeer, Adam Roberts, Katherine Lee, Sharan Narang, Michael Matena, Yanqi Zhou, Wei Li, and Peter J Liu. 2020. Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer. JMLR, Vol. 21 (2020), 1--67.","journal-title":"JMLR"},{"key":"e_1_3_2_2_49_1","volume-title":"Emanuele Coviello, Gabriel Doyle, Gert RG Lanckriet, Roger Levy, and Nuno Vasconcelos.","author":"Rasiwasia Nikhil","year":"2010","unstructured":"Nikhil Rasiwasia, Jose Costa Pereira, Emanuele Coviello, Gabriel Doyle, Gert RG Lanckriet, Roger Levy, and Nuno Vasconcelos. 2010. A new approach to cross-modal multimedia retrieval. In ACM Multimedia. 251--260."},{"key":"e_1_3_2_2_50_1","unstructured":"Shaoqing Ren Kaiming He Ross Girshick and Jian Sun. 2015. Faster r-cnn: Towards real-time object detection with region proposal networks. In NeurIPS."},{"key":"e_1_3_2_2_51_1","first-page":"3351","article-title":"Exploiting subspace relation in semantic labels for cross-modal hashing","volume":"33","author":"Shen Heng Tao","year":"2020","unstructured":"Heng Tao Shen, Luchen Liu, Yang Yang, Xing Xu, Zi Huang, Fumin Shen, and Richang Hong. 2020. Exploiting subspace relation in semantic labels for cross-modal hashing. IEEE TKDE, Vol. 33, 10 (2020), 3351--3365.","journal-title":"IEEE TKDE"},{"key":"e_1_3_2_2_52_1","doi-asserted-by":"crossref","unstructured":"Betty Van Aken Benjamin Winter Alexander L\u00f6ser and Felix A Gers. 2019. How does bert answer questions? a layer-wise analysis of transformer representations. In CIKM. 1823--1832.","DOI":"10.1145\/3357384.3358028"},{"key":"e_1_3_2_2_53_1","volume-title":"JMLR","volume":"9","author":"der Maaten Laurens Van","year":"2008","unstructured":"Laurens Van der Maaten and Geoffrey Hinton. 2008. Visualizing data using t-SNE. JMLR, Vol. 9, 11 (2008)."},{"key":"e_1_3_2_2_54_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N Gomez ?ukasz Kaiser and Illia Polosukhin. 2017. Attention is all you need. In NeurIPS. 5998--6008."},{"key":"e_1_3_2_2_55_1","doi-asserted-by":"crossref","unstructured":"Bokun Wang Yang Yang Xing Xu Alan Hanjalic and Heng Tao Shen. 2017. Adversarial cross-modal retrieval. In ACM Multimedia. 154--162.","DOI":"10.1145\/3123266.3123326"},{"key":"e_1_3_2_2_56_1","volume-title":"Bevt: Bert pretraining of video transformers. arXiv preprint arXiv:2112.01529","author":"Wang Rui","year":"2021","unstructured":"Rui Wang, Dongdong Chen, Zuxuan Wu, Yinpeng Chen, Xiyang Dai, Mengchen Liu, Yu-Gang Jiang, Luowei Zhou, and Lu Yuan. 2021a. Bevt: Bert pretraining of video transformers. arXiv preprint arXiv:2112.01529 (2021)."},{"key":"e_1_3_2_2_57_1","doi-asserted-by":"crossref","unstructured":"Sijin Wang Ruiping Wang Ziwei Yao Shiguang Shan and Xilin Chen. 2020. Cross-modal scene graph matching for relationship-aware image-text retrieval. In WACV. 1508--1517.","DOI":"10.1109\/WACV45572.2020.9093614"},{"key":"e_1_3_2_2_58_1","doi-asserted-by":"crossref","unstructured":"Yuqing Wang Zhaoliang Xu Xinlong Wang Chunhua Shen Baoshan Cheng Hao Shen and Huaxia Xia. 2021b. End-to-end video instance segmentation with transformers. In CVPR. 8741--8750.","DOI":"10.1109\/CVPR46437.2021.00863"},{"key":"e_1_3_2_2_59_1","doi-asserted-by":"crossref","unstructured":"Zheng Wang Zhenwei Gao Kangshuai Guo Yang Yang Xiaoming Wang and Heng Tao Shen. 2023 a. Multilateral Semantic Relations Modeling for Image Text Retrieval. In CVPR. 2830--2839.","DOI":"10.1109\/CVPR52729.2023.00277"},{"key":"e_1_3_2_2_60_1","volume-title":"Camp: Cross-modal adaptive message passing for text-image retrieval. In ICCV. 5764--5773.","author":"Wang Zihao","year":"2019","unstructured":"Zihao Wang, Xihui Liu, Hongsheng Li, Lu Sheng, Junjie Yan, Xiaogang Wang, and Jing Shao. 2019. Camp: Cross-modal adaptive message passing for text-image retrieval. In ICCV. 5764--5773."},{"key":"e_1_3_2_2_61_1","volume-title":"2023 b. Quaternion relation embedding for scene graph generation. TMM","author":"Wang Zheng","year":"2023","unstructured":"Zheng Wang, Xing Xu, Guoqing Wang, Yang Yang, and Heng Tao Shen. 2023 b. Quaternion relation embedding for scene graph generation. TMM (2023)."},{"key":"e_1_3_2_2_62_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6915"},{"key":"e_1_3_2_2_63_1","doi-asserted-by":"crossref","unstructured":"Xi Wei Tianzhu Zhang Yan Li Yongdong Zhang and Feng Wu. 2020. Multi-modality cross attention network for image and sentence matching. In CVPR. 10941--10950.","DOI":"10.1109\/CVPR42600.2020.01095"},{"key":"e_1_3_2_2_64_1","doi-asserted-by":"crossref","unstructured":"Yiling Wu Shuhui Wang Guoli Song and Qingming Huang. 2019. Learning fragment self-attention embeddings for image-text matching. In ACM Multimedia. 2088--2096.","DOI":"10.1145\/3343031.3350940"},{"key":"e_1_3_2_2_65_1","unstructured":"Enze Xie Wenhai Wang Zhiding Yu Anima Anandkumar Jose M Alvarez and Ping Luo. 2021. SegFormer: Simple and efficient design for semantic segmentation with transformers. In NeurIPS."},{"key":"e_1_3_2_2_66_1","first-page":"5412","article-title":"Cross-modal attention with semantic consistence for image-text matching","volume":"31","author":"Xu Xing","year":"2020","unstructured":"Xing Xu, Tan Wang, Yang Yang, Lin Zuo, Fumin Shen, and Heng Tao Shen. 2020. Cross-modal attention with semantic consistence for image-text matching. IEEE TNNLS, Vol. 31, 12 (2020), 5412--5425.","journal-title":"IEEE TNNLS"},{"key":"e_1_3_2_2_67_1","volume-title":"Multi-Modal Transformer with Global-Local Alignment for Composed Query Image Retrieval. TMM","author":"Xu Yahui","year":"2023","unstructured":"Yahui Xu, Yi Bin, Jiwei Wei, Yang Yang, Guoqing Wang, and Heng Tao Shen. 2023. Multi-Modal Transformer with Global-Local Alignment for Composed Query Image Retrieval. TMM (2023)."},{"key":"e_1_3_2_2_68_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2018.2855422"},{"key":"e_1_3_2_2_69_1","volume-title":"Xlnet: Generalized autoregressive pretraining for language understanding. In NeurIPS.","author":"Yang Zhilin","year":"2019","unstructured":"Zhilin Yang, Zihang Dai, Yiming Yang, Jaime Carbonell, Russ R Salakhutdinov, and Quoc V Le. 2019. Xlnet: Generalized autoregressive pretraining for language understanding. In NeurIPS."},{"key":"e_1_3_2_2_70_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00166"},{"key":"e_1_3_2_2_71_1","volume-title":"Visualizing and understanding convolutional networks","author":"Zeiler Matthew D","unstructured":"Matthew D Zeiler and Rob Fergus. 2014. Visualizing and understanding convolutional networks. In ECCV. Springer, 818--833."},{"key":"e_1_3_2_2_72_1","doi-asserted-by":"crossref","unstructured":"Kun Zhang Zhendong Mao Quan Wang and Yongdong Zhang. 2022. Negative-aware attention framework for image-text matching. In CVPR. 15661--15670.","DOI":"10.1109\/CVPR52688.2022.01521"},{"key":"e_1_3_2_2_73_1","doi-asserted-by":"crossref","unstructured":"Qi Zhang Zhen Lei Zhaoxiang Zhang and Stan Z Li. 2020. Context-aware attention network for image-text retrieval. In CVPR. 3536--3545.","DOI":"10.1109\/CVPR42600.2020.00359"},{"key":"e_1_3_2_2_74_1","doi-asserted-by":"publisher","DOI":"10.1145\/3383184"},{"key":"e_1_3_2_2_75_1","doi-asserted-by":"crossref","unstructured":"Luowei Zhou Yingbo Zhou Jason J Corso Richard Socher and Caiming Xiong. 2018. End-to-end dense video captioning with masked transformer. In CVPR. 8739--8748","DOI":"10.1109\/CVPR.2018.00911"}],"event":{"name":"MM '23: The 31st ACM International Conference on Multimedia","location":"Ottawa ON Canada","acronym":"MM '23","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 31st ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612427","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3581783.3612427","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T00:04:14Z","timestamp":1755821054000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612427"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,26]]},"references-count":75,"alternative-id":["10.1145\/3581783.3612427","10.1145\/3581783"],"URL":"https:\/\/doi.org\/10.1145\/3581783.3612427","relation":{},"subject":[],"published":{"date-parts":[[2023,10,26]]},"assertion":[{"value":"2023-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}