{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T00:41:23Z","timestamp":1755823283277,"version":"3.44.0"},"publisher-location":"New York, NY, USA","reference-count":42,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,26]],"date-time":"2023-10-26T00:00:00Z","timestamp":1698278400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100002858","name":"China Postdoctoral Science Foundation","doi-asserted-by":"publisher","award":["2022M712369"],"award-info":[{"award-number":["2022M712369"]}],"id":[{"id":"10.13039\/501100002858","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U21B2024?62202327"],"award-info":[{"award-number":["U21B2024?62202327"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,26]]},"DOI":"10.1145\/3581783.3613786","type":"proceedings-article","created":{"date-parts":[[2023,10,27]],"date-time":"2023-10-27T07:27:30Z","timestamp":1698391650000},"page":"5330-5338","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["External Knowledge Dynamic Modeling for Image-text Retrieval"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0809-6177","authenticated-orcid":false,"given":"Song","family":"Yang","sequence":"first","affiliation":[{"name":"Tianjin University &amp; Institute of Artificial Intelligence, Hefei Comprehensive National Science Center, Tianjin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7129-1456","authenticated-orcid":false,"given":"Qiang","family":"Li","sequence":"additional","affiliation":[{"name":"Tianjin University, Tianjin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9609-6120","authenticated-orcid":false,"given":"Wenhui","family":"Li","sequence":"additional","affiliation":[{"name":"Tianjin University, Tianjin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6406-4896","authenticated-orcid":false,"given":"Min","family":"Liu","sequence":"additional","affiliation":[{"name":"Hunan University, Hunan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9765-0998","authenticated-orcid":false,"given":"Xuanya","family":"Li","sequence":"additional","affiliation":[{"name":"Baidu, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5755-9145","authenticated-orcid":false,"given":"Anan","family":"Liu","sequence":"additional","affiliation":[{"name":"Tianjin University &amp; Institute of Artificial Intelligence, Hefei Comprehensive National Science Center, Tianjin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"SPICE: Semantic Propositional Image Caption Evaluation. In Computer Vision - ECCV 2016 - 14th European Conference","author":"Anderson Peter","year":"2016","unstructured":"Peter Anderson, Basura Fernando, Mark Johnson, and Stephen Gould. 2016. SPICE: Semantic Propositional Image Caption Evaluation. In Computer Vision - ECCV 2016 - 14th European Conference, Vol. 9909. 382--398."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01267"},{"key":"e_1_3_2_1_3_1","volume-title":"Two-stream Hierarchical Similarity Reasoning for Image-text Matching. CoRR","author":"Chen Ran","year":"2022","unstructured":"Ran Chen, Hanli Wang, Lei Wang, and Sam Kwong. 2022. Two-stream Hierarchical Similarity Reasoning for Image-text Matching. CoRR, Vol. abs\/2203.05349 (2022)."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2021.3136857"},{"volume-title":"Similarity Reasoning and Filtration for Image-Text Matching","author":"Diao Haiwen","key":"e_1_3_2_1_5_1","unstructured":"Haiwen Diao, Ying Zhang, Lin Ma, and Huchuan Lu. 2021. Similarity Reasoning and Filtration for Image-Text Matching. In Association for the Advancement of Artificial Intelligence. 1218--1226."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1111\/j.1551-6709.2009.01023.x"},{"key":"e_1_3_2_1_7_1","volume-title":"Proc. BMVC. 12","author":"Faghri Fartash","year":"2018","unstructured":"Fartash Faghri, David J. Fleet, Jamie Ryan Kiros, and Sanja Fidler. 2018. VSE: Improving Visual-Semantic Embeddings with Hard Negatives. In Proc. BMVC. 12."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00750"},{"key":"e_1_3_2_1_9_1","volume-title":"Normalized and Geometry-Aware Self-Attention Network for Image Captioning. In 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 10324--10333","author":"Guo Longteng","year":"2020","unstructured":"Longteng Guo, Jing Liu, Xinxin Zhu, Peng Yao, Shichen Lu, and Hanqing Lu. 2020. Normalized and Geometry-Aware Self-Attention Network for Image Captioning. In 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 10324--10333."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00645"},{"key":"e_1_3_2_1_12_1","volume-title":"Step-Wise Hierarchical Alignment Network for Image-Text Matching. In International Joint Conference on Artificial Intelligence. 765--771","author":"Ji Zhong","year":"2021","unstructured":"Zhong Ji, Kexin Chen, and Haoran Wang. 2021. Step-Wise Hierarchical Alignment Network for Image-Text Matching. In International Joint Conference on Artificial Intelligence. 765--771."},{"key":"e_1_3_2_1_13_1","volume-title":"In Defense of Grid Features for Visual Question Answering. In 2020 IEEE\/CVF Conference on Computer Vision and Pattern. 10264--10273","author":"Jiang Huaizu","year":"2020","unstructured":"Huaizu Jiang, Ishan Misra, Marcus Rohrbach, Erik G. Learned-Miller, and Xinlei Chen. 2020. In Defense of Grid Features for Visual Question Answering. In 2020 IEEE\/CVF Conference on Computer Vision and Pattern. 10264--10273."},{"key":"e_1_3_2_1_14_1","volume-title":"Zemel","author":"Kiros Ryan","year":"2014","unstructured":"Ryan Kiros, Ruslan Salakhutdinov, and Richard S. Zemel. 2014. Unifying Visual-Semantic Embeddings with Multimodal Neural Language Models. CoRR, Vol. abs\/1411.2539 (2014)."},{"key":"e_1_3_2_1_15_1","first-page":"212","article-title":"Stacked Cross Attention for Image-Text Matching","volume":"11208","author":"Lee Kuang-Huei","year":"2018","unstructured":"Kuang-Huei Lee, Xi Chen, Gang Hua, Houdong Hu, and Xiaodong He. 2018. Stacked Cross Attention for Image-Text Matching. In Proc. ECCV, Vol. 11208. 212--228.","journal-title":"Proc. ECCV"},{"key":"e_1_3_2_1_16_1","volume-title":"Relation-Aware Graph Attention Network for Visual Question Answering. In 2019 IEEE\/CVF International Conference on Computer Vision. 10312--10321","author":"Li Linjie","year":"2019","unstructured":"Linjie Li, Zhe Gan, Yu Cheng, and Jingjing Liu. 2019. Relation-Aware Graph Attention Network for Visual Question Answering. In 2019 IEEE\/CVF International Conference on Computer Vision. 10312--10321."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01280"},{"key":"e_1_3_2_1_18_1","first-page":"740","article-title":"Microsoft COCO","volume":"8693","author":"Lin Tsung-Yi","year":"2014","unstructured":"Tsung-Yi Lin, Michael Maire, Serge J. Belongie, James Hays, Pietro Perona, Deva Ramanan, Piotr Doll\u00e1r, and C. Lawrence Zitnick. 2014. Microsoft COCO: Common Objects in Context. In Proc. ECCV, Vol. 8693. 740--755.","journal-title":"Common Objects in Context. In Proc. ECCV"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01093"},{"key":"e_1_3_2_1_20_1","volume-title":"Automatic quality assessment for 2D fetal sonographic standard plane based on multi-task learning. CoRR","author":"Luo Hong","year":"2019","unstructured":"Hong Luo, Han Liu, Kejun Li, and Bo Zhang. 2019. Automatic quality assessment for 2D fetal sonographic standard plane based on multi-task learning. CoRR, Vol. abs\/1912.05260 (2019)."},{"key":"e_1_3_2_1_21_1","volume-title":"Geometric Deep Learning on Graphs and Manifolds Using Mixture Model CNNs. In IEEE Conference on Computer Vision and Pattern Recognition, CVPR","author":"Monti Federico","year":"2017","unstructured":"Federico Monti, Davide Boscaini, Jonathan Masci, Emanuele Rodol\u00e0, Jan Svoboda, and Michael M. Bronstein. 2017. Geometric Deep Learning on Graphs and Manifolds Using Mixture Model CNNs. In IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2017. 5425--5434."},{"volume-title":"Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing. 1532--1543","author":"Pennington Jeffrey","key":"e_1_3_2_1_22_1","unstructured":"Jeffrey Pennington, Richard Socher, and Christopher D. Manning. 2014. Glove: Global Vectors for Word Representation. In Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing. 1532--1543."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-016-0965-7"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413961"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/3404835.3462829"},{"key":"e_1_3_2_1_26_1","unstructured":"Shaoqing Ren Kaiming He Ross B. Girshick and Jian Sun. 2015. Faster R-CNN: Towards Real-Time Object Detection with Region Proposal Networks. In Advances in Neural Information Processing Systems 28: Annual Conference on Neural Information Processing Systems. 91--99."},{"key":"e_1_3_2_1_27_1","volume-title":"Knowledge Aware Semantic Concept Expansion for Image-Text Matching. In International Joint Conference on Artificial Intelligence. 5182--5189","author":"Shi Botian","year":"2019","unstructured":"Botian Shi, Lei Ji, Pan Lu, Zhendong Niu, and Nan Duan. 2019. Knowledge Aware Semantic Concept Expansion for Image-Text Matching. In International Joint Conference on Artificial Intelligence. 5182--5189."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2021.3060291"},{"key":"e_1_3_2_1_29_1","volume-title":"Consensus-Aware Visual-Semantic Embedding for Image-Text Matching. In European Conference on Computer Vision","volume":"12369","author":"Wang Haoran","year":"2020","unstructured":"Haoran Wang, Ying Zhang, Zhong Ji, Yanwei Pang, and Lin Ma. 2020b. Consensus-Aware Visual-Semantic Embedding for Image-Text Matching. In European Conference on Computer Vision, Vol. 12369. 18--34."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.541"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV45572.2020.9093614"},{"key":"e_1_3_2_1_32_1","volume-title":"Wendell and Kadry Abdelhamied","author":"Thomas","year":"1992","unstructured":"Thomas C. Wendell and Kadry Abdelhamied. 1992. A phoneme recognition system using modular construction of time-delay neural networks. In Proc. CBMS. 704--709."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2020.2978386"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N19-1301"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11280-018-0541-x"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2022.3182426"},{"volume-title":"Computer Vision - ECCV 2018 - 15th European Conference","author":"Yao Ting","key":"e_1_3_2_1_37_1","unstructured":"Ting Yao, Yingwei Pan, Yehao Li, and Tao Mei. 2018. Exploring Visual Relationship for Image Captioning. In Computer Vision - ECCV 2018 - 15th European Conference, Vol. 11218. 711--727."},{"key":"e_1_3_2_1_38_1","volume-title":"Neural Motifs: Scene Graph Parsing with Global Context. CoRR","author":"Zellers Rowan","year":"2017","unstructured":"Rowan Zellers, Mark Yatskar, Sam Thomson, and Yejin Choi. 2017. Neural Motifs: Scene Graph Parsing with Global Context. CoRR, Vol. abs\/1711.06640 (2017)."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475380"},{"key":"e_1_3_2_1_40_1","volume-title":"Unified Adaptive Relevance Distinguishable Attention Network for Image-Text Matching","author":"Zhang Kun","year":"2022","unstructured":"Kun Zhang, Zhendong Mao, Anan Liu, and Yongdong Zhang. 2022. Unified Adaptive Relevance Distinguishable Attention Network for Image-Text Matching. IEEE Transactions on Multimedia (2022)."},{"volume-title":"Context-Aware Attention Network for Image-Text Retrieval. In 2020 Conference on Computer Vision and Pattern Recognition. 3533--3542","author":"Zhang Qi","key":"e_1_3_2_1_41_1","unstructured":"Qi Zhang, Zhen Lei, Zhaoxiang Zhang, and Stan Z. Li. 2020. Context-Aware Attention Network for Image-Text Retrieval. In 2020 Conference on Computer Vision and Pattern Recognition. 3533--3542."},{"key":"e_1_3_2_1_42_1","volume-title":"Deep Cross-Modal Projection Learning for Image-Text Matching. In Conference on European Conference Computer Vision","volume":"11205","author":"Zhang Ying","year":"2018","unstructured":"Ying Zhang and Huchuan Lu. 2018. Deep Cross-Modal Projection Learning for Image-Text Matching. In Conference on European Conference Computer Vision, Vol. 11205. 707--723."}],"event":{"name":"MM '23: The 31st ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Ottawa ON Canada","acronym":"MM '23"},"container-title":["Proceedings of the 31st ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3613786","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3581783.3613786","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T00:09:14Z","timestamp":1755821354000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3613786"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,26]]},"references-count":42,"alternative-id":["10.1145\/3581783.3613786","10.1145\/3581783"],"URL":"https:\/\/doi.org\/10.1145\/3581783.3613786","relation":{},"subject":[],"published":{"date-parts":[[2023,10,26]]},"assertion":[{"value":"2023-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}