{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,28]],"date-time":"2026-01-28T04:54:10Z","timestamp":1769576050983,"version":"3.49.0"},"reference-count":61,"publisher":"Springer Science and Business Media LLC","issue":"19","license":[{"start":{"date-parts":[[2023,12,16]],"date-time":"2023-12-16T00:00:00Z","timestamp":1702684800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,12,16]],"date-time":"2023-12-16T00:00:00Z","timestamp":1702684800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61728204"],"award-info":[{"award-number":["61728204"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61802182"],"award-info":[{"award-number":["61802182"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"DOI":"10.1007\/s11042-023-17798-1","type":"journal-article","created":{"date-parts":[[2023,12,16]],"date-time":"2023-12-16T07:01:42Z","timestamp":1702710102000},"page":"57895-57912","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Flexible graph-based attention and pooling network for image-text retrieval"],"prefix":"10.1007","volume":"83","author":[{"given":"Hao","family":"Sun","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaolin","family":"Qin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaojing","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,12,16]]},"reference":[{"key":"17798_CR1","doi-asserted-by":"crossref","unstructured":"Liu D, Cui Y, Cao Z, Chen Y (2020) Indoor navigation for mobile agents: a multimodal vision fusion model, pp 1\u20138","DOI":"10.1109\/IJCNN48605.2020.9207265"},{"key":"17798_CR2","doi-asserted-by":"crossref","unstructured":"Yan L, Liu D, Song Y, Yu C (2020) Multimodal aggregation approach for memory vision-voice indoor navigation with meta-learning 5847\u20135854","DOI":"10.1109\/IROS45743.2020.9341398"},{"key":"17798_CR3","doi-asserted-by":"crossref","unstructured":"Yan L et\u00a0al (2022) Gl-rg: global-local representation granularity for video captioning","DOI":"10.24963\/ijcai.2022\/384"},{"key":"17798_CR4","first-page":"1","volume":"2022","author":"Q Wang","year":"2022","unstructured":"Wang Q et al (2022) Webformer: the web-page transformer for structure information extraction. Proc ACM Web Conf 2022:1\u20132","journal-title":"Proc ACM Web Conf"},{"key":"17798_CR5","doi-asserted-by":"crossref","unstructured":"Yang L et al (2023) Findings of the association for computational linguistics. In: Rogers A, Boyd-Graber J, Okazaki N (eds) Mixpave: mix-prompt tuning for few-shot product attribute value extraction: ACL","DOI":"10.18653\/v1\/2023.findings-acl.633"},{"key":"17798_CR6","first-page":"2405","volume-title":"Mustie: multimodal structural transformer for web information extraction","author":"Q Wang","year":"2023","unstructured":"Wang Q et al (2023) Proceedings of the 61st annual meeting of the association for computational linguistics. In: Rogers A, Boyd-Graber J, Okazaki N (eds) Mustie: multimodal structural transformer for web information extraction, vol 1. Association for Computational Linguistics, Toronto, pp 2405\u20132420"},{"key":"17798_CR7","first-page":"2310","volume":"1","author":"Y Huang","year":"2017","unstructured":"Huang Y, Wang W, Wang L (2017) Instance-aware image and sentence matching with selective multimodal lstm 1:2310\u20132318","journal-title":"Instance-aware image and sentence matching with selective multimodal lstm"},{"key":"17798_CR8","doi-asserted-by":"crossref","unstructured":"Nam H, Ha J-W, Kim J (2017) Dual attention networks for multimodal reasoning and matching, pp 299\u2013307","DOI":"10.1109\/CVPR.2017.232"},{"key":"17798_CR9","doi-asserted-by":"crossref","unstructured":"Lee K-H, Chen X, Hua G, Hu H, He X (2018) Stacked cross attention for image-text matching, pp 201\u2013216","DOI":"10.1007\/978-3-030-01225-0_13"},{"issue":"1","key":"17798_CR10","doi-asserted-by":"publisher","first-page":"388","DOI":"10.1109\/TCSVT.2021.3060713","volume":"32","author":"J Wu","year":"2022","unstructured":"Wu J, Wu C, Lu J, Wang L, Cui X (2022) Region reinforcement network with topic constraint for image-text matching. IEEE Trans Circuits Syst Video Technol 32(1):388\u2013397. https:\/\/doi.org\/10.1109\/TCSVT.2021.3060713","journal-title":"IEEE Trans Circuits Syst Video Technol"},{"key":"17798_CR11","doi-asserted-by":"publisher","unstructured":"Zhang K, Mao Z, Wang Q, Zhang Y (2022) Negative-aware attention framework for image-text matching. IEEE, pp 15640\u201315649. https:\/\/doi.org\/10.1109\/CVPR52688.2022.01521","DOI":"10.1109\/CVPR52688.2022.01521"},{"key":"17798_CR12","doi-asserted-by":"crossref","unstructured":"Wang S, Chen Y, Zhuo J, Huang Q, Tian Q (2018) Joint global and co-attentive representation learning for image-sentence retrieval, pp 1398\u20131406","DOI":"10.1145\/3240508.3240535"},{"key":"17798_CR13","doi-asserted-by":"crossref","unstructured":"Zhang Q, Lei Z, Zhang Z, Li SZ (2020) Context-aware attention network for image-text retrieval, pp 3536\u20133545","DOI":"10.1109\/CVPR42600.2020.00359"},{"key":"17798_CR14","doi-asserted-by":"crossref","unstructured":"Yu T et al (2021) Heterogeneous attention network for effective and efficient cross-modal retrieval, pp 1146\u20131156","DOI":"10.1145\/3404835.3462924"},{"key":"17798_CR15","doi-asserted-by":"crossref","unstructured":"Wei X, Zhang T, Li Y, Zhang Y, Wu F (2020) Multi-modality cross attention network for image and sentence matching, pp 10941\u201310950","DOI":"10.1109\/CVPR42600.2020.01095"},{"key":"17798_CR16","doi-asserted-by":"crossref","unstructured":"Qu L, Liu M, Wu J, Gao Z, Nie L (2021) Dynamic modality interaction modeling for image-text retrieval, pp 1104\u20131113","DOI":"10.1145\/3404835.3462829"},{"key":"17798_CR17","doi-asserted-by":"crossref","unstructured":"Li J, Niu L, Zhang L (2022) Action-aware embedding enhancement for image-text retrieval. AAAI Press 1:1323\u20131331","DOI":"10.1609\/aaai.v36i2.20020"},{"key":"17798_CR18","doi-asserted-by":"crossref","unstructured":"Qu L, Liu M, Cao D, Nie L, Tian Q (2020) Context-aware multi-view summarization network for image-text matching, pp 1047\u20131055","DOI":"10.1145\/3394171.3413961"},{"key":"17798_CR19","doi-asserted-by":"publisher","first-page":"374","DOI":"10.1109\/LSP.2021.3135825","volume":"29","author":"H Lan","year":"2022","unstructured":"Lan H, Zhang P (2022) Learning and integrating multi-level matching features for image-text retrieval. IEEE Signal Process Lett 29:374\u2013378. https:\/\/doi.org\/10.1109\/LSP.2021.3135825","journal-title":"IEEE Signal Process Lett"},{"key":"17798_CR20","unstructured":"Cheng Z et\u00a0al (2023) Fusion is not enough: single-modal attacks to compromise fusion models in autonomous driving. ArXiv:abs\/2304.14614, https:\/\/api.semanticscholar.org\/CorpusID:258417952"},{"key":"17798_CR21","doi-asserted-by":"crossref","unstructured":"Li K, Zhang Y, Li K, Li Y, Fu Y (2019) Visual semantic reasoning for image-text matching, pp 4654\u20134662","DOI":"10.1109\/ICCV.2019.00475"},{"key":"17798_CR22","doi-asserted-by":"crossref","unstructured":"Liu C et al (2020) Graph structured network for image-text matching, pp 10921\u201310930","DOI":"10.1109\/CVPR42600.2020.01093"},{"key":"17798_CR23","doi-asserted-by":"crossref","unstructured":"Zhong, X et\u00a0al (2021) Auxiliary bi-level graph representation for cross-modal image-text retrieval. IEEE, pp 1\u20136","DOI":"10.1109\/ICME51207.2021.9428380"},{"key":"17798_CR24","doi-asserted-by":"crossref","unstructured":"Wang S, Wang R, Yao Z, Shan S, Chen X (2020) Cross-modal scene graph matching for relationship-aware image-text retrieval, pp 1508\u20131517","DOI":"10.1109\/WACV45572.2020.9093614"},{"key":"17798_CR25","doi-asserted-by":"publisher","unstructured":"Long S, Han SC, Wan X, Poon J (2022) Gradual: graph-based dual-modal representation for image-text matching. IEEE, pp 2463\u20132472. https:\/\/doi.org\/10.1109\/WACV51458.2022.00252","DOI":"10.1109\/WACV51458.2022.00252"},{"issue":"8","key":"17798_CR26","doi-asserted-by":"publisher","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter S, Schmidhuber J (1997) Long short-term memory. Neural Comput 9(8):1735\u20131780","journal-title":"Neural Comput"},{"issue":"4","key":"17798_CR27","doi-asserted-by":"publisher","first-page":"2008","DOI":"10.1109\/TIP.2018.2882225","volume":"28","author":"F Huang","year":"2019","unstructured":"Huang F, Zhang X, Zhao Z, Li Z (2019) Bi-directional spatial-semantic attention networks for image-text matching. IEEE Trans Image Process 28(4):2008\u20132020. https:\/\/doi.org\/10.1109\/TIP.2018.2882225","journal-title":"IEEE Trans Image Process"},{"key":"17798_CR28","doi-asserted-by":"crossref","unstructured":"Ji Z, Wang H, Han J, Pang Y (2019) Saliency-guided attention network for image-sentence matching, pp 5754\u20135763","DOI":"10.1109\/ICCV.2019.00585"},{"key":"17798_CR29","first-page":"3","volume-title":"Focus your attention: a bidirectional focal attention network for image-text matching, MM \u201919","author":"C Liu","year":"2019","unstructured":"Liu C et al (2019) Focus your attention: a bidirectional focal attention network for image-text matching, MM \u201919. Association for Computing Machinery, New York, pp 3\u201311"},{"key":"17798_CR30","doi-asserted-by":"crossref","unstructured":"Wang Y et\u00a0al (2019) Position focused attention network for image-text matching. arXiv:1907.09748","DOI":"10.24963\/ijcai.2019\/526"},{"key":"17798_CR31","unstructured":"Vaswani A et al (2017) Attention is all you need 5998\u20136008. https:\/\/proceedings.neurips.cc\/paper\/2017\/hash\/3f5ee243547dee91fbd053c1c4a845aa-Abstract.html"},{"key":"17798_CR32","doi-asserted-by":"crossref","unstructured":"Wu Y, Wang S, Song G, Huang Q (2019) Learning fragment self-attention embeddings for image-text matching, pp 2088\u20132096","DOI":"10.1145\/3343031.3350940"},{"key":"17798_CR33","unstructured":"Kipf TN, Welling M (2017) Semi-supervised classification with graph convolutional networks, OpenReview.net.https:\/\/openreview.net\/forum?id=SJU4ayYgl"},{"key":"17798_CR34","doi-asserted-by":"crossref","unstructured":"Cho K, Van\u00a0Merri\u00ebnboer B, Bahdanau D, Bengio Y (2014) On the properties of neural machine translation: encoder-decoder approaches. arXiv:1409.1259","DOI":"10.3115\/v1\/W14-4012"},{"key":"17798_CR35","doi-asserted-by":"crossref","unstructured":"Zellers R, Yatskar M, Thomson S, Choi Y (2018) Neural motifs: scene graph parsing with global context, pp 5831\u20135840","DOI":"10.1109\/CVPR.2018.00611"},{"key":"17798_CR36","doi-asserted-by":"crossref","unstructured":"Li Y, Ouyang W, Zhou B, Wang K, Wang, X (2017) Scene graph generation from objects, phrases and region captions, pp 1261\u20131270","DOI":"10.1109\/ICCV.2017.142"},{"key":"17798_CR37","unstructured":"Krishna R et\u00a0al (2016) Visual genome: connecting language and vision using crowdsourced dense image annotations. arXiv:1602.07332"},{"key":"17798_CR38","doi-asserted-by":"crossref","unstructured":"Xu D, Zhu Y, Choy CB, Fei-Fei L (2017) Scene graph generation by iterative message passing, pp 3097\u20133106","DOI":"10.1109\/CVPR.2017.330"},{"key":"17798_CR39","doi-asserted-by":"publisher","unstructured":"Liang Y et\u00a0al (2019) Vrr-vg: refocusing visually-relevant relationships. IEEE, pp 10402\u201310411. https:\/\/doi.org\/10.1109\/ICCV.2019.01050","DOI":"10.1109\/ICCV.2019.01050"},{"key":"17798_CR40","doi-asserted-by":"publisher","unstructured":"Lu X, Zhu L, Liu L, Nie L, Zhang H (2021) Graph convolutional multi-modal hashing for flexible multimedia retrieval, pp 1414\u20131422. https:\/\/doi.org\/10.1145\/3474085.3475598","DOI":"10.1145\/3474085.3475598"},{"key":"17798_CR41","doi-asserted-by":"crossref","unstructured":"Ge X et al (2021) Structured multi-modal feature embedding and alignment for image-sentence retrieval. ACM, pp 5185\u20135193","DOI":"10.1145\/3474085.3475634"},{"key":"17798_CR42","doi-asserted-by":"crossref","unstructured":"Wang H, Zhang Y, Ji Z, Pang Y, Ma L (2020) Consensus-aware visual-semantic embedding for image-text matching. Lecture notes in computer science, vol 12369. Springer, pp 18\u201334","DOI":"10.1007\/978-3-030-58586-0_2"},{"key":"17798_CR43","doi-asserted-by":"crossref","unstructured":"Yan L, Cui Y, Chen Y, Liu D (2021) Hierarchical attention fusion for geo-localization, pp 2220\u20132224","DOI":"10.1109\/ICASSP39728.2021.9414517"},{"key":"17798_CR44","doi-asserted-by":"crossref","unstructured":"Cui Y, Yan L, Cao Z, Liu D (2021) Tf-blender: temporal feature blender for video object detection, pp 8138\u20138147","DOI":"10.1109\/ICCV48922.2021.00803"},{"issue":"7","key":"17798_CR45","doi-asserted-by":"publisher","first-page":"6101","DOI":"10.1609\/aaai.v35i7.16760","volume":"35","author":"D Liu","year":"2021","unstructured":"Liu D et al (2021) Densernet: weakly supervised visual localization using multi-scale feature aggregation. Proc AAAI Conf Artif Intell 35(7):6101\u20136109. https:\/\/doi.org\/10.1609\/aaai.v35i7.16760","journal-title":"Proc AAAI Conf Artif Intell"},{"key":"17798_CR46","doi-asserted-by":"publisher","first-page":"300","DOI":"10.1016\/j.neucom.2020.12.067","volume":"432","author":"Y Cui","year":"2021","unstructured":"Cui Y et al (2021) Geometric attentional dynamic graph convolutional neural networks for point cloud analysis. Neurocomputing 432:300\u2013310","journal-title":"Neurocomputing"},{"key":"17798_CR47","first-page":"91","volume":"28","author":"S Ren","year":"2015","unstructured":"Ren S, He K, Girshick R, Sun J (2015) Faster r-cnn: towards real-time object detection with region proposal networks. Adv Neural Inf Process 28:91\u201399","journal-title":"Adv Neural Inf Process"},{"key":"17798_CR48","doi-asserted-by":"crossref","unstructured":"Tang K, Niu Y, Huang J, Shi J, Zhang H (2020) Unbiased scene graph generation from biased training. Computer Vision Foundation\/IEEE 1:3713\u20133722","DOI":"10.1109\/CVPR42600.2020.00377"},{"key":"17798_CR49","unstructured":"Devlin J, Chang MW, Lee K, Toutanova K (2018) Bert: pre-training of deep bidirectional transformers for language understanding"},{"key":"17798_CR50","first-page":"382","volume-title":"Spice: semantic propositional image caption evaluation","author":"P Anderson","year":"2016","unstructured":"Anderson P, Fernando B, Johnson M, Gould S (2016) Computer vision \u2013 ECCV 2016. In: Leibe B, Matas J, Sebe N, Welling M (eds) Spice: semantic propositional image caption evaluation. Springer International Publishing, Cham, pp 382\u2013398"},{"key":"17798_CR51","doi-asserted-by":"publisher","unstructured":"Manning CD et al (2014) The stanford corenlp natural language processing toolkit. Assoc Comput Linguistics 1:55\u201360. https:\/\/doi.org\/10.3115\/v1\/p14-5010","DOI":"10.3115\/v1\/p14-5010"},{"key":"17798_CR52","unstructured":"Hendrycks D, Gimpel K (2016) Gaussian error linear units (gelus)"},{"key":"17798_CR53","unstructured":"Ba JL, Kiros JR, Hinton GE (2016) Layer normalization. arXiv:1607.06450"},{"key":"17798_CR54","unstructured":"Faghri F, Fleet DJ, Kiros JR, Fidler S (2017) Vse++: improving visual-semantic embeddings with hard negatives. arXiv:1707.05612"},{"key":"17798_CR55","unstructured":"Belghazi MI et al (2018) Mutual information neural estimation. (eds Dy JG, Krause A) Proceedings of the 35th international conference on machine learning, ICML 2018, vol\u00a080. Proceedings of Machine Learning Research, Stockholmsm\u00e4ssan, pp 530\u2013539. http:\/\/proceedings.mlr.press\/v80\/belghazi18a.html"},{"key":"17798_CR56","doi-asserted-by":"crossref","unstructured":"Plummer BA et\u00a0al (2015) Flickr30k entities: collecting region-to-phrase correspondences for richer image-to-sentence models, pp 2641\u20132649","DOI":"10.1109\/ICCV.2015.303"},{"key":"17798_CR57","doi-asserted-by":"crossref","unstructured":"Lin T-Y et al (2014) Microsoft coco: common objects in context. Springer, 740\u2013755","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"17798_CR58","doi-asserted-by":"crossref","unstructured":"Karpathy A, Fei-Fei L (2015) Deep visual-semantic alignments for generating image descriptions, pp 3128\u20133137","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"17798_CR59","unstructured":"Collobert R, Kavukcuoglu K, Farabet C (2011) Torch7: a matlab-like environment for machine learning"},{"key":"17798_CR60","unstructured":"Loshchilov I, Hutter F (2017) Decoupled weight decay regularization. arXiv:1711.05101"},{"key":"17798_CR61","first-page":"2579","volume":"9","author":"L van der Maaten","year":"2008","unstructured":"van der Maaten L, Hinton GE (2008) Visualizing data using t-sne. J Mach Learn Res 9:2579\u20132605","journal-title":"J Mach Learn Res"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-17798-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-023-17798-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-17798-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,5,25]],"date-time":"2024-05-25T06:32:22Z","timestamp":1716618742000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-023-17798-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,12,16]]},"references-count":61,"journal-issue":{"issue":"19","published-online":{"date-parts":[[2024,6]]}},"alternative-id":["17798"],"URL":"https:\/\/doi.org\/10.1007\/s11042-023-17798-1","relation":{},"ISSN":["1573-7721"],"issn-type":[{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,12,16]]},"assertion":[{"value":"24 September 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 November 2023","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"30 November 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 December 2023","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflicts of interest"}}]}}