{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,25]],"date-time":"2026-01-25T01:35:56Z","timestamp":1769304956593,"version":"3.49.0"},"reference-count":29,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2024,11,6]],"date-time":"2024-11-06T00:00:00Z","timestamp":1730851200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,6]],"date-time":"2024-11-06T00:00:00Z","timestamp":1730851200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"Science and Technology Research Program of Higher Education Institutions in Hebei Province","award":["ZD2022082"],"award-info":[{"award-number":["ZD2022082"]}]},{"name":"Higher Education Teaching Reform Research and Practice Project of Hebei Province","award":["2022GJJG049"],"award-info":[{"award-number":["2022GJJG049"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"published-print":{"date-parts":[[2025,1]]},"DOI":"10.1007\/s11227-024-06652-2","type":"journal-article","created":{"date-parts":[[2024,11,6]],"date-time":"2024-11-06T18:02:24Z","timestamp":1730916144000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Scene graph fusion and negative sample generation strategy for image-text matching"],"prefix":"10.1007","volume":"81","author":[{"given":"Liqin","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Pengcheng","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xu","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhihong","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yongfeng","family":"Dong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,11,6]]},"reference":[{"key":"6652_CR1","doi-asserted-by":"publisher","first-page":"471","DOI":"10.1109\/TIP.2022.3229631","volume":"32","author":"M Tian","year":"2023","unstructured":"Tian M, Xinxiao W, Jia Y (2023) Adaptive latent graph representation learning for image-text matching. IEEE Trans Image Process 32:471\u2013482. https:\/\/doi.org\/10.1109\/TIP.2022.3229631","journal-title":"IEEE Trans Image Process"},{"key":"6652_CR2","doi-asserted-by":"publisher","first-page":"9067","DOI":"10.1109\/ACCESS.2023.3237966","volume":"11","author":"W Lu","year":"2023","unstructured":"Lu W, Chenyu W, Guo H, Zhao Z (2023) A cross-modal alignment for zero-shot image classification. IEEE Access 11:9067\u20139073. https:\/\/doi.org\/10.1109\/ACCESS.2023.3237966","journal-title":"IEEE Access"},{"key":"6652_CR3","doi-asserted-by":"publisher","first-page":"108005","DOI":"10.1016\/J.ENGAPPAI.2024.108005","volume":"133","author":"T Yao","year":"2024","unstructured":"Yao T, Peng S, Sun Y, Sheng G, Haiyan F, Kong X (2024) Cross-modal semantic interference suppression for image-text matching. Eng Appl Artif Intell 133:108005. https:\/\/doi.org\/10.1016\/J.ENGAPPAI.2024.108005","journal-title":"Eng Appl Artif Intell"},{"key":"6652_CR4","doi-asserted-by":"publisher","first-page":"126389","DOI":"10.1016\/J.NEUCOM.2023.126389","volume":"548","author":"Q Zhao","year":"2023","unstructured":"Zhao Q, Wan Y, Xu J, Fang L (2023) Cross-modal attention fusion network for RGB-D semantic segmentation. Neurocomputing 548:126389. https:\/\/doi.org\/10.1016\/J.NEUCOM.2023.126389","journal-title":"Neurocomputing"},{"key":"6652_CR5","doi-asserted-by":"publisher","first-page":"84","DOI":"10.1016\/J.INS.2022.11.087","volume":"620","author":"Z Li","year":"2023","unstructured":"Li Z, Lu H, Fu H, Gu G (2023) Parallel learned generative adversarial network with multi-path subspaces for cross-modal retrieval. Inf Sci 620:84\u2013104. https:\/\/doi.org\/10.1016\/J.INS.2022.11.087","journal-title":"Inf Sci"},{"key":"6652_CR6","doi-asserted-by":"publisher","unstructured":"Li H, Bin Y, Liao J, Yang Y, Shen HT (2023b) Your negative may not be true negative: Boosting image-text matching with false negative elimination. In Abdulmotaleb El-Saddik, Tao Mei, Rita Cucchiara, Marco Bertini, Diana Patricia\u00a0Tobon Vallejo, Pradeep\u00a0K. Atrey, and M.\u00a0Shamim Hossain, editors, Proceedings of the 31st ACM International Conference on Multimedia, MM 2023, Ottawa, ON, Canada, 29 October 2023- 3 November 2023, pages 924\u2013934. ACM. https:\/\/doi.org\/10.1145\/3581783.3612101","DOI":"10.1145\/3581783.3612101"},{"key":"6652_CR7","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3499027","volume":"18","author":"Y Cheng","year":"2022","unstructured":"Cheng Y, Zhu X, Qian J, Wen F, Liu P (2022) Cross-modal graph matching network for image-text retrieval. ACM Trans Multim Comput Commun Appl 18:1\u201323. https:\/\/doi.org\/10.1145\/3499027","journal-title":"ACM Trans Multim Comput Commun Appl"},{"key":"6652_CR8","unstructured":"Faghri F, Fleet DJ, Kiros JR, Fidler S (2018) VSE++: improving visual-semantic embeddings with hard negatives. In British Machine Vision Conference 2018, BMVC 2018, Newcastle, UK, September 3-6, 2018, page\u00a012. BMVA Press. http:\/\/bmvc2018.org\/contents\/papers\/0344.pdf"},{"issue":"21","key":"6652_CR9","doi-asserted-by":"publisher","first-page":"26126","DOI":"10.1007\/S10489-023-04722-1","volume":"53","author":"X Chang","year":"2023","unstructured":"Chang X, Wang T, Cai S, Sun C (2023) LANDMARK: language-guided representation enhancement framework for scene graph generation. Appl Intell 53(21):26126\u201326138. https:\/\/doi.org\/10.1007\/S10489-023-04722-1","journal-title":"Appl Intell"},{"key":"6652_CR10","doi-asserted-by":"publisher","unstructured":"Wang S, Wang R, Yao Z, Shan S, Chen X (2020) Cross-modal scene graph matching for relationship-aware image-text retrieval. In IEEE Winter Conference on Applications of Computer Vision, WACV 2020, Snowmass Village, CO, USA, March 1-5, 2020, pages 1497\u20131506. IEEE. https:\/\/doi.org\/10.1109\/WACV45572.2020.9093614","DOI":"10.1109\/WACV45572.2020.9093614"},{"key":"6652_CR11","doi-asserted-by":"publisher","unstructured":"Jin D, Wang L, Zheng Y, Li X, Jiang F, Lin W, Pan S (July 2022) CGMN: a contrastive graph matching network for self-supervised graph similarity learning. In Luc\u00a0De Raedt, editor, Proceedings of the Thirty-First International Joint Conference on Artificial Intelligence, IJCAI 2022, Vienna, Austria, 23-29, pages 2101\u20132107. ijcai.org, 2022. https:\/\/doi.org\/10.24963\/IJCAI.2022\/292","DOI":"10.24963\/IJCAI.2022\/292"},{"key":"6652_CR12","doi-asserted-by":"publisher","first-page":"127052","DOI":"10.1016\/J.NEUCOM.2023.127052","volume":"566","author":"H Li","year":"2024","unstructured":"Li H, Zhu G, Zhang L, Jiang Y, Dang Y, Hou H, Shen P, Zhao X, Shah SAA, Bennamoun M (2024) Scene graph generation: A comprehensive survey. Neurocomputing 566:127052. https:\/\/doi.org\/10.1016\/J.NEUCOM.2023.127052","journal-title":"Neurocomputing"},{"key":"6652_CR13","doi-asserted-by":"publisher","unstructured":"Liu C, Mao Z, Zhang T, Xie H, Wang B, Zhang Y (2020) Graph structured network for image-text matching. In 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2020, Seattle, WA, USA, June 13-19, pages 10918\u201310927. Computer Vision Foundation \/ IEEE, 2020. https:\/\/doi.org\/10.1109\/CVPR42600.2020.01093. https:\/\/openaccess.thecvf.com\/content_CVPR_2020\/html\/Liu_Graph_Structured_Network_for_Image-Text_Matching_CVPR_2020_paper.html","DOI":"10.1109\/CVPR42600.2020.01093"},{"key":"6652_CR14","doi-asserted-by":"publisher","unstructured":"Nguyen MD, Nguyen BT, Gurrin C (September 2021) A deep local and global scene-graph matching for image-text retrieval. In Hamido Fujita and H\u00e9ctor P\u00e9rez-Meana, editors, New Trends in Intelligent Software Methodologies, Tools and Techniques - Proceedings of the 20th International Conference on New Trends in Intelligent Software Methodologies, Tools and Techniques, SoMeT 202, Cancun, Mexico, 21-23 , volume 337 of Frontiers in Artificial Intelligence and Applications, pages 510\u2013523. IOS Press, 2021. https:\/\/doi.org\/10.3233\/FAIA210049","DOI":"10.3233\/FAIA210049"},{"key":"6652_CR15","doi-asserted-by":"publisher","unstructured":"Diao H, Zhang Y, Ma L, Lu H (2021) Similarity reasoning and filtration for image-text matching. In Thirty-Fifth AAAI Conference on Artificial Intelligence, AAAI 2021, Thirty-Third Conference on Innovative Applications of Artificial Intelligence, IAAI 2021, The Eleventh Symposium on Educational Advances in Artificial Intelligence, EAAI 2021, Virtual Event, February 2-9, pages 1218\u20131226. AAAI Press, 2021. https:\/\/doi.org\/10.1609\/AAAI.V35I2.16209","DOI":"10.1609\/AAAI.V35I2.16209"},{"key":"6652_CR16","doi-asserted-by":"publisher","first-page":"593","DOI":"10.1016\/J.NEUCOM.2022.11.003","volume":"518","author":"X Yang","year":"2023","unstructured":"Yang X, Li C, Zheng D, Wen P, Yin G (2023) RFE-SRN: image-text similarity reasoning network based on regional feature enhancement. Neurocomputing 518:593\u2013601. https:\/\/doi.org\/10.1016\/J.NEUCOM.2022.11.003","journal-title":"Neurocomputing"},{"issue":"6","key":"6652_CR17","doi-asserted-by":"publisher","first-page":"1137","DOI":"10.1109\/TPAMI.2016.2577031","volume":"39","author":"S Ren","year":"2017","unstructured":"Ren S, He K, Girshick RB, Sun J (2017) Faster R-CNN: towards real-time object detection with region proposal networks. IEEE Trans Pattern Anal Mach Intell 39(6):1137\u20131149. https:\/\/doi.org\/10.1109\/TPAMI.2016.2577031","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"6652_CR18","doi-asserted-by":"publisher","unstructured":"Li Y, Ouyang W, Zhou B, Wang K, Wang X (2017) Scene graph generation from objects, phrases and region captions. In IEEE International Conference on Computer Vision, ICCV 2017, Venice, Italy, October 22-29, 2017, pages 1270\u20131279. IEEE Computer Society. https:\/\/doi.org\/10.1109\/ICCV.2017.142","DOI":"10.1109\/ICCV.2017.142"},{"key":"6652_CR19","doi-asserted-by":"publisher","unstructured":"Anderson P, Fernando B, Johnson M, Gould S (2016) SPICE: semantic propositional image caption evaluation. In Bastian Leibe, Jiri Matas, Nicu Sebe, and Max Welling, editors, Computer Vision - ECCV 2016 - 14th European Conference, Amsterdam, The Netherlands, October 11-14, Proceedings, Part V, volume 9909 of Lecture Notes in Computer Science, pages 382\u2013398. Springer, 2016. https:\/\/doi.org\/10.1007\/978-3-319-46454-1_24","DOI":"10.1007\/978-3-319-46454-1_24"},{"key":"6652_CR20","doi-asserted-by":"publisher","unstructured":"Pennington J, Socher R, Manning CD (2014) Glove: global vectors for word representation. In Alessandro Moschitti, Bo\u00a0Pang, and Walter Daelemans, editors, Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing, EMNLP 2014, October 25-29, Doha, Qatar, A meeting of SIGDAT, a Special Interest Group of the ACL, pages 1532\u20131543. ACL, 2014. https:\/\/doi.org\/10.3115\/V1\/D14-1162","DOI":"10.3115\/V1\/D14-1162"},{"issue":"9","key":"6652_CR21","doi-asserted-by":"publisher","first-page":"11807","DOI":"10.1007\/S11063-023-11388-W","volume":"55","author":"X Zhang","year":"2023","unstructured":"Zhang X, Peng Y, Wang W, Liu S (2023) Image super-resolution based on gated residual and gated convolution networks. Neural Process Lett 55(9):11807\u201311821. https:\/\/doi.org\/10.1007\/S11063-023-11388-W","journal-title":"Neural Process Lett"},{"issue":"1","key":"6652_CR22","doi-asserted-by":"publisher","first-page":"74","DOI":"10.1007\/S11263-016-0965-7","volume":"123","author":"BA Plummer","year":"2017","unstructured":"Plummer BA, Wang L, Cervantes CM, Caicedo JC, Hockenmaier J, Lazebnik S (2017) Flickr30k entities: collecting region-to-phrase correspondences for richer image-to-sentence models. Int J Comput Vis 123(1):74\u201393. https:\/\/doi.org\/10.1007\/S11263-016-0965-7","journal-title":"Int J Comput Vis"},{"key":"6652_CR23","doi-asserted-by":"publisher","unstructured":"Lin TY, Maire M, Belongie SJ, Hays J, Perona P, Ramanan D, Doll\u00e1r P, Zitnick CL (2014) Microsoft COCO: common objects in context. In David\u00a0J. Fleet, Tom\u00e1s Pajdla, Bernt Schiele, and Tinne Tuytelaars, editors, Computer Vision - ECCV 2014 - 13th European Conference, Zurich, Switzerland, September 6-12, Proceedings, Part V, volume 8693 of Lecture Notes in Computer Science, pages 740\u2013755. Springer, 2014. https:\/\/doi.org\/10.1007\/978-3-319-10602-1_48","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"6652_CR24","doi-asserted-by":"publisher","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In 2016 IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2016, Las Vegas, NV, USA, June 27-30, 2016, pages 770\u2013778. IEEE Computer Society. https:\/\/doi.org\/10.1109\/CVPR.2016.90","DOI":"10.1109\/CVPR.2016.90"},{"issue":"1","key":"6652_CR25","doi-asserted-by":"publisher","first-page":"32","DOI":"10.1007\/S11263-016-0981-7","volume":"123","author":"R Krishna","year":"2017","unstructured":"Krishna R, Zhu Y, Groth O, Johnson J, Hata K, Kravitz J, Chen S, Kalantidis Y, Li LJ, Shamma DA, Bernstein MS, Fei-Fei L (2017) Visual genome: connecting language and vision using crowdsourced dense image annotations. Int J Comput Vis 123(1):32\u201373. https:\/\/doi.org\/10.1007\/S11263-016-0981-7","journal-title":"Int J Comput Vis"},{"key":"6652_CR26","doi-asserted-by":"publisher","unstructured":"Wang Z, Liu X, Li H, Sheng L, Yan J, Wang X, Shao J (2019) CAMP: cross-modal adaptive message passing for text-image retrieval. In 2019 IEEE\/CVF International Conference on Computer Vision, ICCV 2019, Seoul, Korea (South), October 27 - November 2, 2019, pages 5763\u20135772. IEEE. https:\/\/doi.org\/10.1109\/ICCV.2019.00586. URL https:\/\/doi.org\/10.1109\/ICCV.2019.00586","DOI":"10.1109\/ICCV.2019.00586"},{"key":"6652_CR27","doi-asserted-by":"publisher","unstructured":"Zhang Q, Lei Z, Zhang Z, Li SZ (2020) Context-aware attention network for image-text retrieval. In 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2020, Seattle, WA, USA, June 13-19, pages 3533\u20133542. Computer Vision Foundation \/ IEEE, 2020. https:\/\/doi.org\/10.1109\/CVPR42600.2020.00359. https:\/\/openaccess.thecvf.com\/content_CVPR_2020\/html\/Zhang_Context-Aware_Attention_Network_for_Image-Text_Retrieval_CVPR_2020_paper.html","DOI":"10.1109\/CVPR42600.2020.00359"},{"issue":"3","key":"6652_CR28","doi-asserted-by":"publisher","first-page":"41","DOI":"10.1109\/MIS.2023.3265176","volume":"38","author":"H Shang","year":"2023","unstructured":"Shang H, Zhao G, Shi J, Qian X (2023) A multiview text imagination network based on latent alignment for image-text matching. IEEE Intell Syst 38(3):41\u201350. https:\/\/doi.org\/10.1109\/MIS.2023.3265176","journal-title":"IEEE Intell Syst"},{"key":"6652_CR29","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3563390","volume":"22","author":"J Pie","year":"2023","unstructured":"Pie J, Zhong K, Wang L, Lakshmanna K (2023) Scene graph semantic inference for image and text matching. ACM Trans Asian Low Res Lang Inf Process 22:1\u201323. https:\/\/doi.org\/10.1145\/3563390","journal-title":"ACM Trans Asian Low Res Lang Inf Process"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-024-06652-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-024-06652-2\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-024-06652-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,6]],"date-time":"2024-11-06T18:04:27Z","timestamp":1730916267000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-024-06652-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,6]]},"references-count":29,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2025,1]]}},"alternative-id":["6652"],"URL":"https:\/\/doi.org\/10.1007\/s11227-024-06652-2","relation":{},"ISSN":["0920-8542","1573-0484"],"issn-type":[{"value":"0920-8542","type":"print"},{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,6]]},"assertion":[{"value":"23 October 2024","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 November 2024","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"138"}}