{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T13:16:29Z","timestamp":1783170989290,"version":"3.54.6"},"reference-count":26,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100017691","name":"Guangxi Key Research and Development Program","doi-asserted-by":"publisher","award":["AB25069258"],"award-info":[{"award-number":["AB25069258"]}],"id":[{"id":"10.13039\/501100017691","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62162003"],"award-info":[{"award-number":["62162003"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100011785","name":"Guangxi Science and Technology Department","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100011785","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition Letters"],"published-print":{"date-parts":[[2026,8]]},"DOI":"10.1016\/j.patrec.2026.05.009","type":"journal-article","created":{"date-parts":[[2026,5,15]],"date-time":"2026-05-15T15:28:40Z","timestamp":1778858920000},"page":"49-55","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["HSAMR: A hierarchical semantic alignment framework for multimodal retrieval"],"prefix":"10.1016","volume":"206","author":[{"given":"Zhichao","family":"Sun","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ningjiang","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hongda","family":"Qin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Long","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.patrec.2026.05.009_bib0001","doi-asserted-by":"crossref","first-page":"3520","DOI":"10.1109\/TMM.2021.3101642","article-title":"Adaptive label-aware graph convolutional networks for cross-modal retrieval","volume":"24","author":"Qian","year":"2022","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.patrec.2026.05.009_bib0002","doi-asserted-by":"crossref","first-page":"1775","DOI":"10.1109\/TMM.2021.3072479","article-title":"Dual attention on pyramid feature maps for image captioning","volume":"24","author":"Yu","year":"2022","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.patrec.2026.05.009_bib0003","series-title":"2024 50th Euromicro Conference on Software Engineering and Advanced Applications (SEAA)","first-page":"122","article-title":"Semantic-aware representation of multi-modal data for data ingress: a literature review","author":"Lamart","year":"2024"},{"key":"10.1016\/j.patrec.2026.05.009_bib0004","series-title":"Proceedings of the 33rd ACM International Conference on Multimedia","first-page":"189-198","article-title":"Object-preserving counterfactual diffusion augmentation for single-domain generalized object detection","author":"Qin","year":"2025"},{"key":"10.1016\/j.patrec.2026.05.009_bib0005","series-title":"Proceedings of the 32nd ACM International Conference on Multimedia","first-page":"6920-6928","article-title":"White-box multimodal jailbreaks against large vision-language models","author":"Wang","year":"2024"},{"issue":"3","key":"10.1016\/j.patrec.2026.05.009_bib0006","doi-asserted-by":"crossref","first-page":"3146","DOI":"10.1109\/TCSS.2022.3216621","article-title":"A web knowledge-driven multimodal retrieval method in computational social systems: unsupervised and robust graph convolutional hashing","volume":"11","author":"Duan","year":"2024","journal-title":"IEEE Trans. Comput. Soc. Syst."},{"key":"10.1016\/j.patrec.2026.05.009_bib0007","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1007\/s10044-025-01554-2","article-title":"An efficient framework for text-to-image retrieval using complete feature aggregation and cross-knowledge conversion","volume":"28","author":"Ngo","year":"2025","journal-title":"Pattern Anal. Appl."},{"issue":"2","key":"10.1016\/j.patrec.2026.05.009_bib0008","doi-asserted-by":"crossref","first-page":"920","DOI":"10.1109\/TCSVT.2022.3203247","article-title":"Deep supervised dual cycle adversarial network for cross-modal retrieval","volume":"33","author":"Liao","year":"2023","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.patrec.2026.05.009_bib0009","doi-asserted-by":"crossref","first-page":"14733","DOI":"10.1007\/s11042-019-7343-8","article-title":"Semantic consistent adversarial cross-modal retrieval exploiting semantic similarity","volume":"79","author":"Ou","year":"2020","journal-title":"Multimed. Tools Appl."},{"issue":"1","key":"10.1016\/j.patrec.2026.05.009_bib0010","doi-asserted-by":"crossref","DOI":"10.1145\/3284750","article-title":"CM-GANs: cross-modal generative adversarial networks for common representation learning","volume":"15","author":"Peng","year":"2019","journal-title":"ACM Trans. Multimed. Comput. Commun. Appl."},{"issue":"11","key":"10.1016\/j.patrec.2026.05.009_bib0011","doi-asserted-by":"crossref","first-page":"11848","DOI":"10.1109\/LRA.2025.3617729","article-title":"Depth transfer: learning to see like a simulator for real-world drone navigation","volume":"10","author":"Yu","year":"2025","journal-title":"IEEE Robot. Autom. Lett."},{"key":"10.1016\/j.patrec.2026.05.009_bib0012","series-title":"Proceedings of the 28th ACM International Conference on Multimedia","first-page":"1047-1055","article-title":"Context-aware multi-view summarization network for image-text matching","author":"Qu","year":"2020"},{"key":"10.1016\/j.patrec.2026.05.009_bib0013","series-title":"2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"15784","article-title":"Learning the best pooling strategy for visual semantic embedding","author":"Chen","year":"2021"},{"issue":"4","key":"10.1016\/j.patrec.2026.05.009_bib0014","doi-asserted-by":"crossref","first-page":"2008","DOI":"10.1109\/TIP.2018.2882225","article-title":"Bi-directional spatial-semantic attention networks for image-text matching","volume":"28","author":"Huang","year":"2019","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.patrec.2026.05.009_bib0015","first-page":"1218","article-title":"Similarity reasoning and filtration for image-text matching","volume":"35","author":"Diao","year":"2021","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"key":"10.1016\/j.patrec.2026.05.009_bib0016","series-title":"Proceedings of the 2022 International Conference on Multimedia Retrieval","first-page":"239-248","article-title":"Learning hierarchical semantic correspondences for cross-modal image-text retrieval","author":"Zeng","year":"2022"},{"key":"10.1016\/j.patrec.2026.05.009_bib0017","series-title":"International Conference on Machine Learning, Vol 139","article-title":"Learning transferable visual models from natural language supervision","volume":"139","author":"Radford","year":"2021"},{"key":"10.1016\/j.patrec.2026.05.009_bib0018","series-title":"2024 IEEE International Conference on Systems, Man, and Cybernetics (SMC)","first-page":"4163","article-title":"High-performance video retrieval by combining contrastive language image pre-training and cross-modal attention","author":"Ou","year":"2024"},{"key":"10.1016\/j.patrec.2026.05.009_bib0019","series-title":"2024 4th Asia-Pacific Conference on Communications Technology and Computer Science (ACCTCS)","first-page":"418","article-title":"Research on image classification and retrieval based on contrastive language-image pre-training","author":"Liu","year":"2024"},{"key":"10.1016\/j.patrec.2026.05.009_bib0020","article-title":"Align before fuse: vision and language representation learning with momentum distillation","volume":"abs\/2107.07651","author":"Li","year":"2021","journal-title":"CoRR"},{"key":"10.1016\/j.patrec.2026.05.009_bib0021","unstructured":"J. Li, D. Li, S. Savarese, S. Hoi, BLIP-2: bootstrapping language-image pre-training with frozen image encoders and large language models, 2023. arXiv: 2301.12597."},{"issue":"4","key":"10.1016\/j.patrec.2026.05.009_bib0022","doi-asserted-by":"crossref","DOI":"10.1145\/3635310","article-title":"Dynamic weighted adversarial learning for semi-supervised classification under intersectional class mismatch","volume":"20","author":"Li","year":"2024","journal-title":"ACM Trans. Multimed. Comput. Commun. Appl."},{"key":"10.1016\/j.patrec.2026.05.009_bib0023","article-title":"VL-BERT: pre-training of generic visual-linguistic representations","volume":"abs\/1908.08530","author":"Su","year":"2019","journal-title":"CoRR"},{"issue":"1","key":"10.1016\/j.patrec.2026.05.009_bib0024","first-page":"746","article-title":"CALIP: zero-shot enhancement of CLIP with parameter-free attention","volume":"37","author":"Guo","year":"2023","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"key":"10.1016\/j.patrec.2026.05.009_bib0025","series-title":"2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"18145","article-title":"An empirical study of training end-to-end vision-and-language transformers","author":"Dou","year":"2022"},{"key":"10.1016\/j.patrec.2026.05.009_bib0026","series-title":"Computer Vision \u2013 ECCV 2020","first-page":"104","article-title":"UNITER: universal image-text representation learning","author":"Chen","year":"2020"}],"container-title":["Pattern Recognition Letters"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0167865526001777?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0167865526001777?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T12:23:39Z","timestamp":1783167819000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0167865526001777"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8]]},"references-count":26,"alternative-id":["S0167865526001777"],"URL":"https:\/\/doi.org\/10.1016\/j.patrec.2026.05.009","relation":{},"ISSN":["0167-8655"],"issn-type":[{"value":"0167-8655","type":"print"}],"subject":[],"published":{"date-parts":[[2026,8]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"HSAMR: A hierarchical semantic alignment framework for multimodal retrieval","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition Letters","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patrec.2026.05.009","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}]}}