{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,22]],"date-time":"2026-07-22T16:46:05Z","timestamp":1784738765974,"version":"3.55.0"},"reference-count":41,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100002338","name":"Ministry of Education of the People's Republic of China","doi-asserted-by":"publisher","award":["24YJCZH289"],"award-info":[{"award-number":["24YJCZH289"]}],"id":[{"id":"10.13039\/501100002338","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100002701","name":"Ministry of Education","doi-asserted-by":"publisher","award":["2024ZD015"],"award-info":[{"award-number":["2024ZD015"]}],"id":[{"id":"10.13039\/501100002701","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100017610","name":"Shenzhen Science and Technology Innovation Program","doi-asserted-by":"publisher","award":["JCYJ20220818103200002"],"award-info":[{"award-number":["JCYJ20220818103200002"]}],"id":[{"id":"10.13039\/501100017610","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100017610","name":"Shenzhen Science and Technology Innovation Program","doi-asserted-by":"publisher","award":["JCYJ20240813114223031"],"award-info":[{"award-number":["JCYJ20240813114223031"]}],"id":[{"id":"10.13039\/501100017610","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003453","name":"Natural Science Foundation of Guangdong Province","doi-asserted-by":"publisher","award":["2024A1515011155"],"award-info":[{"award-number":["2024A1515011155"]}],"id":[{"id":"10.13039\/501100003453","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.patcog.2026.113359","type":"journal-article","created":{"date-parts":[[2026,3,3]],"date-time":"2026-03-03T17:16:36Z","timestamp":1772558196000},"page":"113359","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":1,"special_numbering":"C","title":["Bridging the semantic gap in text-video retrieval: A diffusion generation network for enhanced visual representation"],"prefix":"10.1016","volume":"178","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5439-1508","authenticated-orcid":false,"given":"Songtao","family":"Ding","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-5153-2051","authenticated-orcid":false,"given":"Chun","family":"Mao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-2224-8708","authenticated-orcid":false,"given":"Chun","family":"Geng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4556-9546","authenticated-orcid":false,"given":"Hongyu","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-8869-7476","authenticated-orcid":false,"given":"Yaqiong","family":"Xing","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3550-8026","authenticated-orcid":false,"given":"Guohua","family":"Lv","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-5567-2349","authenticated-orcid":false,"given":"Xiaoying","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7013-9081","authenticated-orcid":false,"given":"Shaohua","family":"Wan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.patcog.2026.113359_bib0001","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"5612","article-title":"T2V2T: text-to-video-to-text fusion for text-to-video retrieval","author":"Kim","year":"2023"},{"key":"10.1016\/j.patcog.2026.113359_bib0002","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV)","first-page":"13723","article-title":"UATVR: uncertainty-adaptive text-video retrieval","author":"Fang","year":"2023"},{"key":"10.1016\/j.patcog.2026.113359_bib0003","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"3650","article-title":"On semantic similarity in video retrieval","author":"Wray","year":"2021"},{"key":"10.1016\/j.patcog.2026.113359_bib0004","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"10704","article-title":"Cap4Video: what can auxiliary captions do for text-video retrieval?","author":"Wu","year":"2023"},{"key":"10.1016\/j.patcog.2026.113359_bib0005","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"10638","article-title":"Fine-grained video-text retrieval with hierarchical graph reasoning","author":"Chen","year":"2020"},{"key":"10.1016\/j.patcog.2026.113359_bib0006","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV)","first-page":"4100","article-title":"Progressive spatio-temporal prototype matching for text-video retrieval","author":"Li","year":"2023"},{"issue":"8","key":"10.1016\/j.patcog.2026.113359_bib0007","doi-asserted-by":"crossref","first-page":"5680","DOI":"10.1109\/TCSVT.2022.3150959","article-title":"Reading-strategy inspired visual representation learning for text-to-video retrieval","volume":"32","author":"Dong","year":"2022","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.patcog.2026.113359_bib0008","series-title":"Proceedings of the European Conference on Computer Vision (ECCV)","first-page":"214","article-title":"Multi-modal transformer for video retrieval","author":"Gabeur","year":"2020"},{"key":"10.1016\/j.patcog.2026.113359_bib0009","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV)","first-page":"12054","article-title":"Audio-enhanced text-to-video retrieval using text-conditioned feature alignment","author":"Ibrahimi","year":"2023"},{"key":"10.1016\/j.patcog.2026.113359_bib0010","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2021.115541","article-title":"Multimodal video-text matching using a deep bifurcation network and joint embedding of visual and textual features","volume":"184","author":"Nabati","year":"2021","journal-title":"Expert Syst. Appl."},{"issue":"6","key":"10.1016\/j.patcog.2026.113359_bib0011","doi-asserted-by":"crossref","first-page":"76","DOI":"10.1109\/MC.2010.183","article-title":"Visual-concept search solved?","volume":"43","author":"Snoek","year":"2010","journal-title":"Computer"},{"key":"10.1016\/j.patcog.2026.113359_bib0012","unstructured":"Y. Liu, S. Albanie, A. Nagrani, A. Zisserman, Use what you have: video retrieval using representations from collaborative experts, arXiv: 1907.13487(2019)."},{"key":"10.1016\/j.patcog.2026.113359_bib0013","series-title":"Proceedings of the 43rd International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR)","first-page":"1339","article-title":"Tree-augmented cross-modal encoding for complex-query video retrieval","author":"Yang","year":"2020"},{"key":"10.1016\/j.patcog.2026.113359_bib0014","series-title":"Proceedings of the 25th ACM International Conference on Multimedia (ACM MM)","first-page":"154","article-title":"Adversarial cross-modal retrieval","author":"Wang","year":"2017"},{"key":"10.1016\/j.patcog.2026.113359_bib0015","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"15789","article-title":"Learning the best pooling strategy for visual semantic embedding","author":"Chen","year":"2021"},{"key":"10.1016\/j.patcog.2026.113359_bib0016","series-title":"Proceedings of the 12th IEEE International Symposium on Electronics and Telecommunications (ISETC)","first-page":"327","article-title":"Video retrieval using relevant topics extraction from movie subtitles","author":"Mocanu","year":"2016"},{"key":"10.1016\/j.patcog.2026.113359_bib0017","doi-asserted-by":"crossref","first-page":"11954","DOI":"10.1109\/TCSVT.2024.3429192","article-title":"UMP: unified modality-aware prompt tuning for text-video retrieval","volume":"34","author":"Zhang","year":"2024","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.patcog.2026.113359_bib0018","doi-asserted-by":"crossref","first-page":"3463","DOI":"10.1109\/TIP.2025.3574925","article-title":"Text-video retrieval with global-local semantic consistent learning","volume":"34","author":"Zhang","year":"2025","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.patcog.2026.113359_bib0019","unstructured":"Z. Shao, Z. Zhang, Z. Lin, J. Wu, J. Xiao, Bridging Video-text Retrieval with Video Captioning, arXiv: 2209.11391(2022)."},{"key":"10.1016\/j.patcog.2026.113359_bib0020","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV)","first-page":"2470","article-title":"DiffusionRet: generative text-video retrieval with diffusion model","author":"Jin","year":"2023"},{"key":"10.1016\/j.patcog.2026.113359_bib0021","doi-asserted-by":"crossref","first-page":"319","DOI":"10.22219\/kinetik.v3i4.680","article-title":"Indonesian dataset expansion of microsoft research video description corpus and its similarity analysis","author":"Rahutomo","year":"2018","journal-title":"Kinetik: Game Technol. Inf. Syst. Comput. Netw. Comput. Electr. Control"},{"key":"10.1016\/j.patcog.2026.113359_bib0022","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV)","first-page":"4581","article-title":"VATEX: a large-scale, high-quality multilingual dataset for video-and-language research","author":"Wang","year":"2019"},{"key":"10.1016\/j.patcog.2026.113359_bib0023","series-title":"Proceedings of the 49th Annual Meeting of the Association for Computational Linguistics (ACL)","first-page":"190","article-title":"Collecting highly parallel data for paraphrase evaluation","author":"Chen","year":"2011"},{"key":"10.1016\/j.patcog.2026.113359_bib0024","series-title":"Proceedings of the IEEE International Conference on Computer Vision (ICCV)","first-page":"5803","article-title":"Localizing moments in video with natural language","author":"Anne Hendricks","year":"2017"},{"issue":"8","key":"10.1016\/j.patcog.2026.113359_bib0025","first-page":"4065","article-title":"Dual encoding for video retrieval by text","volume":"44","author":"Dong","year":"2021","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.patcog.2026.113359_bib0026","series-title":"Proceedings of the 27th ACM International Conference on Multimedia (ACM MM)","first-page":"1786","article-title":"W2VV++: fully deep learning for ad-hoc video search","author":"Li","year":"2019"},{"key":"10.1016\/j.patcog.2026.113359_bib0027","unstructured":"H. Luo, L. Ji, B. Shi, H. Huang, N. Duan, T. Li, J. Li, T. Bharti, M. Zhou, Univl: a unified video and language pre-training model for multimodal understanding and generation, arXiv: 2002.06353(2020)."},{"issue":"3","key":"10.1016\/j.patcog.2026.113359_bib0028","doi-asserted-by":"crossref","first-page":"1438","DOI":"10.1109\/TCSVT.2022.3207910","article-title":"Temporal multimodal graph transformer with global-local alignment for video-text retrieval","volume":"33","author":"Feng","year":"2022","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"issue":"10","key":"10.1016\/j.patcog.2026.113359_bib0029","doi-asserted-by":"crossref","first-page":"5486","DOI":"10.1109\/TCSVT.2023.3257193","article-title":"Using multimodal contrastive knowledge distillation for video-text retrieval","volume":"33","author":"Ma","year":"2023","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.patcog.2026.113359_bib0030","doi-asserted-by":"crossref","DOI":"10.1016\/j.imavis.2024.105168","article-title":"Transferable dual multi-granularity semantic excavating for partially relevant video retrieval","volume":"149","author":"Cheng","year":"2024","journal-title":"Image Vis. Comput."},{"key":"10.1016\/j.patcog.2026.113359_bib0031","doi-asserted-by":"crossref","DOI":"10.1016\/j.artint.2024.104235","article-title":"TeachText: cross-modal text-video retrieval through generalized distillation","volume":"338","author":"Croitoru","year":"2025","journal-title":"Artif. Intell."},{"key":"10.1016\/j.patcog.2026.113359_bib0032","series-title":"International Conference on Machine Learning (ICML)","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.patcog.2026.113359_bib0033","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV)","first-page":"1728","article-title":"Frozen in time: a joint video and image encoder for end-to-end retrieval","author":"Bain","year":"2021"},{"key":"10.1016\/j.patcog.2026.113359_bib0034","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence (AAAI)","first-page":"6644","article-title":"Noise estimation using density estimation for self-supervised multimodal learning","volume":"35","author":"Amrani","year":"2021"},{"key":"10.1016\/j.patcog.2026.113359_bib0035","first-page":"38655","article-title":"Text-adaptive multiple visual prototype matching for video-text retrieval","volume":"35","author":"Lin","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.113359_bib0036","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"5194","article-title":"Cross modal retrieval with querybank normalisation","author":"Bogolin","year":"2022"},{"key":"10.1016\/j.patcog.2026.113359_bib0037","series-title":"Proceedings of the European Conference on Computer Vision (ECCV)","first-page":"319","article-title":"Ts2-Net: token shift and selection transformer for text-video retrieval","author":"Liu","year":"2022"},{"key":"10.1016\/j.patcog.2026.113359_bib0038","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"7331","article-title":"Less is more: clipbert for video-and-language learning via sparse sampling","author":"Lei","year":"2021"},{"key":"10.1016\/j.patcog.2026.113359_bib0039","series-title":"Proceedings of the 45th International ACM SIGIR Conference on Research and Development in Information Retrieval (SIGIR)","first-page":"949","article-title":"Cret: cross-modal retrieval transformer for efficient text-video retrieval","author":"Ji","year":"2022"},{"key":"10.1016\/j.patcog.2026.113359_bib0040","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"16167","article-title":"Bridging video-text retrieval with multiple choice questions","author":"Ge","year":"2022"},{"key":"10.1016\/j.patcog.2026.113359_bib0041","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"4953","article-title":"Align and prompt: video-and-language pre-training with entity prompts","author":"Li","year":"2022"}],"container-title":["Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326003249?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326003249?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,5,21]],"date-time":"2026-05-21T16:56:08Z","timestamp":1779382568000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0031320326003249"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":41,"alternative-id":["S0031320326003249"],"URL":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113359","relation":{},"ISSN":["0031-3203"],"issn-type":[{"value":"0031-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Bridging the semantic gap in text-video retrieval: A diffusion generation network for enhanced visual representation","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113359","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Published by Elsevier Ltd.","name":"copyright","label":"Copyright"}],"article-number":"113359"}}