{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T14:57:08Z","timestamp":1784300228197,"version":"3.55.0"},"reference-count":54,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2024,7,22]],"date-time":"2024-07-22T00:00:00Z","timestamp":1721606400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,7,22]],"date-time":"2024-07-22T00:00:00Z","timestamp":1721606400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62202204"],"award-info":[{"award-number":["62202204"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"crossref","award":["JUSRP123032"],"award-info":[{"award-number":["JUSRP123032"]}],"id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Multimed Info Retr"],"published-print":{"date-parts":[[2024,9]]},"DOI":"10.1007\/s13735-024-00335-7","type":"journal-article","created":{"date-parts":[[2024,7,22]],"date-time":"2024-07-22T08:01:46Z","timestamp":1721635306000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":12,"title":["LSECA: local semantic enhancement and cross aggregation for video-text retrieval"],"prefix":"10.1007","volume":"13","author":[{"given":"Zhiwen","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Donglin","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhikai","family":"Hu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,7,22]]},"reference":[{"issue":"2","key":"335_CR1","doi-asserted-by":"publisher","first-page":"32","DOI":"10.1007\/s13735-023-00298-1","volume":"12","author":"J Wang","year":"2023","unstructured":"Wang J, Hua Y, Yang Y, Kou H (2023) Spsd: similarity-preserving self-distillation for video-text retrieval. Int J Multimed Inf Retr 12(2):32","journal-title":"Int J Multimed Inf Retr"},{"key":"335_CR2","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1007\/s13735-018-00166-3","volume":"8","author":"NC Mithun","year":"2019","unstructured":"Mithun NC, Li J, Metze F, Chowdhury AKR (2019) Joint embeddings with multimodal cues for video-text retrieval. Int J Multimed Inf Retr 8:3\u201318","journal-title":"Int J Multimed Inf Retr"},{"key":"335_CR3","doi-asserted-by":"crossref","unstructured":"Gabeur V, Sun C, Alahari K, Schmid C (2020) Multi-modal transformer for video retrieval. In Computer vision\u2013ECCV 2020: 16th european conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part IV 16, pp. 214\u2013229. Springer","DOI":"10.1007\/978-3-030-58548-8_13"},{"key":"335_CR4","unstructured":"Liu Y, Albanie S, Nagrani A, Zisserman A (2019) Use what you have: video retrieval using representations from collaborative experts. arXiv preprint arXiv:1907.13487"},{"key":"335_CR5","doi-asserted-by":"crossref","unstructured":"Lei J, Li L, Zhou L, Gan Z, Berg TL, Bansal M, Liu J (2021) Less is more: clipbert for video-and-language learning via sparse sampling. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 7331\u20137341","DOI":"10.1109\/CVPR46437.2021.00725"},{"key":"335_CR6","doi-asserted-by":"crossref","unstructured":"Bain M, Nagrani A, Varol G, Zisserman A (2021) Frozen in time: A joint video and image encoder for end-to-end retrieval. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp. 1728\u20131738","DOI":"10.1109\/ICCV48922.2021.00175"},{"key":"335_CR7","doi-asserted-by":"crossref","unstructured":"Liu S, Fan H, Qian S, Chen Y, Ding W, Wang Z (2021) Hit: Hierarchical transformer with momentum contrast for video-text retrieval. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp. 11915\u201311925","DOI":"10.1109\/ICCV48922.2021.01170"},{"key":"335_CR8","doi-asserted-by":"crossref","unstructured":"Ge Y, Ge Y, Liu X, Li D, Shan Y, Qie X, Luo P (2022) Bridging video-text retrieval with multiple choice questions. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 16167\u201316176","DOI":"10.1109\/CVPR52688.2022.01569"},{"key":"335_CR9","doi-asserted-by":"crossref","unstructured":"Arnab A, Dehghani M, Heigold G, Sun C, Lu\u010di\u0107 M, Schmid C (2021) Vivit: a video vision transformer. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp. 6836\u20136846","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"335_CR10","unstructured":"Bertasius G, Wang H, Torresani L (2021) Is space-time attention all you need for video understanding? In: Proceedings of the 38th international conference on machine learning, vol 139. PMLR, pp 813\u2013824"},{"key":"335_CR11","doi-asserted-by":"crossref","unstructured":"Zhang D, Wu X-J, Yu J (2021) Discrete bidirectional matrix factorization hashing for zero-shot cross-media retrieval. In: Chinese conference on pattern recognition and computer vision (PRCV), pp. 524\u2013536. Springer","DOI":"10.1007\/978-3-030-88007-1_43"},{"key":"335_CR12","doi-asserted-by":"crossref","unstructured":"Dong J, Li X, Xu C, Ji S, He Y, Yang G, Wang X (2019) Dual encoding for zero-example video retrieval. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 9346\u20139355","DOI":"10.1109\/CVPR.2019.00957"},{"key":"335_CR13","doi-asserted-by":"crossref","unstructured":"Dzabraev M, Kalashnikov M, Komkov S, Petiushko A (2021) Mdmmt: multidomain multimodal transformer for video retrieval. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp. 3354\u20133363","DOI":"10.1109\/CVPRW53098.2021.00374"},{"key":"335_CR14","unstructured":"Radford A, Kim JW, Hallacy C, Ramesh A, Goh G, Agarwal S, Sastry G, Askell A, Mishkin P, Clark J et\u00a0al. (2021) Learning transferable visual models from natural language supervision. In: International conference on machine learning, pp 8748\u20138763. PMLR"},{"key":"335_CR15","unstructured":"Jia C, Yang Y, Xia Y, Chen Y-T, Parekh Z, Pham H, Le Q, Sung Y-H, Li Z, Duerig T (2021) Scaling up visual and vision-language representation learning with noisy text supervision. In: International conference on machine learning, pp 4904\u20134916. PMLR"},{"key":"335_CR16","unstructured":"Yu J, Wang Z, Vasudevan V, Yeung L, Seyedhosseini M, Wu Y (2022) Coca: contrastive captioners are image-text foundation models. arXiv preprint arXiv:2205.01917"},{"key":"335_CR17","doi-asserted-by":"publisher","first-page":"293","DOI":"10.1016\/j.neucom.2022.07.028","volume":"508","author":"H Luo","year":"2022","unstructured":"Luo H, Ji L, Zhong M, Chen Y, Lei W, Duan N, Li T (2022) Clip4clip: an empirical study of clip for end to end video clip retrieval and captioning. Neurocomputing 508:293\u2013304","journal-title":"Neurocomputing"},{"key":"335_CR18","first-page":"38655","volume":"35","author":"C Lin","year":"2022","unstructured":"Lin C, Ancong W, Liang J, Zhang J, Ge W, Zheng W-S, Shen C (2022) Text-adaptive multiple visual prototype matching for video-text retrieval. Adv Neural Inf Process Syst 35:38655\u201338666","journal-title":"Adv Neural Inf Process Syst"},{"key":"335_CR19","doi-asserted-by":"crossref","unstructured":"He F, Wang Q, Feng Z, Jiang W, L\u00fc Y, Zhu Y, Tan X (2021) Improving video retrieval by adaptive margin. In: Proceedings of the 44th international ACM SIGIR conference on research and development in information retrieval, pp 1359\u20131368","DOI":"10.1145\/3404835.3462927"},{"key":"335_CR20","doi-asserted-by":"crossref","unstructured":"Sun C, Myers A, Vondrick C, Murphy K, Schmid C (2019) Videobert: a joint model for video and language representation learning. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 7464\u20137473","DOI":"10.1109\/ICCV.2019.00756"},{"key":"335_CR21","doi-asserted-by":"crossref","unstructured":"Liu Y, Xiong P, Xu L, Cao S, Jin Q (2022) Ts2-net: Token shift and selection transformer for text-video retrieval. In: European conference on computer vision, pp 319\u2013335. Springer","DOI":"10.1007\/978-3-031-19781-9_19"},{"key":"335_CR22","doi-asserted-by":"crossref","unstructured":"Ma Y, Xu G, Sun X, Yan M, Zhang J, Ji R (2022) X-clip: End-to-end multi-grained contrastive learning for video-text retrieval. In: Proceedings of the 30th ACM international conference on multimedia, pp 638\u2013647","DOI":"10.1145\/3503161.3547910"},{"key":"335_CR23","doi-asserted-by":"crossref","unstructured":"Yang J, Bisk Y, Gao J (2021) Taco: token-aware cascade contrastive learning for video-text alignment. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 11562\u201311572","DOI":"10.1109\/ICCV48922.2021.01136"},{"key":"335_CR24","unstructured":"Yao L, Huang R, Hou L, Lu G, Niu M, Xu H, Liang X, Li Z, Jiang X, Xu C (2021) Filip: fine-grained interactive language-image pre-training. arXiv preprint arXiv:2111.07783"},{"key":"335_CR25","doi-asserted-by":"crossref","unstructured":"Li L, Chen Y-C, Cheng Y, Gan Z, Yu L, Liu J (2020) Hero: hierarchical encoder for video+ language omni-representation pre-training. arXiv preprint arXiv:2005.00200","DOI":"10.18653\/v1\/2020.emnlp-main.161"},{"issue":"3","key":"335_CR26","doi-asserted-by":"publisher","first-page":"507","DOI":"10.1109\/TBDATA.2019.2948924","volume":"6","author":"X-Y Tong","year":"2019","unstructured":"Tong X-Y, Xia G-S, Fan H, Zhong Y, Datcu M, Zhang L (2019) Exploiting deep features for remote sensing image retrieval: a systematic investigation. IEEE Transactions on Big Data 6(3):507\u2013521","journal-title":"IEEE Transactions on Big Data"},{"key":"335_CR27","doi-asserted-by":"crossref","unstructured":"Xu J, Mei T, Yao T, Rui Y (2016) Msr-vtt: a large video description dataset for bridging video and language. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 5288\u20135296","DOI":"10.1109\/CVPR.2016.571"},{"key":"335_CR28","doi-asserted-by":"publisher","unstructured":"Wu Z, Yao T, Fu Y, Jiang Y-G (2017) Deep learning for video classification and captioning. In: Frontiers of multimedia research. Association for Computing Machinery and Morgan & Claypool, pp 3\u201329. https:\/\/doi.org\/10.1145\/3122865.3122867","DOI":"10.1145\/3122865.3122867"},{"key":"335_CR29","doi-asserted-by":"crossref","unstructured":"Rohrbach A, Rohrbach M, Schiele B (2015) The long-short story of movie description. In: Pattern recognition: 37th german conference, GCPR 2015, Aachen, Germany, October 7-10, 2015, Proceedings 37, pp 209\u2013221. Springer","DOI":"10.1007\/978-3-319-24947-6_17"},{"key":"335_CR30","doi-asserted-by":"publisher","unstructured":"Sharma P, Li Y (2019) Self-supervised contextual keyword and keyphrase retrieval with self-labelling. Preprints. https:\/\/doi.org\/10.20944\/preprints201908.0073.v1","DOI":"10.20944\/preprints201908.0073.v1"},{"issue":"7","key":"335_CR31","doi-asserted-by":"publisher","first-page":"5947","DOI":"10.1109\/TCYB.2020.3032017","volume":"52","author":"D Zhang","year":"2020","unstructured":"Zhang D, Xiao-Jun W (2020) Scalable discrete matrix factorization and semantic autoencoder for cross-media retrieval. IEEE Transactions on Cybernetics 52(7):5947\u20135960","journal-title":"IEEE Transactions on Cybernetics"},{"key":"335_CR32","unstructured":"Cheng X, Lin H, Wu X, Yang F, Shen D (2021) Improving video-text retrieval by multi-stream corpus alignment and dual softmax loss. arXiv preprint arXiv:2109.04290"},{"key":"335_CR33","doi-asserted-by":"crossref","unstructured":"Wang X, Zhu L, Yang Y (2021) T2vlad: global-local sequence alignment for text-video retrieval. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 5079\u20135088","DOI":"10.1109\/CVPR46437.2021.00504"},{"issue":"2","key":"335_CR34","first-page":"1365","volume":"35","author":"D Zhang","year":"2021","unstructured":"Zhang D, Wu X-J, Xu T, Yin H-F (2021) DAH: discrete asymmetric hashing for efficient cross-media retrieval. IEEE Trans Knowl Data Eng 35(2):1365\u20131378","journal-title":"IEEE Trans Knowl Data Eng"},{"key":"335_CR35","doi-asserted-by":"crossref","unstructured":"Zhao S, Zhu L, Wang X, Yang Y (2022) Centerclip: token clustering for efficient text-video retrieval. In: Proceedings of the 45th international ACM SIGIR conference on research and development in information retrieval, pp 970\u2013981","DOI":"10.1145\/3477495.3531950"},{"key":"335_CR36","unstructured":"Fang H, Xiong P, Xu L, Chen Y (2021) Clip2video: mastering video-text retrieval via image clip. arXiv preprint arXiv:2106.11097"},{"key":"335_CR37","doi-asserted-by":"crossref","unstructured":"Gorti SK, Vouitsis N, Ma J, Golestan K, Volkovs M, Garg A, Yu G (2022) X-pool: Cross-modal language-video attention for text-video retrieval. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 5006\u20135015","DOI":"10.1109\/CVPR52688.2022.00495"},{"issue":"6","key":"335_CR38","doi-asserted-by":"crossref","first-page":"6461","DOI":"10.1109\/TKDE.2022.3178819","volume":"35","author":"D Zhang","year":"2023","unstructured":"Zhang D, Wu X-J, Xu T, Kittler J (2023) Watch: two-stage discrete cross-media hashing. IEEE Trans Knowl Data Eng 35(6):6461\u20136474","journal-title":"IEEE Trans Knowl Data Eng"},{"key":"335_CR39","doi-asserted-by":"publisher","first-page":"19","DOI":"10.1016\/j.patrec.2021.07.018","volume":"151","author":"D Zhang","year":"2021","unstructured":"Zhang D, Xiao-Jun W, Yin H-F, Kittler J (2021) Moon: multi-hash codes joint learning for cross-media retrieval. Pattern Recogn Lett 151:19\u201325","journal-title":"Pattern Recogn Lett"},{"key":"335_CR40","doi-asserted-by":"crossref","unstructured":"Zhang D, Wu X-J, Liu Z, Yu J, Kitter J (2021) Fast discrete cross-modal hashing based on label relaxation and matrix factorization. In: 2020 25th International conference on pattern recognition (ICPR), pp 4845\u20134850. IEEE","DOI":"10.1109\/ICPR48806.2021.9412497"},{"key":"335_CR41","unstructured":"Wang Q, Zhang Y, Zheng Y, Pan P, Hua X-S (2022) Disentangled representation learning for text-video retrieval. arXiv preprint arXiv:2203.07111"},{"key":"335_CR42","doi-asserted-by":"crossref","unstructured":"Wang Z, Sung Y-L, Cheng F, Bertasius G, Bansal M(2023) Unified coarse-to-fine alignment for video-text retrieval. In: Proceedings of the IEEE\/CVF international conference on computer vision (ICCV), pp 2816\u20132827, October","DOI":"10.1109\/ICCV51070.2023.00264"},{"key":"335_CR43","doi-asserted-by":"crossref","unstructured":"Tian K, Zhao R, Xin Z, Lan B, Li X (2024) Holistic features are almost sufficient for text-to-video retrieval. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","DOI":"10.1109\/CVPR52733.2024.01622"},{"key":"335_CR44","doi-asserted-by":"publisher","unstructured":"Chen L, Deng Z, Liu L, Yin S (2024) Multilevel semantic interaction alignment for video\u2013text cross-modal retrieval. IEEE Trans Circuits Syst Video Technol. https:\/\/doi.org\/10.1109\/TCSVT.2024.3360530","DOI":"10.1109\/TCSVT.2024.3360530"},{"key":"335_CR45","doi-asserted-by":"crossref","unstructured":"Deng C, Chen Q, Qin P, Chen D, Wu Q (2023) Prompt switch: efficient clip adaptation for text-video retrieval. In: Proceedings of the IEEE\/CVF international conference on computer vision (ICCV), pp 15648\u201315658, October","DOI":"10.1109\/ICCV51070.2023.01434"},{"key":"335_CR46","doi-asserted-by":"crossref","unstructured":"Fang B, Wu W, Liu C, Zhou Y, Song Y, Wang W, Shu X, Ji X, Wang J(2023) Uatvr: uncertainty-adaptive text-video retrieval. In: Proceedings of the IEEE\/CVF international conference on computer vision (ICCV), pp 13723\u201313733, October","DOI":"10.1109\/ICCV51070.2023.01262"},{"key":"335_CR47","doi-asserted-by":"crossref","unstructured":"Amrani E, Ben-Ari R, Rotman D, Bronstein A (2021) Noise estimation using density estimation for self-supervised multimodal learning. In: Proceedings of the AAAI conference on artificial intelligence 35:6644\u20136652","DOI":"10.1609\/aaai.v35i8.16822"},{"key":"335_CR48","doi-asserted-by":"crossref","unstructured":"Mithun NC, Li J, Metze F, Roy-Chowdhury AK (2018) Learning joint embedding with multimodal cues for cross-modal video-text retrieval. In: Proceedings of the 2018 ACM on international conference on multimedia retrieval, pp 19\u201327","DOI":"10.1145\/3206025.3206064"},{"key":"335_CR49","unstructured":"Patrick M, Huang P-Y, Asano Y, Metze F, Hauptmann A, Henriques J, Vedaldi A (2020) Support-set bottlenecks for video-text representation learning. arXiv preprint arXiv:2010.02824"},{"key":"335_CR50","doi-asserted-by":"crossref","unstructured":"Croitoru I, Bogolin S-V, Leordeanu M, Jin H, Zisserman A, Albanie S, Liu (2021) Teachtext: Crossmodal generalized distillation for text-video retrieval. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 11583\u201311593","DOI":"10.1109\/ICCV48922.2021.01138"},{"key":"335_CR51","doi-asserted-by":"crossref","unstructured":"Yu Y, Ko H, Choi J, Kim G (2017) End-to-end concept word detection for video captioning, retrieval, and question answering. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 3165\u20133173","DOI":"10.1109\/CVPR.2017.347"},{"key":"335_CR52","doi-asserted-by":"publisher","first-page":"2914","DOI":"10.1109\/TMM.2021.3090595","volume":"24","author":"X Song","year":"2021","unstructured":"Song X, Chen J, Zuxuan W, Jiang Y-G (2021) Spatial-temporal graphs for cross-modal text2video retrieval. IEEE Trans Multimedia 24:2914\u20132923","journal-title":"IEEE Trans Multimedia"},{"key":"335_CR53","doi-asserted-by":"crossref","unstructured":"Yu Y, Kim J, Kim G (2018) A joint sequence fusion model for video question answering and retrieval. In: Proceedings of the European conference on computer vision (ECCV), pp 471\u2013487","DOI":"10.1007\/978-3-030-01234-2_29"},{"key":"335_CR54","doi-asserted-by":"crossref","unstructured":"Bogolin S-V, Croitoru I, Jin H, Liu Y, Albanie S (2022) Cross modal retrieval with querybank normalisation. In: Proceedings of the IEEE\/cvf conference on computer vision and pattern recognition, pp 5194\u20135205","DOI":"10.1109\/CVPR52688.2022.00513"}],"container-title":["International Journal of Multimedia Information Retrieval"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13735-024-00335-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s13735-024-00335-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13735-024-00335-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,24]],"date-time":"2024-11-24T15:02:08Z","timestamp":1732460528000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s13735-024-00335-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,7,22]]},"references-count":54,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2024,9]]}},"alternative-id":["335"],"URL":"https:\/\/doi.org\/10.1007\/s13735-024-00335-7","relation":{},"ISSN":["2192-6611","2192-662X"],"issn-type":[{"value":"2192-6611","type":"print"},{"value":"2192-662X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,7,22]]},"assertion":[{"value":"27 September 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 April 2024","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 May 2024","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 July 2024","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that there is no Conflict of interest regarding the content of the study.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"30"}}