{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T16:46:56Z","timestamp":1783788416924,"version":"3.55.0"},"reference-count":37,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2024,10,15]],"date-time":"2024-10-15T00:00:00Z","timestamp":1728950400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,15]],"date-time":"2024-10-15T00:00:00Z","timestamp":1728950400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62376016"],"award-info":[{"award-number":["62376016"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61902204"],"award-info":[{"award-number":["61902204"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2024,12]]},"DOI":"10.1007\/s00530-024-01525-3","type":"journal-article","created":{"date-parts":[[2024,10,15]],"date-time":"2024-10-15T05:01:35Z","timestamp":1728968495000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":14,"title":["Hierarchical bi-directional conceptual interaction for text-video retrieval"],"prefix":"10.1007","volume":"30","author":[{"given":"Wenpeng","family":"Han","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Guanglin","family":"Niu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mingliang","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaowei","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,10,15]]},"reference":[{"key":"1525_CR1","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., et al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763 (2021)"},{"key":"1525_CR2","doi-asserted-by":"crossref","unstructured":"Gorti, S.K., Vouitsis, N., Ma, J., Golestan, K., Volkovs, M., Garg, A., Yu, G.: X-pool: cross-modal language-video attention for text-video retrieval. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5006\u20135015 (2022)","DOI":"10.1109\/CVPR52688.2022.00495"},{"key":"1525_CR3","doi-asserted-by":"publisher","first-page":"293","DOI":"10.1016\/j.neucom.2022.07.028","volume":"508","author":"H Luo","year":"2022","unstructured":"Luo, H., Ji, L., Zhong, M., Chen, Y., Lei, W., Duan, N., Li, T.: Clip4clip: an empirical study of clip for end to end video clip retrieval and captioning. Neurocomputing 508, 293\u2013304 (2022)","journal-title":"Neurocomputing"},{"key":"1525_CR4","doi-asserted-by":"crossref","unstructured":"Liu, R., Huang, J., Li, G., Feng, J., Wu, X., Li, T.H.: Revisiting temporal modeling for clip-based image-to-video knowledge transferring. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6555\u20136564 (2023)","DOI":"10.1109\/CVPR52729.2023.00634"},{"key":"1525_CR5","doi-asserted-by":"crossref","unstructured":"Jin, P., Huang, J., Xiong, P., Tian, S., Liu, C., Ji, X., Yuan, L., Chen, J.: Video-text as game players: hierarchical Banzhaf interaction for cross-modal representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2472\u20132482 (2023)","DOI":"10.1109\/CVPR52729.2023.00244"},{"key":"1525_CR6","doi-asserted-by":"crossref","unstructured":"Zhao, S., Zhu, L., Wang, X., Yang, Y.: Centerclip: token clustering for efficient text-video retrieval. In: Proceedings of the 45th International ACM SIGIR Conference on Research and Development in Information Retrieval, pp. 970\u2013981 (2022)","DOI":"10.1145\/3477495.3531950"},{"key":"1525_CR7","doi-asserted-by":"crossref","unstructured":"Jin, P., Li, H., Cheng, Z., Huang, J., Wang, Z., Yuan, L., Liu, C., Chen, J.: Text-video retrieval with disentangled conceptualization and set-to-set alignment. In: Proceedings of the Thirty-Second International Joint Conference on Artificial Intelligence, pp. 938\u2013946 (2023)","DOI":"10.24963\/ijcai.2023\/104"},{"issue":"1","key":"1525_CR8","doi-asserted-by":"publisher","first-page":"35","DOI":"10.1007\/s00530-023-01205-8","volume":"30","author":"G Lv","year":"2024","unstructured":"Lv, G., Sun, Y., Nian, F.: Video-text retrieval via multi-modal masked transformer and adaptive attribute-aware graph convolutional network. Multimed. Syst. 30(1), 35 (2024)","journal-title":"Multimed. Syst."},{"key":"1525_CR9","doi-asserted-by":"crossref","unstructured":"Jin, P., Li, H., Cheng, Z., Li, K., Ji, X., Liu, C., Yuan, L., Chen, J.: Diffusionret: generative text-video retrieval with diffusion model. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2470\u20132481 (2023)","DOI":"10.1109\/ICCV51070.2023.00234"},{"key":"1525_CR10","doi-asserted-by":"crossref","unstructured":"Li, Q., Su, L., Zhao, J., Xia, L., Cai, H., Cheng, S., Tang, H., Wang, J., Yin, D.: Text-video retrieval via multi-modal hypergraph networks. In: Proceedings of the 17th ACM International Conference on Web Search and Data Mining, pp. 369\u2013377 (2024)","DOI":"10.1145\/3616855.3635757"},{"key":"1525_CR11","doi-asserted-by":"crossref","unstructured":"Chen, Y., Wang, J., Lin, L., Qi, Z., Ma, J., Shan, Y.: Tagging before alignment: integrating multi-modal tags for video-text retrieval. In: Proceedings of the AAAI Conference on Artificial Intelligence, pp. 396\u2013404 (2023)","DOI":"10.1609\/aaai.v37i1.25113"},{"key":"1525_CR12","doi-asserted-by":"crossref","unstructured":"Wu, W., Luo, H., Fang, B., Wang, J., Ouyang, W.: Cap4video: what can auxiliary captions do for text-video retrieval? In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10704\u201310713 (2023)","DOI":"10.1109\/CVPR52729.2023.01031"},{"key":"1525_CR13","doi-asserted-by":"crossref","unstructured":"Chen, S., Zhao, Y., Jin, Q., Wu, Q.: Fine-grained video-text retrieval with hierarchical graph reasoning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10638\u201310647 (2020)","DOI":"10.1109\/CVPR42600.2020.01065"},{"key":"1525_CR14","doi-asserted-by":"crossref","unstructured":"Wang, Y., Yang, Y., Zhang, W., Wang, J.: Improved clip cross-modal retrieval model for fine-grained interactions. In: 2023 IEEE Smart World Congress (SWC), pp. 593\u2013598 (2023)","DOI":"10.1109\/SWC57546.2023.10449025"},{"key":"1525_CR15","doi-asserted-by":"crossref","unstructured":"Ge, X., Chen, F., Xu, S., Tao, F., Jose, J.M.: Cross-modal semantic enhanced interaction for image-sentence retrieval. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 1022\u20131031 (2023)","DOI":"10.1109\/WACV56688.2023.00108"},{"issue":"6","key":"1525_CR16","doi-asserted-by":"publisher","DOI":"10.1016\/j.ipm.2023.103508","volume":"60","author":"L Xiao","year":"2023","unstructured":"Xiao, L., Wu, X., Yang, S., Xu, J., Zhou, J., He, L.: Cross-modal fine-grained alignment and fusion network for multimodal aspect-based sentiment analysis. Inf. Process. Manag. 60(6), 103508 (2023)","journal-title":"Inf. Process. Manag."},{"key":"1525_CR17","doi-asserted-by":"crossref","unstructured":"Che, Z., Cui, G.: Cross-modal fine-grained interaction fusion in fake news detection. Int. J. Adv. Comput. Sci. Appl. 15(5) (2024)","DOI":"10.14569\/IJACSA.2024.0150596"},{"issue":"5","key":"1525_CR18","doi-asserted-by":"publisher","first-page":"2469","DOI":"10.1007\/s00530-023-01130-w","volume":"29","author":"M Zhong","year":"2023","unstructured":"Zhong, M., Chen, Y., Zhang, H., Xiong, H., Wang, Z.: Multimodal-enhanced hierarchical attention network for video captioning. Multimed. Syst. 29(5), 2469\u20132482 (2023)","journal-title":"Multimed. Syst."},{"issue":"2","key":"1525_CR19","doi-asserted-by":"publisher","first-page":"83","DOI":"10.1007\/s00530-024-01282-3","volume":"30","author":"M Xue","year":"2024","unstructured":"Xue, M., Xu, Z., Qiao, S., Zheng, J., Li, T., Wang, Y., Peng, D.: Driver intention prediction based on multi-dimensional cross-modality information interaction. Multimed. Syst. 30(2), 83 (2024)","journal-title":"Multimed. Syst."},{"issue":"1","key":"1525_CR20","doi-asserted-by":"publisher","DOI":"10.1016\/j.ipm.2023.103546","volume":"61","author":"P Liu","year":"2024","unstructured":"Liu, P., Wang, G., Li, H., Liu, J., Ren, Y., Zhu, H., Sun, L.: Multi-granularity cross-modal representation learning for named entity recognition on social media. Inf. Process. Manag. 61(1), 103546 (2024)","journal-title":"Inf. Process. Manag."},{"key":"1525_CR21","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2023.102132","volume":"103","author":"J Wang","year":"2024","unstructured":"Wang, J., Yang, Y., Jiang, Y., Ma, M., Xie, Z., Li, T.: Cross-modal incongruity aligning and collaborating for multi-modal sarcasm detection. Inf. Fusion 103, 102132 (2024)","journal-title":"Inf. Fusion"},{"key":"1525_CR22","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2024.102460","volume":"110","author":"H Tang","year":"2024","unstructured":"Tang, H., Hu, Y., Wang, Y., Zhang, S., Xu, M., Zhu, J., Zheng, Q.: Listen as you wish: fusion of audio and text for cross-modal event detection in smart cities. Inf. Fusion 110, 102460 (2024)","journal-title":"Inf. Fusion"},{"key":"1525_CR23","doi-asserted-by":"crossref","unstructured":"Long, S., Han, S.C., Wan, X., Poon, J.: Gradual: graph-based dual-modal representation for image-text matching. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 3459\u20133468 (2022)","DOI":"10.1109\/WACV51458.2022.00252"},{"key":"1525_CR24","doi-asserted-by":"crossref","unstructured":"Xu, J., Mei, T., Yao, T., Rui, Y.: Msr-vtt: a large video description dataset for bridging video and language. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5288\u20135296 (2016)","DOI":"10.1109\/CVPR.2016.571"},{"key":"1525_CR25","doi-asserted-by":"crossref","unstructured":"Miech, A., Zhukov, D., Alayrac, J.-B., Tapaswi, M., Laptev, I., Sivic, J.: Howto100m: learning a text-video embedding by watching hundred million narrated video clips. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2630\u20132640 (2019)","DOI":"10.1109\/ICCV.2019.00272"},{"key":"1525_CR26","doi-asserted-by":"crossref","unstructured":"Yu, Y., Kim, J., Kim, G.: A joint sequence fusion model for video question answering and retrieval. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 471\u2013487 (2018)","DOI":"10.1007\/978-3-030-01234-2_29"},{"key":"1525_CR27","doi-asserted-by":"crossref","unstructured":"Krishna, R., Hata, K., Ren, F., Fei-Fei, L., Carlos\u00a0Niebles, J.: Dense-captioning events in videos. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 706\u2013715 (2017)","DOI":"10.1109\/ICCV.2017.83"},{"key":"1525_CR28","doi-asserted-by":"crossref","unstructured":"Anne\u00a0Hendricks, L., Wang, O., Shechtman, E., Sivic, J., Darrell, T., Russell, B.: Localizing moments in video with natural language. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 5803\u20135812 (2017)","DOI":"10.1109\/ICCV.2017.618"},{"key":"1525_CR29","doi-asserted-by":"crossref","unstructured":"Zhu, L., Yang, Y.: Actbert: learning global-local video-text representations. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8746\u20138755 (2020)","DOI":"10.1109\/CVPR42600.2020.00877"},{"key":"1525_CR30","doi-asserted-by":"crossref","unstructured":"Amrani, E., Ben-Ari, R., Rotman, D., Bronstein, A.: Noise estimation using density estimation for self-supervised multimodal learning. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 35, pp. 6644\u20136652 (2021)","DOI":"10.1609\/aaai.v35i8.16822"},{"key":"1525_CR31","doi-asserted-by":"crossref","unstructured":"Lei, J., Li, L., Zhou, L., Gan, Z., Berg, T.L., Bansal, M., Liu, J.: Less is more: Clipbert for video-and-language learning via sparse sampling. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7331\u20137341 (2021)","DOI":"10.1109\/CVPR46437.2021.00725"},{"key":"1525_CR32","doi-asserted-by":"crossref","unstructured":"Gabeur, V., Sun, C., Alahari, K., Schmid, C.: Multi-modal transformer for video retrieval. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, August 23\u201328, 2020, Proceedings, Part IV 16, pp. 214\u2013229 (2020)","DOI":"10.1007\/978-3-030-58548-8_13"},{"key":"1525_CR33","doi-asserted-by":"crossref","unstructured":"Croitoru, I., Bogolin, S.-V., Leordeanu, M., Jin, H., Zisserman, A., Albanie, S., Liu, Y.: Teachtext: crossmodal generalized distillation for text-video retrieval. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 11583\u201311593 (2021)","DOI":"10.1109\/ICCV48922.2021.01138"},{"key":"1525_CR34","doi-asserted-by":"crossref","unstructured":"Bain, M., Nagrani, A., Varol, G., Zisserman, A.: Frozen in time: a joint video and image encoder for end-to-end retrieval. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1728\u20131738 (2021)","DOI":"10.1109\/ICCV48922.2021.00175"},{"key":"1525_CR35","doi-asserted-by":"crossref","unstructured":"Wang, X., Zhu, L., Yang, Y.: T2vlad: global-local sequence alignment for text-video retrieval. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5079\u20135088 (2021)","DOI":"10.1109\/CVPR46437.2021.00504"},{"key":"1525_CR36","doi-asserted-by":"crossref","unstructured":"Liu, Y., Xiong, P., Xu, L., Cao, S., Jin, Q.: Ts2-net: token shift and selection transformer for text-video retrieval. In: European Conference on Computer Vision, pp. 319\u2013335 (2022)","DOI":"10.1007\/978-3-031-19781-9_19"},{"key":"1525_CR37","doi-asserted-by":"crossref","unstructured":"Zhang, B., Hu, H., Sha, F.: Cross-modal and hierarchical modeling of video and text. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 374\u2013390 (2018)","DOI":"10.1007\/978-3-030-01261-8_23"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-024-01525-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-024-01525-3\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-024-01525-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,16]],"date-time":"2024-12-16T09:08:29Z","timestamp":1734340109000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-024-01525-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,15]]},"references-count":37,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2024,12]]}},"alternative-id":["1525"],"URL":"https:\/\/doi.org\/10.1007\/s00530-024-01525-3","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,15]]},"assertion":[{"value":"16 July 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"1 October 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 October 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"317"}}