{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,19]],"date-time":"2026-01-19T07:04:23Z","timestamp":1768806263891,"version":"3.49.0"},"reference-count":51,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2024,1,22]],"date-time":"2024-01-22T00:00:00Z","timestamp":1705881600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,1,22]],"date-time":"2024-01-22T00:00:00Z","timestamp":1705881600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"University Synergy Innovation Program of Anhui Province","award":["GXXT-2022-043"],"award-info":[{"award-number":["GXXT-2022-043"]}]},{"name":"University Synergy Innovation Program of Anhui Province","award":["GXXT-2022-037"],"award-info":[{"award-number":["GXXT-2022-037"]}]},{"name":"Anhui Provincial Key Research and Development Program","award":["2022a05020042"],"award-info":[{"award-number":["2022a05020042"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["61902104"],"award-info":[{"award-number":["61902104"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2024,2]]},"DOI":"10.1007\/s00530-023-01205-8","type":"journal-article","created":{"date-parts":[[2024,1,22]],"date-time":"2024-01-22T09:02:18Z","timestamp":1705914138000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":6,"title":["Video\u2013text retrieval via multi-modal masked transformer and adaptive attribute-aware graph convolutional network"],"prefix":"10.1007","volume":"30","author":[{"given":"Gang","family":"Lv","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yining","family":"Sun","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fudong","family":"Nian","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,1,22]]},"reference":[{"key":"1205_CR1","doi-asserted-by":"crossref","unstructured":"Amrani, E., Ben-Ari, R., Rotman, D., et\u00a0al: Noise estimation using density estimation for self-supervised multimodal learning. In: Proceedings of the AAAI Conference on Artificial Intelligence, pp 6644\u20136652 (2021)","DOI":"10.1609\/aaai.v35i8.16822"},{"key":"1205_CR2","doi-asserted-by":"crossref","unstructured":"Bain, M., Nagrani, A., Varol, G., et\u00a0al: Frozen in time: A joint video and image encoder for end-to-end retrieval. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 1728\u20131738 (2021)","DOI":"10.1109\/ICCV48922.2021.00175"},{"key":"1205_CR3","doi-asserted-by":"crossref","unstructured":"Barraco, M., Cornia, M., Cascianelli, S., et\u00a0al: The unreasonable effectiveness of clip features for image captioning: an experimental analysis. In: proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 4662\u20134670 (2022)","DOI":"10.1109\/CVPRW56347.2022.00512"},{"key":"1205_CR4","doi-asserted-by":"crossref","unstructured":"Bogolin, S.V., Croitoru, I., Jin, H., et\u00a0al: Cross modal retrieval with querybank normalisation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 5194\u20135205 (2022)","DOI":"10.1109\/CVPR52688.2022.00513"},{"key":"1205_CR5","unstructured":"Chen, D., Dolan, W.B.: Collecting highly parallel data for paraphrase evaluation. In: Proceedings of the 49th annual meeting of the association for computational linguistics: human language technologies, pp 190\u2013200 (2011)"},{"key":"1205_CR6","doi-asserted-by":"crossref","unstructured":"Croitoru, I., Bogolin, S.V., Leordeanu, M., et\u00a0al: Teachtext: Crossmodal generalized distillation for text-video retrieval. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 11583\u201311593 (2021)","DOI":"10.1109\/ICCV48922.2021.01138"},{"key":"1205_CR7","doi-asserted-by":"crossref","unstructured":"Dzabraev, M., Kalashnikov, M., Komkov, S., et\u00a0al: Mdmmt: Multidomain multimodal transformer for video retrieval. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 3354\u20133363 (2021)","DOI":"10.1109\/CVPRW53098.2021.00374"},{"key":"1205_CR8","unstructured":"Fu, T.J., Li, L., Gan, Z., et\u00a0al: Violet: End-to-end video-language transformers with masked visual-token modeling (2021). arXiv preprint arXiv:2111.12681"},{"key":"1205_CR9","doi-asserted-by":"crossref","unstructured":"Gabeur, V., Sun, C., Alahari, K., et\u00a0al: Multi-modal transformer for video retrieval. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part IV 16, Springer, pp 214\u2013229 (2020)","DOI":"10.1007\/978-3-030-58548-8_13"},{"key":"1205_CR10","doi-asserted-by":"crossref","unstructured":"Ge, Y., Ge, Y., Liu, X., et\u00a0al: Bridging video-text retrieval with multiple choice questions. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 16167\u201316176 (2022)","DOI":"10.1109\/CVPR52688.2022.01569"},{"key":"1205_CR11","first-page":"22605","volume":"33","author":"S Ging","year":"2020","unstructured":"Ging, S., Zolfaghari, M., Pirsiavash, H., et al.: Coot: cooperative hierarchical transformer for video-text representation learning. Adv. Neural Inform. Process. Syst. 33, 22605\u201322618 (2020)","journal-title":"Adv. Neural Inform. Process. Syst."},{"key":"1205_CR12","doi-asserted-by":"crossref","unstructured":"Gorti, S.K., Vouitsis, N., Ma, J., et\u00a0al: X-pool: Cross-modal language-video attention for text-video retrieval. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 5006\u20135015 (2022)","DOI":"10.1109\/CVPR52688.2022.00495"},{"key":"1205_CR13","doi-asserted-by":"crossref","unstructured":"Huang, J., Li, Y., Feng, J., et\u00a0al: Clover: Towards a unified video-language alignment and fusion model. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 14856\u201314866 (2023)","DOI":"10.1109\/CVPR52729.2023.01427"},{"key":"1205_CR14","doi-asserted-by":"crossref","unstructured":"Kaufman, D., Levi, G., Hassner, T., et\u00a0al: Temporal tessellation: A unified approach for video analysis. In: Proceedings of the IEEE International Conference on Computer Vision, pp 94\u2013104 (2017)","DOI":"10.1109\/ICCV.2017.20"},{"key":"1205_CR15","doi-asserted-by":"crossref","unstructured":"Kim, D., Park, J., Lee, J., et\u00a0al: Language-free training for zero-shot video grounding. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp 2539\u20132548 (2023)","DOI":"10.1109\/WACV56688.2023.00257"},{"key":"1205_CR16","unstructured":"Kiros, R., Salakhutdinov, R., Zemel, R.S.: Unifying visual-semantic embeddings with multimodal neural language models (2014). arXiv preprint arXiv:1411.2539"},{"key":"1205_CR17","doi-asserted-by":"crossref","unstructured":"Lei, J., Li, L., Zhou, L., et\u00a0al: Less is more: Clipbert for video-and-language learning via sparse sampling. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 7331\u20137341 (2021)","DOI":"10.1109\/CVPR46437.2021.00725"},{"key":"1205_CR18","doi-asserted-by":"crossref","unstructured":"Li, L., Chen, Y.C., Cheng, Y., et\u00a0al: Hero: Hierarchical encoder for video+ language omni-representation pre-training (2020). arXiv preprint arXiv:2005.00200","DOI":"10.18653\/v1\/2020.emnlp-main.161"},{"key":"1205_CR19","doi-asserted-by":"crossref","unstructured":"Li, Q., Han, Z., Wu, X.M.: Deeper insights into graph convolutional networks for semi-supervised learning. In: Proceedings of the AAAI conference on artificial intelligence (2018)","DOI":"10.1609\/aaai.v32i1.11604"},{"key":"1205_CR20","unstructured":"Liu, Y., Albanie, S., Nagrani, A., et\u00a0al: Use what you have: Video retrieval using representations from collaborative experts (2019). arXiv preprint arXiv:1907.13487"},{"key":"1205_CR21","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization (2017). arXiv preprint arXiv:1711.05101"},{"key":"1205_CR22","doi-asserted-by":"publisher","first-page":"293","DOI":"10.1016\/j.neucom.2022.07.028","volume":"508","author":"H Luo","year":"2022","unstructured":"Luo, H., Ji, L., Zhong, M., et al.: Clip4clip: an empirical study of clip for end to end video clip retrieval and captioning. Neurocomputing 508, 293\u2013304 (2022)","journal-title":"Neurocomputing"},{"key":"1205_CR23","doi-asserted-by":"crossref","unstructured":"Ma, Y., Xu, G., Sun, X., et\u00a0al: X-clip: End-to-end multi-grained contrastive learning for video-text retrieval. In: Proceedings of the 30th ACM International Conference on Multimedia, pp 638\u2013647 (2022)","DOI":"10.1145\/3503161.3547910"},{"key":"1205_CR24","unstructured":"Maas, A.L., Hannun, A.Y., Ng, A.Y., et\u00a0al: Rectifier nonlinearities improve neural network acoustic models. In: Proc. icml, Atlanta, Georgia, USA, p 3 (2013)"},{"issue":"11","key":"1205_CR25","doi-asserted-by":"publisher","first-page":"39","DOI":"10.1145\/219717.219748","volume":"38","author":"GA Miller","year":"1995","unstructured":"Miller, G.A.: Wordnet: a lexical database for English. Commun. ACM 38(11), 39\u201341 (1995)","journal-title":"Commun. ACM"},{"issue":"18","key":"1205_CR26","doi-asserted-by":"publisher","first-page":"3346","DOI":"10.3390\/math10183346","volume":"10","author":"F Nian","year":"2022","unstructured":"Nian, F., Ding, L., Hu, Y., et al.: Multi-level cross-modal semantic alignment network for video-text retrieval. Mathematics 10(18), 3346 (2022)","journal-title":"Mathematics"},{"key":"1205_CR27","unstructured":"Oord, A.v.d., Li, Y., Vinyals, O.: Representation learning with contrastive predictive coding (2018). arXiv preprint arXiv:1807.03748"},{"key":"1205_CR28","unstructured":"Patrick, M., Huang, P.Y., Asano, Y., et\u00a0al: Support-set bottlenecks for video-text representation learning (2020). arXiv preprint arXiv:2010.02824"},{"key":"1205_CR29","doi-asserted-by":"crossref","unstructured":"Portillo-Quintero, J.A., Ortiz-Bayliss, J.C., Terashima-Mar\u00edn, H.: A straightforward framework for video retrieval using clip. In: Pattern Recognition: 13th Mexican Conference, MCPR 2021, Mexico City, Mexico, June 23\u201326, 2021, Proceedings, Springer, pp 3\u201312 (2021)","DOI":"10.1007\/978-3-030-77004-4_1"},{"key":"1205_CR30","doi-asserted-by":"crossref","unstructured":"Qi, P., Dozat, T., Zhang, Y., et\u00a0al: Universal dependency parsing from scratch (2019). arXiv preprint arXiv:1901.10457","DOI":"10.18653\/v1\/K18-2016"},{"key":"1205_CR31","doi-asserted-by":"publisher","first-page":"3520","DOI":"10.1109\/TMM.2021.3101642","volume":"24","author":"S Qian","year":"2021","unstructured":"Qian, S., Xue, D., Fang, Q., et al.: Adaptive label-aware graph convolutional networks for cross-modal retrieval. IEEE Trans. Multimed. 24, 3520\u20133532 (2021)","journal-title":"IEEE Trans. Multimed."},{"key":"1205_CR32","unstructured":"Radford, A., Kim, J.W., Hallacy, C., et\u00a0al: Learning transferable visual models from natural language supervision. In: International conference on machine learning, PMLR, pp 8748\u20138763 (2021)"},{"key":"1205_CR33","doi-asserted-by":"publisher","first-page":"94","DOI":"10.1007\/s11263-016-0987-1","volume":"123","author":"A Rohrbach","year":"2017","unstructured":"Rohrbach, A., Torabi, A., Rohrbach, M., et al.: Movie description. Int. J. Comput. Vis. 123, 94\u2013120 (2017)","journal-title":"Int. J. Comput. Vis."},{"key":"1205_CR34","doi-asserted-by":"crossref","unstructured":"Sun, C., Myers, A., Vondrick, C., et\u00a0al: Videobert: a joint model for video and language representation learning. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 7464\u20137473 (2019)","DOI":"10.1109\/ICCV.2019.00756"},{"key":"1205_CR35","unstructured":"Torabi, A., Tandon, N., Sigal, L.: Learning language-visual embedding for movie understanding with natural-language (2016). arXiv preprint arXiv:1609.08124"},{"key":"1205_CR36","doi-asserted-by":"crossref","unstructured":"Wang, J., Ge, Y., Cai, G., et\u00a0al: Object-aware video-language pre-training for retrieval. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp 3313\u20133322 (2022)","DOI":"10.1109\/CVPR52688.2022.00331"},{"key":"1205_CR37","unstructured":"Wang, P., Yang, A., Men, R., et\u00a0al: Ofa: Unifying architectures, tasks, and modalities through a simple sequence-to-sequence learning framework. In: International Conference on Machine Learning, PMLR, pp 23318\u201323340 (2022)"},{"key":"1205_CR38","unstructured":"Wang, Q., Zhang, Y., Zheng, Y., et\u00a0al: Disentangled representation learning for text-video retrieval (2022). arXiv preprint arXiv:2203.07111"},{"key":"1205_CR39","doi-asserted-by":"crossref","unstructured":"Wang, J., Qian, S., Hu, J., et\u00a0al: Positive unlabeled fake news detection via multi-modal masked transformer network. IEEE Trans. Multimed. (2023)","DOI":"10.1109\/TMM.2023.3263552"},{"key":"1205_CR40","unstructured":"Wu, P., He, X., Tang, M., et\u00a0al: Hanet: Hierarchical alignment networks for video-text retrieval. In: Proceedings of the 29th ACM international conference on Multimedia, pp 3518\u20133527 (2021)"},{"key":"1205_CR41","doi-asserted-by":"crossref","unstructured":"Xu, J., Mei, T., Yao, T., et\u00a0al: Msr-vtt: A large video description dataset for bridging video and language. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 5288\u20135296 (2016)","DOI":"10.1109\/CVPR.2016.571"},{"key":"1205_CR42","doi-asserted-by":"crossref","unstructured":"Xu, H., Ghosh, G., Huang, P.Y., et\u00a0al: Videoclip: Contrastive pre-training for zero-shot video-text understanding (2021). arXiv preprint arXiv:2109.14084","DOI":"10.18653\/v1\/2021.emnlp-main.544"},{"key":"1205_CR43","doi-asserted-by":"crossref","unstructured":"Xu, H., Ghosh, G., Huang, P.Y., et\u00a0al: Vlm: Task-agnostic video-language model pre-training for video understanding (2021). arXiv preprint arXiv:2105.09996","DOI":"10.18653\/v1\/2021.findings-acl.370"},{"key":"1205_CR44","unstructured":"Xue, H., Sun, Y., Liu, B., et\u00a0al: Clip-vip: Adapting pre-trained image-text model to video-language representation alignment (2022). arXiv preprint arXiv:2209.06430"},{"key":"1205_CR45","doi-asserted-by":"crossref","unstructured":"Yang, J., Bisk, Y., Gao, J.: Taco: token-aware cascade contrastive learning for video-text alignment. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp 11562\u201311572 (2021)","DOI":"10.1109\/ICCV48922.2021.01136"},{"key":"1205_CR46","doi-asserted-by":"crossref","unstructured":"Yu, Y., Ko, H., Choi, J., et\u00a0al: End-to-end concept word detection for video captioning, retrieval, and question answering. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 3165\u20133173 (2017)","DOI":"10.1109\/CVPR.2017.347"},{"key":"1205_CR47","doi-asserted-by":"crossref","unstructured":"Yu, Y., Kim, J., Kim, G.: A joint sequence fusion model for video question answering and retrieval. In: Proceedings of the European Conference on Computer Vision (ECCV), pp 471\u2013487 (2018)","DOI":"10.1007\/978-3-030-01234-2_29"},{"key":"1205_CR48","doi-asserted-by":"crossref","unstructured":"Zhang, H., Yang, Y., Qi, F., et\u00a0al: Debiased video-text retrieval via soft positive sample calibration. IEEE Transa. Circ. Syst. Video Technol. (2023)","DOI":"10.1109\/TCSVT.2023.3248873"},{"key":"1205_CR49","doi-asserted-by":"crossref","unstructured":"Zhang, H., Yang, Y., Qi, F., et\u00a0al: Robust video-text retrieval via noisy pair calibration. IEEE Trans. Multimed. (2023)","DOI":"10.1109\/TMM.2023.3239183"},{"key":"1205_CR50","doi-asserted-by":"crossref","unstructured":"Zhu, L., Yang, Y.: Actbert: learning global-local video-text representations. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 8746\u20138755 (2020)","DOI":"10.1109\/CVPR42600.2020.00877"},{"issue":"1","key":"1205_CR51","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1007\/s13735-023-00267-8","volume":"12","author":"C Zhu","year":"2023","unstructured":"Zhu, C., Jia, Q., Chen, W., et al.: Deep learning for video-text retrieval: a review. Int. J. Multimed. Inform. Retri. 12(1), 3 (2023)","journal-title":"Int. J. Multimed. Inform. Retri."}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-023-01205-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-023-01205-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-023-01205-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,2,14]],"date-time":"2024-02-14T06:17:29Z","timestamp":1707891449000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-023-01205-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,1,22]]},"references-count":51,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2024,2]]}},"alternative-id":["1205"],"URL":"https:\/\/doi.org\/10.1007\/s00530-023-01205-8","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,1,22]]},"assertion":[{"value":"27 June 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 December 2023","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 January 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"35"}}