{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,9]],"date-time":"2026-03-09T21:16:17Z","timestamp":1773090977568,"version":"3.50.1"},"reference-count":56,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2023,9,1]],"date-time":"2023-09-01T00:00:00Z","timestamp":1693526400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,9,1]],"date-time":"2023-09-01T00:00:00Z","timestamp":1693526400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"National Key Research and Development Program of China","award":["No.2020YFB1406800"],"award-info":[{"award-number":["No.2020YFB1406800"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Multimed Info Retr"],"published-print":{"date-parts":[[2023,12]]},"DOI":"10.1007\/s13735-023-00298-1","type":"journal-article","created":{"date-parts":[[2023,9,1]],"date-time":"2023-09-01T15:02:26Z","timestamp":1693580546000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["SPSD: Similarity-preserving self-distillation for video\u2013text retrieval"],"prefix":"10.1007","volume":"12","author":[{"given":"Jiachen","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yan","family":"Hua","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yingyun","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hongwei","family":"Kou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,9,1]]},"reference":[{"key":"298_CR1","unstructured":"Vaswani A, Shazeer N, Parmar N, et\u00a0al (2017) Attention is all you need. In: Proceedings of the 31st international conference on neural information processing systems, pp 6000\u20146010"},{"key":"298_CR2","unstructured":"Devlin J, Chang MW, Lee K, et\u00a0al (2018) Bert: pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805"},{"key":"298_CR3","unstructured":"Radford A, Kim JW, Hallacy C, et\u00a0al (2021) Learning transferable visual models from natural language supervision. In: International conference on machine learning. PMLR, pp 8748\u20138763, arXiv:1609.08124"},{"key":"298_CR4","doi-asserted-by":"publisher","unstructured":"Liu S, Fan H, Qian S, et\u00a0al (2021) Hit: hierarchical transformer with momentum contrast for video-text retrieval. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 11,915\u201311,925, https:\/\/doi.org\/10.1109\/ICCV48922.2021.01170","DOI":"10.1109\/ICCV48922.2021.01170"},{"key":"298_CR5","doi-asserted-by":"publisher","unstructured":"Gabeur V, Sun C, Alahari K, et\u00a0al (2020) Multi-modal transformer for video retrieval. In: European conference on computer vision. Springer, pp 214\u2013229, https:\/\/doi.org\/10.1007\/978-3-030-58548-8_13","DOI":"10.1007\/978-3-030-58548-8_13"},{"key":"298_CR6","doi-asserted-by":"publisher","unstructured":"Yang J, Bisk Y, Gao J (2021) Taco: token-aware cascade contrastive learning for video-text alignment. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 11,562\u201311,572, https:\/\/doi.org\/10.1109\/ICCV48922.2021.01136","DOI":"10.1109\/ICCV48922.2021.01136"},{"key":"298_CR7","unstructured":"Li LH, Yatskar M, Yin D, et\u00a0al (2019) Visualbert: a simple and performant baseline for vision and language. arXiv preprint arXiv:1908.03557"},{"key":"298_CR8","unstructured":"Patrick M, Huang PY, Asano Y, et\u00a0al (2021) Support-set bottlenecks for video-text representation learning. In: International conference on learning representations, Vienna, Austria"},{"key":"298_CR9","unstructured":"Liu Y, Albanie S, Nagrani A, et\u00a0al (2019) Use what you have: video retrieval using representations from collaborative experts. arXiv preprint arXiv:1907.13487"},{"key":"298_CR10","unstructured":"Huang Z, Zeng Z, Liu B, et\u00a0al (2020) Pixel-bert: aligning image pixels with text by deep multi-modal transformers. arXiv preprint arXiv:2004.00849v2"},{"key":"298_CR11","doi-asserted-by":"publisher","unstructured":"Li G, Duan N, Fang Y, et\u00a0al (2020) Unicoder-vl: a universal encoder for vision and language by cross-modal pre-training. In: Proceeding of the AAAI conference on artificial intelligence, pp 11,336\u201311,344, https:\/\/doi.org\/10.1609\/aaai.v34i07.6795","DOI":"10.1609\/aaai.v34i07.6795"},{"key":"298_CR12","unstructured":"Su W, Zhu X, Cao Y, et\u00a0al (2020) Vl-bert: pre-training of generic visual-linguistic representations. In: Proceedings of the international conference on learning representation, Addis Ababa, Ethiopia"},{"key":"298_CR13","unstructured":"Lample G, Conneau A (2019) Cross-lingual language model pretraining. arXiv preprint arXiv: 1901.07291"},{"key":"298_CR14","doi-asserted-by":"publisher","unstructured":"Huang H, Liang Y, Duan N, et\u00a0al (2019) Unicoder: a universal language encoder by pre-training with multiple cross-lingual tasks. arXiv preprint https:\/\/doi.org\/10.48550\/arXiv.1909.00964","DOI":"10.48550\/arXiv.1909.00964"},{"key":"298_CR15","doi-asserted-by":"publisher","unstructured":"Tan H, Bansal M (2019) Lxmert: Leaning cross-modality encoder representations from transformers. In: Conference on empirical methods in natural language processing and 9th international joint conference on natural language processing, https:\/\/doi.org\/10.18653\/v1\/D19-1514","DOI":"10.18653\/v1\/D19-1514"},{"key":"298_CR16","doi-asserted-by":"publisher","unstructured":"Khattab O, Zaharia M (2020) Colbert: efficient and effective passage search via contextualized late interaction over Bert. In: Proceedings of the 43rd international ACM SIGIR conference on research and development in information retrieval, pp 39\u201348, https:\/\/doi.org\/10.1145\/3397271.3401075","DOI":"10.1145\/3397271.3401075"},{"key":"298_CR17","doi-asserted-by":"publisher","unstructured":"Yao L, Huang R, Hou L, et\u00a0al (2021) Filip: Fine-grained interactive language-image pre-training. arXiv preprint https:\/\/doi.org\/10.48550\/arXiv.2111.07783","DOI":"10.48550\/arXiv.2111.07783"},{"key":"298_CR18","doi-asserted-by":"publisher","unstructured":"Zhang L, Song J, Gao A, et\u00a0al (2019) Be your own teacher: improve the performance of convolutional neural networks via self distillation. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 3713\u20133722, https:\/\/doi.org\/10.1109\/ICCV.2019.00381","DOI":"10.1109\/ICCV.2019.00381"},{"key":"298_CR19","doi-asserted-by":"publisher","unstructured":"Tung F, Mori G (2019) Similarity-preserving knowledge distillation. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 1365\u20131374, https:\/\/doi.org\/10.1109\/ICCV.2019.00145","DOI":"10.1109\/ICCV.2019.00145"},{"key":"298_CR20","doi-asserted-by":"publisher","unstructured":"Zhu J, Tang S, Chen D, et\u00a0al (2021) Complementary relation contrastive distillation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 9260\u20139269, https:\/\/doi.org\/10.1109\/CVPR46437.2021.00914","DOI":"10.1109\/CVPR46437.2021.00914"},{"key":"298_CR21","doi-asserted-by":"publisher","unstructured":"Ji M, Shin S, Hwang S, et\u00a0al (2021a) Refine myself by teaching myself: feature refinement via self-knowledge distillation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 10,664\u201310,673, https:\/\/doi.org\/10.1109\/CVPR46437.2021.01052","DOI":"10.1109\/CVPR46437.2021.01052"},{"key":"298_CR22","doi-asserted-by":"publisher","unstructured":"Tenney I, Das D, Pavlick E (2019) Bert rediscovers the classical NLP pipeline. In: The 57th annual meeting of the association for computational linguistics (ACL), https:\/\/doi.org\/10.18653\/v1\/P19-1452","DOI":"10.18653\/v1\/P19-1452"},{"key":"298_CR23","doi-asserted-by":"publisher","unstructured":"Hao Y, Dong L, Wei F, et\u00a0al (2019) Visualizing and understanding the effectiveness of Bert. In: conference on empirical methods in natural language processing and 9th international joint conference on natural language processing, https:\/\/doi.org\/10.18653\/v1\/D19-1424","DOI":"10.18653\/v1\/D19-1424"},{"key":"298_CR24","unstructured":"Qiao Y, Xiong C, Liu Z, et\u00a0al (2019) Understanding the behaviors of Bert in ranking. arXiv preprint. arXiv: 1904.07531"},{"key":"298_CR25","doi-asserted-by":"publisher","unstructured":"Vig J (2019) A multiscale visualization of attention in the transformer model. In: The 57th annual meeting of the association for computational linguistics (ACL), p\u00a037, https:\/\/doi.org\/10.18653\/v1\/P19-3007","DOI":"10.18653\/v1\/P19-3007"},{"key":"298_CR26","doi-asserted-by":"crossref","unstructured":"Peters M, Neumann M, Zettlemoyer L, et\u00a0al (2018) Dissecting contextual word embeddings: Architecture and representation. In: Proceedings of the 2018 conference on empirical methods in natural language processing (EMNLP), arXiv:1808.08949","DOI":"10.18653\/v1\/D18-1179"},{"key":"298_CR27","unstructured":"Van\u00a0den Oord A, Li Y, Vinyals O (2018) Representation learning with contrastive predictive coding. arXiv preprint 2(3):4. arXiv:1807.03748"},{"key":"298_CR28","doi-asserted-by":"publisher","unstructured":"Zhu L, Yang Y (2020) Actbert: learning global-local video-text representations. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 8746\u20138755, https:\/\/doi.org\/10.1109\/CVPR42600.2020.00877","DOI":"10.1109\/CVPR42600.2020.00877"},{"key":"298_CR29","unstructured":"Lu J, Batra D, Parikh D, et\u00a0al (2019) Vilbert: pretraining task-agnostic visiolinguistic representations for vision-and-language tasks. In: Proceedings of the 33rd international conference on neural information processing systems, p 13\u201323"},{"key":"298_CR30","doi-asserted-by":"publisher","unstructured":"Yu Y, Kim J, Kim G (2018) A joint sequence fusion model for video question answering and retrieval. In: Proceedings of the European conference on computer vision (ECCV), pp 471\u2013487, https:\/\/doi.org\/10.1007\/978-3-030-01234-2_29","DOI":"10.1007\/978-3-030-01234-2_29"},{"key":"298_CR31","doi-asserted-by":"publisher","unstructured":"Wang X, Zhu L, Yang Y (2021) T2vlad: global-local sequence alignment for text-video retrieval. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 5079\u20135088, https:\/\/doi.org\/10.1109\/CVPR46437.2021.00504","DOI":"10.1109\/CVPR46437.2021.00504"},{"key":"298_CR32","doi-asserted-by":"publisher","unstructured":"Lee KH, Chen X, Hua G, et\u00a0al (2018) Stacked cross attention for image-text matching. In: Proceedings of the European conference on computer vision (ECCV), pp 201\u2013216, https:\/\/doi.org\/10.1007\/978-3-030-01225-0_13","DOI":"10.1007\/978-3-030-01225-0_13"},{"key":"298_CR33","doi-asserted-by":"crossref","unstructured":"Ji K, Liu J, Hong W, et\u00a0al (2022) Cret: cross-modal retrieval transformer for efficient text-video retrieval. In: Proceedings of the 45th international ACM SIGIR conference on research and development in information retrieval, pp 949\u2013959","DOI":"10.1145\/3477495.3531960"},{"key":"298_CR34","doi-asserted-by":"crossref","unstructured":"Gao Y, Lu Z (2023) Cmmt: cross-modal meta-transformer for video-text retrieval. In: Proceedings of the 2023 ACM international conference on multimedia retrieval, pp 76\u201384","DOI":"10.1145\/3591106.3592238"},{"key":"298_CR35","doi-asserted-by":"crossref","unstructured":"Gorti SK, Vouitsis N, Ma J, et\u00a0al (2022) X-pool: cross-modal language-video attention for text-video retrieval. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 5006\u20135015","DOI":"10.1109\/CVPR52688.2022.00495"},{"issue":"108","key":"298_CR36","first-page":"354","volume":"242","author":"M Jin","year":"2022","unstructured":"Jin M, Zhang H, Zhu L et al (2022) Coarse-to-fine dual-level attention for video-text cross modal retrieval. Knowl Syst 242(108):354","journal-title":"Knowl Syst"},{"key":"298_CR37","doi-asserted-by":"publisher","unstructured":"Zolfaghari M, Zhu Y, Gehler P, et\u00a0al (2021) Crossclr: cross-modal contrastive learning for multi-modal video representations. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 1450\u20131459, https:\/\/doi.org\/10.1109\/ICCV48922.2021.00148","DOI":"10.1109\/ICCV48922.2021.00148"},{"key":"298_CR38","unstructured":"Ging S, Zolfaghari M, Pirsiavash H, et\u00a0al (2020) Coot: cooperative hierarchical transformer for video-text representation learning. In: 34th conference on neural information processing systems (NeurIPS 2020), pp 22,605\u201322,618"},{"key":"298_CR39","doi-asserted-by":"publisher","unstructured":"Ji Z, Chen K, Wang H (2021b) Step-wise hierarchical alignment network for image-text matching. In: Proceedings of the thirtieth international joint conference on artificial intelligence, IJCAI-21. international joint conferences on artificial intelligence organization, pp 765\u2013771, https:\/\/doi.org\/10.24963\/ijcai.2021\/106","DOI":"10.24963\/ijcai.2021\/106"},{"key":"298_CR40","doi-asserted-by":"crossref","unstructured":"Jiang J, Min S, Kong W, et\u00a0al (2022b) Tencent text-video retrieval: hierarchical cross-modal interactions with multi-level representations. IEEE Access","DOI":"10.1109\/ACCESS.2022.3227973"},{"key":"298_CR41","doi-asserted-by":"publisher","unstructured":"Hinton G, Vinyals O, Dean J (2015) Distilling the knowledge in a neural network. In: NIPS deep learning workshop, https:\/\/doi.org\/10.4140\/TCP.n.2015.249","DOI":"10.4140\/TCP.n.2015.249"},{"issue":"6","key":"298_CR42","doi-asserted-by":"publisher","first-page":"3048","DOI":"10.1109\/TPAMI.2021.3055564","volume":"44","author":"L Wang","year":"2022","unstructured":"Wang L, Yoon KJ (2022) Knowledge distillation and student-teacher learning for visual intelligence: a review and new outlooks. IEEE Trans Pattern Anal Mach Intell 44(6):3048\u20133068. https:\/\/doi.org\/10.1109\/TPAMI.2021.3055564","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"298_CR43","doi-asserted-by":"publisher","unstructured":"Park W, Kim D, Lu Y, et\u00a0al (2019) Relational knowledge distillation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 3967\u20133976, https:\/\/doi.org\/10.1109\/CVPR.2019.00409","DOI":"10.1109\/CVPR.2019.00409"},{"key":"298_CR44","unstructured":"Tian Y, Krishnan D, Isola P (2020) Contrastive representation distillation. In: International conference on learning representations"},{"key":"298_CR45","doi-asserted-by":"crossref","unstructured":"Li J, Ji Z, Wang G, et\u00a0al (2022) Learning from students: online contrastive distillation network for general continual learning. In: Proc 31st Int Joint Conf Artif Intell, pp 3215\u20133221","DOI":"10.24963\/ijcai.2022\/446"},{"key":"298_CR46","doi-asserted-by":"publisher","unstructured":"Hou Y, Ma Z, Liu C, et\u00a0al (2019) Learning lightweight lane detection cnns by self attention distillation. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 1013\u20131021, https:\/\/doi.org\/10.1109\/ICCV.2019.00110","DOI":"10.1109\/ICCV.2019.00110"},{"key":"298_CR47","doi-asserted-by":"publisher","first-page":"293","DOI":"10.1016\/j.neucom.2022.07.028","volume":"508","author":"H Luo","year":"2022","unstructured":"Luo H, Ji L, Zhong M et al (2022) Clip4clip: an empirical study of clip for end to end video clip retrieval and captioning. Neurocomputing 508:293\u2013304","journal-title":"Neurocomputing"},{"key":"298_CR48","unstructured":"Xue H, Sun Y, Liu B, et\u00a0al (2023) Clip-vip: adapting pre-trained image-text model to video-language representation alignment. In: International conference on learning representations"},{"key":"298_CR49","unstructured":"Jiang H, Zhang J, Huang R, et\u00a0al (2022a) Cross-modal adapter for text-video retrieval. arXiv preprint arXiv:2211.09623"},{"key":"298_CR50","doi-asserted-by":"crossref","unstructured":"Huang S, Gong B, Pan Y, et\u00a0al (2023) Vop: text-video co-operative prompt tuning for cross-modal retrieval. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 6565\u20136574","DOI":"10.1109\/CVPR52729.2023.00635"},{"key":"298_CR51","doi-asserted-by":"crossref","unstructured":"Wang H, He D, Wu W, et\u00a0al (2022) Coder: coupled diversity-sensitive momentum contrastive learning for image-text retrieval. In: 17th European conference on computer vision\u2013ECCV 2022, Springer, pp 700\u2013716","DOI":"10.1007\/978-3-031-20059-5_40"},{"key":"298_CR52","doi-asserted-by":"publisher","unstructured":"Wu Z, Xiong Y, Yu SX, et\u00a0al (2018) Unsupervised feature learning via non-parametric instance discrimination. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 3733\u20133742, https:\/\/doi.org\/10.1109\/CVPR.2018.00393","DOI":"10.1109\/CVPR.2018.00393"},{"key":"298_CR53","doi-asserted-by":"publisher","unstructured":"Xu J, Mei T, Yao T, et\u00a0al (2016) Msr-vtt: A large video description dataset for bridging video and language. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 5288\u20135296, https:\/\/doi.org\/10.1109\/CVPR.2016.571","DOI":"10.1109\/CVPR.2016.571"},{"key":"298_CR54","doi-asserted-by":"publisher","unstructured":"Rohrbach A, Rohrbach M, Tandon N, et\u00a0al (2015) A dataset for movie description. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 3202\u20133212, https:\/\/doi.org\/10.1109\/CVPR.2015.7298940","DOI":"10.1109\/CVPR.2015.7298940"},{"key":"298_CR55","doi-asserted-by":"crossref","unstructured":"Krishna R, Hata K, Ren F, et\u00a0al (2017) Dense-captioning events in videos. In: Proceedings of the IEEE international conference on computer vision, pp 706\u2013715","DOI":"10.1109\/ICCV.2017.83"},{"key":"298_CR56","unstructured":"Loshchilov I, Hutter F (2019) Decoupled weight decay regularization. In: International conference on learning representations"}],"container-title":["International Journal of Multimedia Information Retrieval"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13735-023-00298-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s13735-023-00298-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13735-023-00298-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,12,2]],"date-time":"2023-12-02T14:15:21Z","timestamp":1701526521000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s13735-023-00298-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,9,1]]},"references-count":56,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2023,12]]}},"alternative-id":["298"],"URL":"https:\/\/doi.org\/10.1007\/s13735-023-00298-1","relation":{},"ISSN":["2192-6611","2192-662X"],"issn-type":[{"value":"2192-6611","type":"print"},{"value":"2192-662X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,9,1]]},"assertion":[{"value":"8 March 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 June 2023","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"1 August 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"1 September 2023","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"32"}}