{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,6]],"date-time":"2026-07-06T05:18:14Z","timestamp":1783315094032,"version":"3.54.6"},"reference-count":40,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2026,3,10]],"date-time":"2026-03-10T00:00:00Z","timestamp":1773100800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,3,10]],"date-time":"2026-03-10T00:00:00Z","timestamp":1773100800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"China National Social Science Fund Art General Project","award":["21BD073"],"award-info":[{"award-number":["21BD073"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1007\/s00530-026-02260-7","type":"journal-article","created":{"date-parts":[[2026,3,10]],"date-time":"2026-03-10T13:30:34Z","timestamp":1773149434000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Efficient video-to-music retrieval with dual-path encoding"],"prefix":"10.1007","volume":"32","author":[{"given":"Tao","family":"Xu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qiangsheng","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dong","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,3,10]]},"reference":[{"key":"2260_CR1","doi-asserted-by":"crossref","unstructured":"Pr\u00e9tet, L., Richard, G., Peeters, G.: Cross-modal music-video recommendation: a study of design choices. In: 2021 International Joint Conference on Neural Networks (IJCNN), pp. 1\u20139 (2021). IEEE","DOI":"10.1109\/IJCNN52387.2021.9533662"},{"key":"2260_CR2","doi-asserted-by":"crossref","unstructured":"Zeng, D., Yu, Y., Oyama, K.: Audio-visual embedding for cross-modal music video retrieval through supervised deep cca. In: 2018 IEEE International Symposium on Multimedia (ISM), pp. 143\u2013150 (2018). IEEE","DOI":"10.1109\/ISM.2018.00-21"},{"key":"2260_CR3","first-page":"9758","volume":"33","author":"H Alwassel","year":"2020","unstructured":"Alwassel, H., Mahajan, D., Korbar, B., Torresani, L., Ghanem, B., Tran, D.: Self-supervised learning by cross-modal audio-video clustering. Adv. Neural. Inf. Process. Syst. 33, 9758\u20139770 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"issue":"2","key":"2260_CR4","doi-asserted-by":"publisher","first-page":"805","DOI":"10.3390\/s23020805","volume":"23","author":"X Gu","year":"2023","unstructured":"Gu, X., Shen, Y., Lv, C.: A dual-path cross-modal network for video-music retrieval. Sensors 23(2), 805 (2023)","journal-title":"Sensors"},{"key":"2260_CR5","doi-asserted-by":"crossref","unstructured":"Dong, Z., Liu, X., Chen, B., Polak, P., Zhang, P.: Musechat: A conversational music recommendation system for videos. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12775\u201312785 (2024)","DOI":"10.1109\/CVPR52733.2024.01214"},{"key":"2260_CR6","doi-asserted-by":"crossref","unstructured":"Hong, S., Im, W., Yang, H.S.: Cbvmr: content-based video-music retrieval using soft intra-modal structure constraint. In: Proceedings of the 2018 ACM on International Conference on Multimedia Retrieval, pp. 353\u2013361 (2018)","DOI":"10.1145\/3206025.3206046"},{"key":"2260_CR7","unstructured":"Li, B., Kumar, A.: Query by video: cross-modal music retrieval. In: ISMIR, pp. 604\u2013611 (2019)"},{"key":"2260_CR8","doi-asserted-by":"crossref","unstructured":"Sur\u00eds, D., Vondrick, C., Russell, B., Salamon, J.: It\u2019s time for artistic correspondence in music and video. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10564\u201310574 (2022)","DOI":"10.1109\/CVPR52688.2022.01031"},{"key":"2260_CR9","doi-asserted-by":"crossref","unstructured":"Girdhar, R., El-Nouby, A., Liu, Z., Singh, M., Alwala, K.V., Joulin, A., Misra, I.: Imagebind: One embedding space to bind them all. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15180\u201315190 (2023)","DOI":"10.1109\/CVPR52729.2023.01457"},{"key":"2260_CR10","doi-asserted-by":"crossref","unstructured":"Szegedy, C., Vanhoucke, V., Ioffe, S., Shlens, J., Wojna, Z.: Rethinking the inception architecture for computer vision. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2818\u20132826 (2016)","DOI":"10.1109\/CVPR.2016.308"},{"key":"2260_CR11","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763 (2021). PmLR"},{"key":"2260_CR12","doi-asserted-by":"crossref","unstructured":"Gemmeke, J.F., Ellis, D.P., Freedman, D., Jansen, A., Lawrence, W., Moore, R.C., Plakal, M., Ritter, M.: Audio set: An ontology and human-labeled dataset for audio events. In: 2017 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 776\u2013780 (2017). IEEE","DOI":"10.1109\/ICASSP.2017.7952261"},{"key":"2260_CR13","doi-asserted-by":"crossref","unstructured":"Lee, J., Bryan, N.J., Salamon, J., Jin, Z., Nam, J.: Disentangled multidimensional metric learning for music similarity. In: ICASSP 2020-2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6\u201310 (2020). IEEE","DOI":"10.1109\/ICASSP40776.2020.9053442"},{"issue":"3s","key":"2260_CR14","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3575658","volume":"19","author":"J Zhang","year":"2023","unstructured":"Zhang, J., Yu, Y., Tang, S., Wu, J., Li, W.: Variational autoencoder with cca for audio-visual cross-modal retrieval. ACM Trans. Multimed. Comput. Commun. Appl. 19(3s), 1\u201321 (2023)","journal-title":"ACM Trans. Multimed. Comput. Commun. Appl."},{"key":"2260_CR15","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, \u0141., Polosukhin, I.: Attention is all you need. Advances in neural information processing systems 30 (2017)"},{"key":"2260_CR16","doi-asserted-by":"crossref","unstructured":"Shang, L., Yue, Z.D., Karim, K.S., Shen, J., Wang, D.: Camr: Towards connotation-aware music retrieval on social media with visual inputs. In: 2020 IEEE\/ACM International Conference on Advances in Social Networks Analysis and Mining (ASONAM), pp. 425\u2013429 (2020). IEEE","DOI":"10.1109\/ASONAM49781.2020.9381371"},{"issue":"6","key":"2260_CR17","doi-asserted-by":"publisher","first-page":"3068","DOI":"10.3390\/app15063068","volume":"15","author":"H Chen","year":"2025","unstructured":"Chen, H., Zou, Z., Liu, Y., Zhu, X.: Deep class-guided hashing for multi-label cross-modal retrieval. Appl. Sci. 15(6), 3068 (2025)","journal-title":"Appl. Sci."},{"key":"2260_CR18","doi-asserted-by":"publisher","first-page":"7027","DOI":"10.1109\/TMM.2024.3358995","volume":"26","author":"G Song","year":"2024","unstructured":"Song, G., Huang, K., Su, H., Song, F., Yang, M.: Deep ranking distribution preserving hashing for robust multi-label cross-modal retrieval. IEEE Trans. Multimedia 26, 7027\u20137042 (2024)","journal-title":"IEEE Trans. Multimedia"},{"key":"2260_CR19","doi-asserted-by":"crossref","unstructured":"Zhao, S., Zhu, L., Wang, X., Yang, Y.: Centerclip: Token clustering for efficient text-video retrieval. In: Proceedings of the 45th International ACM SIGIR Conference on Research and Development in Information Retrieval, pp. 970\u2013981 (2022)","DOI":"10.1145\/3477495.3531950"},{"key":"2260_CR20","doi-asserted-by":"crossref","unstructured":"Tariq, U., Hu, Z., Tasneem, K.T., Heyat, M.B.B., Iqbal, M.S., Aziz, K.: Clustere-zsl: a novel cluster-based embeddings for enhanced zero-shot learning in contrastive pre-training cross-model retrieval. IEEE Access (2024)","DOI":"10.1109\/ACCESS.2024.3476082"},{"key":"2260_CR21","doi-asserted-by":"crossref","unstructured":"Hu, Z., Cheung, Y.-M., Li, M., Lan, W., Zhang, D.: Key points centered sparse hashing for cross-modal retrieval. In: ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 8431\u20138435 (2024). IEEE","DOI":"10.1109\/ICASSP48485.2024.10446586"},{"issue":"6","key":"2260_CR22","doi-asserted-by":"publisher","first-page":"3865","DOI":"10.1109\/TSMC.2024.3373612","volume":"54","author":"J Huang","year":"2024","unstructured":"Huang, J., Kang, P., Fang, X., Han, N., Xie, S., Gao, H.: Efficient discriminative hashing for cross-modal retrieval. IEEE Trans. Syst. Man Cybern.: Syst. 54(6), 3865\u20133878 (2024)","journal-title":"IEEE Trans. Syst. Man Cybern.: Syst."},{"key":"2260_CR23","doi-asserted-by":"crossref","unstructured":"Wang, H., Chen, Y., Yao, M., Liu, W., Peng, J., Fu, X.: Tensor completion framework by graph refinement for incomplete multi-view clustering. IEEE Trans. Multimedia (2025)","DOI":"10.1109\/TMM.2025.3613125"},{"key":"2260_CR24","doi-asserted-by":"crossref","unstructured":"Tan, J., Peng, J., Zhang, S., Wang, Z., Wang, H.: Unsupervised lifelong person re-identification via affinity harmonization. ACM Trans. Multimedia Comput. Commun. Appl. (2025)","DOI":"10.1145\/3779124"},{"key":"2260_CR25","doi-asserted-by":"crossref","unstructured":"Cai, B., Wang, H., Yao, M., Fu, X.: Focus more on what? guiding multi-task training for end-to-end person search. IEEE Trans. Circuits Syst. Video Technol. (2025)","DOI":"10.1109\/TCSVT.2025.3540089"},{"key":"2260_CR26","doi-asserted-by":"crossref","unstructured":"Zhou, L., Li, Y.: Coarse-to-fine alignment makes better speech-image retrieval. In: 2024 IEEE International Conference on Multimedia and Expo (ICME), pp. 1\u20136 (2024). IEEE","DOI":"10.1109\/ICME57554.2024.10687580"},{"issue":"1","key":"2260_CR27","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s44267-025-00073-2","volume":"3","author":"Y Xu","year":"2025","unstructured":"Xu, Y., Wu, M., Guo, Z., Cao, M., Ye, M., Laaksonen, J.: Efficient text-to-video retrieval via multi-modal multi-tagger derived pre-screening. Visual Intell. 3(1), 1\u201313 (2025)","journal-title":"Visual Intell."},{"key":"2260_CR28","doi-asserted-by":"crossref","unstructured":"Nagrani, A., Seo, P.H., Seybold, B., Hauth, A., Manen, S., Sun, C., Schmid, C.: Learning audio-video modalities from image captions. In: European Conference on Computer Vision, pp. 407\u2013426 (2022). Springer","DOI":"10.1007\/978-3-031-19781-9_24"},{"key":"2260_CR29","doi-asserted-by":"crossref","unstructured":"Guzhov, A., Raue, F., Hees, J., Dengel, A.: Audioclip: Extending clip to image, text and audio. In: ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 976\u2013980 (2022). IEEE","DOI":"10.1109\/ICASSP43922.2022.9747631"},{"key":"2260_CR30","doi-asserted-by":"crossref","unstructured":"Zhang, R., Guo, Z., Zhang, W., Li, K., Miao, X., Cui, B., Qiao, Y., Gao, P., Li, H.: Pointclip: point cloud understanding by clip. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8552\u20138562 (2022)","DOI":"10.1109\/CVPR52688.2022.00836"},{"key":"2260_CR31","unstructured":"Wu, S., Fei, H., Qu, L., Ji, W., Chua, T.-S.: Next-gpt: Any-to-any multimodal llm. In: Forty-First International Conference on Machine Learning (2024)"},{"key":"2260_CR32","doi-asserted-by":"crossref","unstructured":"Lei, W., Ge, Y., Yi, K., Zhang, J., Gao, D., Sun, D., Ge, Y., Shan, Y., Shou, M.Z.: Vit-lens: towards omni-modal representations. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 26647\u201326657 (2024)","DOI":"10.1109\/CVPR52733.2024.02516"},{"key":"2260_CR33","doi-asserted-by":"crossref","unstructured":"Lyu, Y., Zheng, X., Zhou, J., Wang, L.: Unibind: Llm-augmented unified and balanced representation space to bind them all. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 26752\u201326762 (2024)","DOI":"10.1109\/CVPR52733.2024.02526"},{"key":"2260_CR34","doi-asserted-by":"crossref","unstructured":"Zhou, B., Li, L., Wang, Y., Liu, H., Yao, Y., Wang, W.: Unialign: Scaling multimodal alignment within one unified model. In: Proceedings of the Computer Vision and Pattern Recognition Conference, pp. 29644\u201329655 (2025)","DOI":"10.1109\/CVPR52734.2025.02760"},{"key":"2260_CR35","unstructured":"Su, Y., Lan, T., Li, H., Xu, J., Wang, Y., Cai, D.: Pandagpt: One model to instruction-follow them all. arXiv preprint arXiv:2305.16355 (2023)"},{"key":"2260_CR36","unstructured":"Abu-El-Haija, S., Kothari, N., Lee, J., Natsev, P., Toderici, G., Varadarajan, B., Vijayanarasimhan, S.: Youtube-8m: A large-scale video classification benchmark. arXiv preprint arXiv:1609.08675 (2016)"},{"key":"2260_CR37","doi-asserted-by":"crossref","unstructured":"McKee, D., Salamon, J., Sivic, J., Russell, B.: Language-guided music recommendation for video via prompt analogies. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14784\u201314793 (2023)","DOI":"10.1109\/CVPR52729.2023.01420"},{"key":"2260_CR38","unstructured":"Chen, Z., Zhang, P., Ye, K., Dong, W., Feng, X., Zhang, Y.: Start from video-music retrieval: an inter\u2013intra modal loss for cross modal retrieval. arXiv preprint arXiv:2407.19415 (2024)"},{"key":"2260_CR39","doi-asserted-by":"crossref","unstructured":"Wu, H.-H., Seetharaman, P., Kumar, K., Bello, J.P.: Wav2clip: learning robust audio representations from clip. In: ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 4563\u20134567 (2022). IEEE","DOI":"10.1109\/ICASSP43922.2022.9747669"},{"key":"2260_CR40","unstructured":"Cheng, Z., Leng, S., Zhang, H., Xin, Y., Li, X., Chen, G., Zhu, Y., Zhang, W., Luo, Z., Zhao, D., et al.: Videollama 2: advancing spatial-temporal modeling and audio understanding in video-llms. arXiv preprint arXiv:2406.07476 (2024)"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-026-02260-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-026-02260-7","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-026-02260-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,6]],"date-time":"2026-07-06T05:06:44Z","timestamp":1783314404000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-026-02260-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,10]]},"references-count":40,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2026,6]]}},"alternative-id":["2260"],"URL":"https:\/\/doi.org\/10.1007\/s00530-026-02260-7","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,3,10]]},"assertion":[{"value":"2 October 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"31 January 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"10 March 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"183"}}