{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T16:45:56Z","timestamp":1777567556187,"version":"3.51.4"},"publisher-location":"Cham","reference-count":34,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031938054","type":"print"},{"value":"9783031938061","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-93806-1_16","type":"book-chapter","created":{"date-parts":[[2025,5,31]],"date-time":"2025-05-31T18:29:06Z","timestamp":1748716146000},"page":"214-226","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["CMMD: Contrastive Multi-modal Diffusion for\u00a0Video-Audio Conditional Modeling"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3154-4150","authenticated-orcid":false,"given":"Ruihan","family":"Yang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hannes","family":"Gamper","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sebastian","family":"Braun","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,5,20]]},"reference":[{"key":"16_CR1","first-page":"8780","volume":"34","author":"P Dhariwal","year":"2021","unstructured":"Dhariwal, P., Nichol, A.: Diffusion models beat gans on image synthesis. Adv. Neural. Inf. Process. Syst. 34, 8780\u20138794 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"16_CR2","doi-asserted-by":"publisher","unstructured":"Elizalde, B., Deshmukh, S., Ismail, M.A., Wang, H.: Clap learning audio concepts from natural language supervision. In: ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a01\u20135 (2023). https:\/\/doi.org\/10.1109\/ICASSP49357.2023.10095889","DOI":"10.1109\/ICASSP49357.2023.10095889"},{"key":"16_CR3","doi-asserted-by":"crossref","unstructured":"Gemmeke, J.F., et al.: Audio set: an ontology and human-labeled dataset for audio events. In: Proceedings of IEEE ICASSP 2017, New Orleans, LA (2017)","DOI":"10.1109\/ICASSP.2017.7952261"},{"key":"16_CR4","doi-asserted-by":"crossref","unstructured":"Gui, A., Gamper, H., Braun, S., Emmanouilidou, D.: Adapting frechet audio distance for generative music evaluation (2023)","DOI":"10.1109\/ICASSP48485.2024.10446663"},{"key":"16_CR5","doi-asserted-by":"crossref","unstructured":"Hang, T., et al.: Efficient diffusion training via min-snr weighting strategy. arXiv preprint arXiv:2303.09556 (2023)","DOI":"10.1109\/ICCV51070.2023.00684"},{"key":"16_CR6","first-page":"6840","volume":"33","author":"J Ho","year":"2020","unstructured":"Ho, J., Jain, A., Abbeel, P.: Denoising diffusion probabilistic models. Adv. Neural. Inf. Process. Syst. 33, 6840\u20136851 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"16_CR7","unstructured":"Ho, J., Salimans, T., Gritsenko, A., Chan, W., Norouzi, M., Fleet, D.J.: Video diffusion models. In: Koyejo, S., Mohamed, S., Agarwal, A., Belgrave, D., Cho, K., Oh, A. (eds.) Advances in Neural Information Processing Systems, vol.\u00a035, pp. 8633\u20138646. Curran Associates, Inc. (2022). https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2022\/file\/39235c56aef13fb05a6adc95eb9d8d66-Paper-Conference.pdf"},{"key":"16_CR8","doi-asserted-by":"crossref","unstructured":"Huh, J., Chalk, J., Kazakos, E., Damen, D., Zisserman, A.: EPIC-SOUNDS: a large-scale dataset of actions that sound. In: IEEE International Conference on Acoustics, Speech, & Signal Processing (ICASSP) (2023)","DOI":"10.1109\/ICASSP49357.2023.10096198"},{"key":"16_CR9","doi-asserted-by":"crossref","unstructured":"Jeong, Y., et al.: The power of sound (TPOS): audio reactive video generation with stable diffusion. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7822\u20137832 (2023)","DOI":"10.1109\/ICCV51070.2023.00719"},{"key":"16_CR10","first-page":"26565","volume":"35","author":"T Karras","year":"2022","unstructured":"Karras, T., Aittala, M., Aila, T., Laine, S.: Elucidating the design space of diffusion-based generative models. Adv. Neural. Inf. Process. Syst. 35, 26565\u201326577 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"16_CR11","first-page":"23593","volume":"35","author":"B Kawar","year":"2022","unstructured":"Kawar, B., Elad, M., Ermon, S., Song, J.: Denoising diffusion restoration models. Adv. Neural. Inf. Process. Syst. 35, 23593\u201323606 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"16_CR12","doi-asserted-by":"publisher","unstructured":"Kilgour, K., Zuluaga, M., Roblek, D., Sharifi, M.: Fr\u00e9chet Audio Distance: A Metric for Evaluating Music Enhancement Algorithms (2019). https:\/\/doi.org\/10.48550\/arXiv.1812.08466. http:\/\/arxiv.org\/abs\/1812.08466. arXiv:1812.08466","DOI":"10.48550\/arXiv.1812.08466"},{"key":"16_CR13","unstructured":"Kong, J., Kim, J., Bae, J.: HiFi-GAN: generative adversarial networks for efficient and high fidelity speech synthesis. In: Proceedings of 34th Conference on Neural Information Processing Systems (2020)"},{"key":"16_CR14","unstructured":"Lee, S.H., et al.: Soundini: sound-guided diffusion for natural video editing. arXiv preprint arXiv:2304.06818 (2023)"},{"key":"16_CR15","unstructured":"Li, R., Yang, S., Ross, D.A., Kanazawa, A.: Learn to dance with AIST++: music conditioned 3D dance generation (2021)"},{"key":"16_CR16","doi-asserted-by":"crossref","unstructured":"Liu, H., et al.: Audioldm 2: learning holistic audio generation with self-supervised pretraining (2023)","DOI":"10.1109\/TASLP.2024.3399607"},{"key":"16_CR17","first-page":"5775","volume":"35","author":"C Lu","year":"2022","unstructured":"Lu, C., Zhou, Y., Bao, F., Chen, J., Li, C., Zhu, J.: DPM-solver: a fast ode solver for diffusion probabilistic model sampling in around 10 steps. Adv. Neural. Inf. Process. Syst. 35, 5775\u20135787 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"16_CR18","unstructured":"Luo, S., Yan, C., Hu, C., Zhao, H.: Diff-foley: synchronized video-to-audio synthesis with latent diffusion models. arXiv preprint arXiv:2306.17203 (2023)"},{"key":"16_CR19","doi-asserted-by":"publisher","unstructured":"McFee, B., et al.: librosa\/librosa: 0.10.1 (2023). https:\/\/doi.org\/10.5281\/zenodo.8252662","DOI":"10.5281\/zenodo.8252662"},{"key":"16_CR20","unstructured":"Oord, A.V.D., Li, Y., Vinyals, O.: Representation learning with contrastive predictive coding. arXiv preprint arXiv:1807.03748 (2018)"},{"key":"16_CR21","unstructured":"Ramesh, A., Dhariwal, P., Nichol, A., Chu, C., Chen, M.: Hierarchical text-conditional image generation with clip latents. arXiv preprint arXiv:2204.06125 (2022)"},{"key":"16_CR22","unstructured":"Ramesh, A., et al.: Zero-shot text-to-image generation. In: International Conference on Machine Learning, pp. 8821\u20138831. PMLR (2021)"},{"key":"16_CR23","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., Ommer, B.: High-resolution image synthesis with latent diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10684\u201310695 (2022)","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"16_CR24","doi-asserted-by":"crossref","unstructured":"Ruan, L., et al.: Mm-diffusion: learning multi-modal diffusion models for joint audio and video generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10219\u201310228 (2023)","DOI":"10.1109\/CVPR52729.2023.00985"},{"key":"16_CR25","unstructured":"Salimans, T., Ho, J.: Progressive distillation for fast sampling of diffusion models. In: International Conference on Learning Representations (2022)"},{"key":"16_CR26","unstructured":"Song, J., Meng, C., Ermon, S.: Denoising diffusion implicit models. In: International Conference on Learning Representations (2021)"},{"key":"16_CR27","unstructured":"Song, Y., Sohl-Dickstein, J., Kingma, D.P., Kumar, A., Ermon, S., Poole, B.: Score-based generative modeling through stochastic differential equations. In: International Conference on Learning Representations (2021)"},{"key":"16_CR28","unstructured":"Tsuchida, S., Fukayama, S., Hamasaki, M., Goto, M.: AIST dance video database: multi-genre, multi-dancer, and multi-camera database for dance information processing. In: Proceedings of the 20th International Society for Music Information Retrieval Conference, ISMIR 2019, Delft, Netherlands (2019)"},{"key":"16_CR29","unstructured":"Unterthiner, T., van Steenkiste, S., Kurach, K., Marinier, R., Michalski, M., Gelly, S.: Towards accurate generative models of video: a new metric & challenges. arXiv preprint arXiv:1812.01717 (2018)"},{"key":"16_CR30","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Advances in Neural Information Processing Systems, vol. 30 (2017)"},{"key":"16_CR31","first-page":"23371","volume":"35","author":"V Voleti","year":"2022","unstructured":"Voleti, V., Jolicoeur-Martineau, A., Pal, C.: MCVD-masked conditional video diffusion for prediction, generation, and interpolation. Adv. Neural. Inf. Process. Syst. 35, 23371\u201323385 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"16_CR32","doi-asserted-by":"crossref","unstructured":"Yang, R., Srivastava, P., Mandt, S.: Diffusion probabilistic modeling for video generation. arXiv preprint arXiv:2203.09481 (2022)","DOI":"10.3390\/e25101469"},{"issue":"3","key":"16_CR33","doi-asserted-by":"publisher","first-page":"623","DOI":"10.1109\/TBC.2008.2002102","volume":"54","author":"AC Younkin","year":"2008","unstructured":"Younkin, A.C., Corriveau, P.J.: Determining the amount of audio-video synchronization errors perceptible to the average end-user. IEEE Trans. Broadcast. 54(3), 623\u2013627 (2008). https:\/\/doi.org\/10.1109\/TBC.2008.2002102","journal-title":"IEEE Trans. Broadcast."},{"key":"16_CR34","unstructured":"Zhu, Y., Wu, Y., Olszewski, K., Ren, J., Tulyakov, S., Yan, Y.: Discrete contrastive diffusion for cross-modal music and image generation. In: The Eleventh International Conference on Learning Representations (2023)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024 Workshops"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-93806-1_16","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,31]],"date-time":"2025-05-31T18:29:16Z","timestamp":1748716156000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-93806-1_16"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9783031938054","9783031938061"],"references-count":34,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-93806-1_16","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"20 May 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}