{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,2]],"date-time":"2026-01-02T03:25:07Z","timestamp":1767324307971,"version":"3.48.0"},"publisher-location":"Cham","reference-count":68,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032128393","type":"print"},{"value":"9783032128409","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-12840-9_8","type":"book-chapter","created":{"date-parts":[[2026,1,2]],"date-time":"2026-01-02T03:22:44Z","timestamp":1767324164000},"page":"106-122","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Video Object Segmentation-Aware Audio Generation"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-2361-8067","authenticated-orcid":false,"given":"Ilpo","family":"Viertola","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8879-587X","authenticated-orcid":false,"given":"Vladimir","family":"Iashin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8767-0864","authenticated-orcid":false,"given":"Esa","family":"Rahtu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,1,2]]},"reference":[{"key":"8_CR1","unstructured":"Alayrac, J.B., et al.: Flamingo: a visual language model for few-shot learning. In: Advances in Neural Information Processing Systems (NeurIPS), vol. 35, pp. 23716\u201323736 (2022)"},{"key":"8_CR2","doi-asserted-by":"crossref","unstructured":"Chang, H., Zhang, H., Jiang, L., Liu, C., Freeman, W.T.: Maskgit: masked generative image transformer. In: Proceedings of the Computer Vision and Pattern Recognition Conference (CVPR), pp. 11315\u201311325 (2022)","DOI":"10.1109\/CVPR52688.2022.01103"},{"key":"8_CR3","doi-asserted-by":"crossref","unstructured":"Chen, C., et al.: Action2sound: ambient-aware generation of action sounds from egocentric videos. In: European Conference on Computer Vision (ECCV), pp. 277\u2013295 (2024)","DOI":"10.1007\/978-3-031-72897-6_16"},{"key":"8_CR4","doi-asserted-by":"crossref","unstructured":"Chen, H., Xie, W., Vedaldi, A., Zisserman, A.: Vggsound: a large-scale audio-visual dataset. In: Proceedings of International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 721\u2013725 (2020)","DOI":"10.1109\/ICASSP40776.2020.9053174"},{"key":"8_CR5","doi-asserted-by":"crossref","unstructured":"Chen, Z., et al.: Video-guided foley sound generation with multimodal controls. In: Proceedings of the Computer Vision and Pattern Recognition Conference (CVPR), pp. 18770\u201318781 (2025)","DOI":"10.1109\/CVPR52734.2025.01749"},{"key":"8_CR6","doi-asserted-by":"crossref","unstructured":"Cheng, H.K., Ishii, M., Hayakawa, A., Shibuya, T., Schwing, A., Mitsufuji, Y.: Mmaudio: taming multimodal joint training for high-quality video-to-audio synthesis. In: Proceedings of the Computer Vision and Pattern Recognition Conference (CVPR), pp. 28901\u201328911 (2025)","DOI":"10.1109\/CVPR52734.2025.02691"},{"key":"8_CR7","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.J., Li, K., Fei-Fei, L.: Imagenet: a large-scale hierarchical image database. In: Proceedings of the Computer Vision and Pattern Recognition Conference (CVPR), pp. 248\u2013255 (2009)","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"8_CR8","unstructured":"Dosovitskiy, A., et al.: An image is worth 16x16 words: transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)"},{"key":"8_CR9","doi-asserted-by":"crossref","unstructured":"Fragkiadaki, K., Arbelaez, P., Felsen, P., Malik, J.: Learning to segment moving objects in videos. In: Proceedings of the International Conference on Computer Vision (CVPR), pp. 4083\u20134090 (2015)","DOI":"10.1109\/CVPR.2015.7299035"},{"key":"8_CR10","doi-asserted-by":"crossref","unstructured":"Gemmeke, J.F., et al.: Audio set: an ontology and human-labeled dataset for audio events. In: Proceedings of International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 776\u2013780 (2017)","DOI":"10.1109\/ICASSP.2017.7952261"},{"key":"8_CR11","doi-asserted-by":"crossref","unstructured":"Girdhar, R., et al.: Imagebind: one embedding space to bind them all. In: Proceedings of the Computer Vision and Pattern Recognition Conference (CVPR), pp. 15180\u201315190 (2023)","DOI":"10.1109\/CVPR52729.2023.01457"},{"key":"8_CR12","doi-asserted-by":"crossref","unstructured":"Gong, Y., Chung, Y.A., Glass, J.: AST: audio spectrogram transformer. arXiv preprint arXiv:2104.01778 (2021)","DOI":"10.21437\/Interspeech.2021-698"},{"key":"8_CR13","unstructured":"Ho, J., Salimans, T.: Classifier-free diffusion guidance. arXiv preprint arXiv:2207.12598 (2022)"},{"key":"8_CR14","doi-asserted-by":"crossref","unstructured":"Hou, S., et al.: Editing music with melody and text: using controlnet for diffusion transformer. In: Proceedings of International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a01\u20135 (2025)","DOI":"10.1109\/ICASSP49660.2025.10890309"},{"key":"8_CR15","unstructured":"Hu, E.J., et al.: Lora: low-rank adaptation of large language models. arxiv 2021. arXiv preprint arXiv:2106.09685 (2021)"},{"key":"8_CR16","unstructured":"Iashin, V., Xie, W., Rahtu, E., Zisserman, A.: Sparse in space and time: audio-visual synchronisation with trainable selectors. In: British Machine Vision Conference (BMVC) (2022)"},{"key":"8_CR17","doi-asserted-by":"crossref","unstructured":"Iashin, V., Rahtu, E.: Taming visually guided sound generation. arXiv preprint arXiv:2110.08791 (2021)","DOI":"10.5244\/C.35.336"},{"key":"8_CR18","doi-asserted-by":"crossref","unstructured":"Iashin, V., Xie, W., Rahtu, E., Zisserman, A.: Synchformer: efficient synchronization from sparse cues. In: Proceedings of International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 5325\u20135329 (2024)","DOI":"10.1109\/ICASSP48485.2024.10448489"},{"key":"8_CR19","doi-asserted-by":"crossref","unstructured":"Jeong, Y., Kim, Y., Chun, S., Lee, J.: Read, watch and scream! sound generation from text and video. In: AAAI Conference on Artificial Intelligence (AAAI), pp. 17590\u201317598 (2025)","DOI":"10.1609\/aaai.v39i17.33934"},{"key":"8_CR20","unstructured":"Jocher, G., et al.: ultralytics\/yolov5: v7. 0-yolov5 sota realtime instance segmentation. Zenodo (2022)"},{"key":"8_CR21","unstructured":"Kim, G., et al.: A versatile diffusion transformer with mixture of noise levels for audiovisual generation. arXiv preprint arXiv:2405.13762 (2024)"},{"key":"8_CR22","unstructured":"Kirillov, A., et al.: Segment anything. arXiv preprint arXiv:2304.02643 (2023)"},{"key":"8_CR23","doi-asserted-by":"publisher","first-page":"2880","DOI":"10.1109\/TASLP.2020.3030497","volume":"28","author":"Q Kong","year":"2020","unstructured":"Kong, Q., Cao, Y., Iqbal, T., Wang, Y., Wang, W., Plumbley, M.D.: PANNs: large-scale pretrained audio neural networks for audio pattern recognition. Trans. Audio Speech Lang. Process. 28, 2880\u20132894 (2020)","journal-title":"Trans. Audio Speech Lang. Process."},{"key":"8_CR24","doi-asserted-by":"crossref","unstructured":"Koutini, K., Schl\u00fcter, J., Eghbal-Zadeh, H., Widmer, G.: Efficient training of audio transformers with patchout. arXiv preprint arXiv:2110.05069 (2021)","DOI":"10.21437\/Interspeech.2022-227"},{"issue":"2","key":"8_CR25","doi-asserted-by":"publisher","first-page":"522","DOI":"10.1109\/TMM.2018.2856090","volume":"21","author":"B Li","year":"2018","unstructured":"Li, B., Liu, X., Dinesh, K., Duan, Z., Sharma, G.: Creating a multitrack classical music performance dataset for multimodal music analysis: challenges, insights, and applications. Trans. Multimedia 21(2), 522\u2013535 (2018)","journal-title":"Trans. Multimedia"},{"key":"8_CR26","unstructured":"Li, J., Li, D., Xiong, C., Hoi, S.: Blip: bootstrapping language-image pre-training for unified vision-language understanding and generation. In: International Conference on Machine Learning (ICML), pp. 12888\u201312900 (2022)"},{"key":"8_CR27","unstructured":"Lian, L., et al.: Describe anything: detailed localized image and video captioning. arXiv preprint arXiv:2504.16072 (2025)"},{"key":"8_CR28","unstructured":"Lipman, Y., Chen, R.T., Ben-Hamu, H., Nickel, M., Le, M.: Flow matching for generative modeling. arXiv preprint arXiv:2210.02747 (2022)"},{"key":"8_CR29","unstructured":"Liu, H., et al.: Audioldm: text-to-audio generation with latent diffusion models. arXiv preprint arXiv:2301.12503 (2023)"},{"key":"8_CR30","doi-asserted-by":"crossref","unstructured":"Liu, S., et al.: Grounding dino: marrying dino with grounded pre-training for open-set object detection. In: European Conference on Computer Vision (ECCV), pp. 38\u201355 (2024)","DOI":"10.1007\/978-3-031-72970-6_3"},{"key":"8_CR31","unstructured":"Liu, X., Gong, C., Liu, Q.: Flow straight and fast: learning to generate and transfer data with rectified flow. arXiv preprint arXiv:2209.03003 (2022)"},{"key":"8_CR32","doi-asserted-by":"crossref","unstructured":"Liu, X., Su, K., Shlizerman, E.: Tell what you hear from what you see\u2013video to audio generation through text. arXiv preprint arXiv:2411.05679 (2024)","DOI":"10.52202\/079017-3213"},{"key":"8_CR33","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101 (2017)"},{"key":"8_CR34","unstructured":"Luo, S., Yan, C., Hu, C., Zhao, H.: Diff-foley: synchronized video-to-audio synthesis with latent diffusion models. In: Advances in Neural Information Processing Systems (NeurIPS), pp. 48855\u201348876 (2023)"},{"key":"8_CR35","doi-asserted-by":"crossref","unstructured":"Mei, X., et al.: Foleygen: visually-guided audio generation. In: International Workshop on Machine Learning for Signal Processing (MLSP), pp.\u00a01\u20136 (2024)","DOI":"10.1109\/MLSP58920.2024.10734721"},{"key":"8_CR36","unstructured":"Mo, S., Shi, J., Tian, Y.: Text-to-audio generation synchronized with videos. arXiv preprint arXiv:2403.07938 (2024)"},{"key":"8_CR37","doi-asserted-by":"crossref","unstructured":"Montesinos, J.F., Slizovskaia, O., Haro, G.: Solos: a dataset for audio-visual music analysis. In: International Workshop on Multimedia Signal Processing (MMSP), pp.\u00a01\u20136 (2020)","DOI":"10.1109\/MMSP48831.2020.9287124"},{"key":"8_CR38","doi-asserted-by":"crossref","unstructured":"Pascual, S., Yeh, C., Tsiamas, I., Serr\u00e0, J.: Masked generative video-to-audio transformers with enhanced synchronicity. In: European Conference on Computer Vision (ECCV), pp. 247\u2013264 (2024)","DOI":"10.1007\/978-3-031-73021-4_15"},{"key":"8_CR39","doi-asserted-by":"crossref","unstructured":"Peebles, W., Xie, S.: Scalable diffusion models with transformers. In: Proceedings of the Computer Vision and Pattern Recognition Conference (CVPR), pp. 4195\u20134205 (2023)","DOI":"10.1109\/ICCV51070.2023.00387"},{"key":"8_CR40","unstructured":"Polyak, A., et al.: Movie gen: a cast of media foundation models. arXiv preprint arXiv:2410.13720 (2024)"},{"key":"8_CR41","unstructured":"Pont-Tuset, J., Perazzi, F., Caelles, S., Arbel\u00e1ez, P., Sorkine-Hornung, A., Van\u00a0Gool, L.: The 2017 davis challenge on video object segmentation. arXiv preprint arXiv:1704.00675 (2017)"},{"key":"8_CR42","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning (ICML), pp. 8748\u20138763 (2021)"},{"key":"8_CR43","unstructured":"Ravi, N., et al.: Sam 2: segment anything in images and videos. arXiv preprint arXiv:2408.00714 (2024)"},{"key":"8_CR44","unstructured":"Ren, T., et al.: Grounding dino 1.5: advance the \u201cedge\u201d of open-set object detection. arXiv preprint arXiv:2405.10300 (2024)"},{"key":"8_CR45","unstructured":"Ren, T., et al.: Grounded SAM: assembling open-world models for diverse visual tasks. arXiv preprint arXiv:2401.14159 (2024)"},{"key":"8_CR46","unstructured":"Ren, T., et al.: Grounded SAM: assembling open-world models for diverse visual tasks. arXiv preprint arXiv:2401.14159 (2024)"},{"key":"8_CR47","doi-asserted-by":"crossref","unstructured":"Ruan, L., et al.: Mm-diffusion: learning multi-modal diffusion models for joint audio and video generation. In: Proceedings of the Computer Vision and Pattern Recognition Conference (CVPR), pp. 10219\u201310228 (2023)","DOI":"10.1109\/CVPR52729.2023.00985"},{"key":"8_CR48","unstructured":"Salimans, T., Goodfellow, I., Zaremba, W., Cheung, V., Radford, A., Chen, X.: Improved techniques for training GANs. In: Advances in Neural Information Processing Systems (NeurIPS), vol. 29 (2016)"},{"key":"8_CR49","doi-asserted-by":"crossref","unstructured":"Sheffer, R., Adi, Y.: I hear your true colors: image guided audio generation. In: Proceedings of International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a01\u20135 (2023)","DOI":"10.1109\/ICASSP49357.2023.10096023"},{"key":"8_CR50","unstructured":"Song, K., Tan, X., Qin, T., Lu, J., Liu, T.Y.: Mpnet: masked and permuted pre-training for language understanding. In: Advances in Neural \u0131nformation Processing Systems (NeurIPS), vol. 33, pp. 16857\u201316867 (2020)"},{"key":"8_CR51","unstructured":"Su, K., Liu, X., Shlizerman, E.: From vision to audio and beyond: a unified model for audio-visual representation and generation. arXiv preprint arXiv:2409.19132 (2024)"},{"key":"8_CR52","doi-asserted-by":"crossref","unstructured":"Tang, Z., Yang, Z., Khademi, M., Liu, Y., Zhu, C., Bansal, M.: Codi-2: in-context interleaved and interactive any-to-any generation. In: Proceedings of the Computer Vision and Pattern Recognition Conference (CVPR), pp. 27425\u201327434 (2024)","DOI":"10.1109\/CVPR52733.2024.02589"},{"key":"8_CR53","unstructured":"Tang, Z., Yang, Z., Zhu, C., Zeng, M., Bansal, M.: Any-to-any generation via composable diffusion. In: Advances in Neural Information Processing Systems (NeurIPS), pp. 16083\u201316099 (2023)"},{"key":"8_CR54","unstructured":"Tong, A., et al.: Improving and generalizing flow-based generative models with minibatch optimal transport. arXiv preprint arXiv:2302.00482 (2023)"},{"key":"8_CR55","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Advances in Neural Information Processing Systems (NeurIPS) (2017)"},{"key":"8_CR56","doi-asserted-by":"crossref","unstructured":"Viertola, I., Iashin, V., Rahtu, E.: Temporally aligned audio for video with autoregression. In: Proceedings of International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a01\u20135 (2025)","DOI":"10.1109\/ICASSP49660.2025.10890587"},{"key":"8_CR57","doi-asserted-by":"crossref","unstructured":"Wang, H., Ma, J., Pascual, S., Cartwright, R., Cai, W.: V2a-mapper: a lightweight solution for vision-to-audio generation by connecting foundation models. In: Proceedings of the AAAI Conference on Artificial Intelligence (AAAI), vol.\u00a038, pp. 15492\u201315501 (2024)","DOI":"10.1609\/aaai.v38i14.29475"},{"key":"8_CR58","doi-asserted-by":"crossref","unstructured":"Wang, Y., et al.: Frieren: efficient video-to-audio generation network with rectified flow matching. In: Advances in Neural Information Processing Systems (NeurIPS), vol. 37, pp. 128118\u2013128138 (2024)","DOI":"10.52202\/079017-4068"},{"key":"8_CR59","doi-asserted-by":"publisher","first-page":"2692","DOI":"10.1109\/TASLP.2024.3399026","volume":"32","author":"SL Wu","year":"2024","unstructured":"Wu, S.L., Donahue, C., Watanabe, S., Bryan, N.J.: Music controlnet: multiple time-varying controls for music generation. Trans. Audio Speech Lang. Process. 32, 2692\u20132703 (2024)","journal-title":"Trans. Audio Speech Lang. Process."},{"key":"8_CR60","doi-asserted-by":"crossref","unstructured":"Xiao, B., et al.: Florence-2: advancing a unified representation for a variety of vision tasks. arXiv preprint arXiv:2311.06242 (2023)","DOI":"10.1109\/CVPR52733.2024.00461"},{"key":"8_CR61","doi-asserted-by":"crossref","unstructured":"Xing, Y., He, Y., Tian, Z., Wang, X., Chen, Q.: Seeing and hearing: open-domain visual-audio generation with diffusion latent aligners. In: Proceedings of the Conference on Computer Vision and Pattern Recognition (CVPR), pp. 7151\u20137161 (2024)","DOI":"10.1109\/CVPR52733.2024.00683"},{"key":"8_CR62","unstructured":"Xu, M., et al.: Video-to-audio generation with hidden alignment. arXiv preprint arXiv:2407.07464 (2024)"},{"key":"8_CR63","doi-asserted-by":"crossref","unstructured":"Zhang, L., Rao, A., Agrawala, M.: Adding conditional control to text-to-image diffusion models. In: Proceedings of the Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3836\u20133847 (2023)","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"8_CR64","doi-asserted-by":"crossref","unstructured":"Zhao, H., Gan, C., Ma, W.C., Torralba, A.: The sound of motions. In: Proceedings of the International Conference on Computer Vision (CVPR), pp. 1735\u20131744 (2019)","DOI":"10.1109\/ICCV.2019.00182"},{"key":"8_CR65","doi-asserted-by":"crossref","unstructured":"Zhao, H., Gan, C., Rouditchenko, A., Vondrick, C., McDermott, J., Torralba, A.: The sound of pixels. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 570\u2013586 (2018)","DOI":"10.1007\/978-3-030-01246-5_35"},{"key":"8_CR66","unstructured":"Zhou, J., Shen, X.: Audio-visual segmentation with semantics. arXiv preprint arXiv:2301.13190 (2023)"},{"key":"8_CR67","doi-asserted-by":"crossref","unstructured":"Zhou, J., et al.: Audio\u2013visual segmentation. In: European Conference on Computer Vision (ECCV), pp. 386\u2013403 (2022)","DOI":"10.1007\/978-3-031-19836-6_22"},{"key":"8_CR68","unstructured":"Ziv, A., et al.: Masked audio generation using a single non-autoregressive transformer. arXiv preprint arXiv:2401.04577 (2024)"}],"container-title":["Lecture Notes in Computer Science","Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-12840-9_8","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,2]],"date-time":"2026-01-02T03:22:50Z","timestamp":1767324170000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-12840-9_8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9783032128393","9783032128409"],"references-count":68,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-12840-9_8","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"2 January 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"DAGM GCPR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"DAGM German Conference on Pattern Recognition","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Freiburg","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Germany","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 September 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 September 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"47","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"dagm2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.dagm-gcpr.de\/year\/2025","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}