{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,23]],"date-time":"2026-06-23T18:38:10Z","timestamp":1782239890041,"version":"3.54.5"},"publisher-location":"Cham","reference-count":51,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031728969","type":"print"},{"value":"9783031728976","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,12,2]],"date-time":"2024-12-02T00:00:00Z","timestamp":1733097600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,12,2]],"date-time":"2024-12-02T00:00:00Z","timestamp":1733097600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72897-6_16","type":"book-chapter","created":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T21:36:18Z","timestamp":1733088978000},"page":"277-295","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":14,"title":["Action2Sound: Ambient-Aware Generation of\u00a0Action Sounds from\u00a0Egocentric Videos"],"prefix":"10.1007","author":[{"given":"Changan","family":"Chen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Puyuan","family":"Peng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ami","family":"Baid","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zihui","family":"Xue","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wei-Ning","family":"Hsu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"David","family":"Harwath","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kristen","family":"Grauman","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,12,2]]},"reference":[{"key":"16_CR1","unstructured":"Blattmann, A., Rombach, R., Oktay, K., Ommer, B.: Retrieval-augmented diffusion models. ArXiv arxiv:2204.11824 (2022). https:\/\/api.semanticscholar.org\/CorpusID:248377386"},{"key":"16_CR2","unstructured":"Borgeaud, S., et al.: Improving language models by retrieving from trillions of tokens. In: International Conference on Machine Learning (2021). https:\/\/api.semanticscholar.org\/CorpusID:244954723"},{"key":"16_CR3","doi-asserted-by":"crossref","unstructured":"Chen, C., Ashutosh, K., Girdhar, R., Harwath, D., Grauman, K.: Soundingactions: learning how actions sound from narrated egocentric videos. In: CVPR (2024)","DOI":"10.1109\/CVPR52733.2024.02573"},{"key":"16_CR4","unstructured":"Chen, C., et al.: Soundspaces 2.0: a simulation platform for visual-acoustic learning. In: NeurIPS (2023)"},{"key":"16_CR5","doi-asserted-by":"crossref","unstructured":"Chen, H., Xie, W., Vedaldi, A., Zisserman, A.: Vggsound: a large-scale audio-visual dataset. In: ICASSP (2020)","DOI":"10.1109\/ICASSP40776.2020.9053174"},{"key":"16_CR6","first-page":"8292","volume":"29","author":"P Chen","year":"2020","unstructured":"Chen, P., Zhang, Y., Tan, M., Xiao, H., Huang, D., Gan, C.: Generating visually aligned sound from videos. TIP 29, 8292\u20138302 (2020)","journal-title":"TIP"},{"key":"16_CR7","unstructured":"Chen, W., Hu, H., Saharia, C., Cohen, W.W.: Re-imagen: retrieval-augmented text-to-image generator. ArXiv arxiv:2209.14491 (2022). https:\/\/api.semanticscholar.org\/CorpusID:252596087"},{"key":"16_CR8","doi-asserted-by":"crossref","unstructured":"Clarke, S., et al.: Realimpact: a dataset of impact sound fields for real objects. In: Proceedings of the IEEE International Conference on Computer Vision and Pattern Recognition (2023)","DOI":"10.1109\/CVPR52729.2023.00152"},{"key":"16_CR9","unstructured":"Clarke, S., et al.: Diffimpact: differentiable rendering and identification of impact sounds. In: 5th Annual Conference on Robot Learning (2021)"},{"key":"16_CR10","doi-asserted-by":"crossref","unstructured":"Damen, D., et al.: Scaling egocentric vision: the epic-kitchens dataset. In: ECCV (2018)","DOI":"10.1007\/978-3-030-01225-0_44"},{"key":"16_CR11","unstructured":"Dhariwal, P., Nichol, A.: Diffusion models beat gans on image synthesis. ArXiv arxiv:2105.05233 (2021). https:\/\/api.semanticscholar.org\/CorpusID:234357997"},{"key":"16_CR12","doi-asserted-by":"crossref","unstructured":"Du, Y., Chen, Z., Salamon, J., Russell, B., Owens, A.: Conditional generation of audio from video via foley analogies. In: 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 2426\u20132436 (2023)","DOI":"10.1109\/CVPR52729.2023.00240"},{"key":"16_CR13","doi-asserted-by":"crossref","unstructured":"Heilbron, F.C., Victor\u00a0Escorcia, B.G., Niebles, J.C.: Activitynet: a large-scale video benchmark for human activity understanding. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 961\u2013970 (2015)","DOI":"10.1109\/CVPR.2015.7298698"},{"key":"16_CR14","unstructured":"Gan, C., et al.: Threedworld: a platform for interactive multi-modal physical simulation. In: NeurIPS Datasets and Benchmarks Track (2021)"},{"key":"16_CR15","unstructured":"Gandhi, D., Gupta, A., Pinto, L.: Swoosh! rattle! thump! - actions that sound. In: RSS (2022)"},{"key":"16_CR16","doi-asserted-by":"crossref","unstructured":"Gemmeke, J.F., et al.: Audio set: an ontology and human-labeled dataset for audio events. In: 2017 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 776\u2013780 (2017)","DOI":"10.1109\/ICASSP.2017.7952261"},{"key":"16_CR17","doi-asserted-by":"crossref","unstructured":"Girdhar, R., El-et al.: Imagebind: one embedding space to bind them all. In: CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.01457"},{"key":"16_CR18","unstructured":"Grauman, K., et al.: Ego4d: around the world in 3,000 hours of egocentric video. In: 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 18973\u201318990 (2022)"},{"key":"16_CR19","unstructured":"Grauman, K., et al.: Ego-exo4d: understanding skilled human activity from first- and third-person perspectives. In: CVPR (2024)"},{"key":"16_CR20","unstructured":"Guu, K., Lee, K., Tung, Z., Pasupat, P., Chang, M.W.: Realm: retrieval-augmented language model pre-training. ArXiv arxiv:2002.08909 (2020). https:\/\/api.semanticscholar.org\/CorpusID:211204736"},{"key":"16_CR21","unstructured":"Ho, J., Jain, A., Abbeel, P.: Denoising diffusion probabilistic models. In: NeurIPS (2020)"},{"key":"16_CR22","unstructured":"Ho, J., Salimans, T.: Classifier-free diffusion guidance (2022)"},{"key":"16_CR23","doi-asserted-by":"crossref","unstructured":"Huang, C., Tian, Y., Kumar, A., Xu, C.: Egocentric audio-visual object localization. In: CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.02194"},{"key":"16_CR24","unstructured":"Huang, R., et al.: Make-an-audio: text-to-audio generation with prompt-enhanced diffusion models. ArXiv arxiv:2301.12661 (2023). https:\/\/api.semanticscholar.org\/CorpusID:256390046"},{"key":"16_CR25","doi-asserted-by":"crossref","unstructured":"Huh, J., Chalk, J., Kazakos, E., Damen, D., Zisserman, A.: Epic-sounds: a large-scale dataset of actions that sound. In: ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a01\u20135 (2023)","DOI":"10.1109\/ICASSP49357.2023.10096198"},{"key":"16_CR26","unstructured":"Iashin, V., Rahtu, E.: Taming visually guided sound generation. In: BMVC (2021)"},{"key":"16_CR27","doi-asserted-by":"crossref","unstructured":"Jiang, H., Murdock, C., Ithapu, V.K.: Egocentric deep multi-channel audio-visual active speaker localization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10544\u201310552 (2022)","DOI":"10.1109\/CVPR52688.2022.01029"},{"key":"16_CR28","unstructured":"Kay, W., et al.: The kinetics human action video dataset. CoRR arxiv:1705.06950 (2017). http:\/\/arxiv.org\/abs\/1705.06950"},{"key":"16_CR29","doi-asserted-by":"crossref","unstructured":"Kazakos, E., Nagrani, A., Zisserman, A., Damen, D.: Epic-fusion: audio-visual temporal binding for egocentric action recognition. In: 2019 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 5491\u20135500 (2019)","DOI":"10.1109\/ICCV.2019.00559"},{"key":"16_CR30","unstructured":"Khandelwal, U., Levy, O., Jurafsky, D., Zettlemoyer, L., Lewis, M.: Generalization through memorization: nearest neighbor language models. ArXiv arxiv:1911.00172 (2019). https:\/\/api.semanticscholar.org\/CorpusID:207870430"},{"key":"16_CR31","doi-asserted-by":"crossref","unstructured":"Kilgour, K., Zuluaga, M., Roblek, D., Sharifi, M.: Fr\u00e9chet audio distance: a metric for evaluating music enhancement algorithms. arxiv (2018)","DOI":"10.21437\/Interspeech.2019-2219"},{"key":"16_CR32","first-page":"17022","volume":"33","author":"J Kong","year":"2020","unstructured":"Kong, J., Kim, J., Bae, J.: Hifi-gan: generative adversarial networks for efficient and high fidelity speech synthesis. Adv. Neural. Inf. Process. Syst. 33, 17022\u201317033 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"16_CR33","unstructured":"Kong, Z., Ping, W., Huang, J., Zhao, K., Catanzaro, B.: Diffwave: a versatile diffusion model for audio synthesis. ArXiv arxiv:2009.09761 (2020). https:\/\/api.semanticscholar.org\/CorpusID:221818900"},{"key":"16_CR34","unstructured":"Lewis, P., et al.: Retrieval-augmented generation for knowledge-intensive nlp tasks. ArXiv arxiv:2005.11401 (2020). https:\/\/api.semanticscholar.org\/CorpusID:218869575"},{"key":"16_CR35","unstructured":"Lin, K.Q., et al.: Egocentric video-language pretraining. Adv. Neural Inf. Process. Syst. (2022)"},{"key":"16_CR36","unstructured":"Liu, H., et al.: Audioldm: text-to-audio generation with latent diffusion models. In: International Conference on Machine Learning (2023). https:\/\/api.semanticscholar.org\/CorpusID:256390486"},{"key":"16_CR37","unstructured":"Lu, C., Zhou, Y., Bao, F., Chen, J., Li, C., Zhu, J.: Dpm-solver: a fast ode solver for diffusion probabilistic model sampling in around 10 steps. arXiv preprint arXiv:2206.00927 (2022)"},{"key":"16_CR38","unstructured":"Luo, S., Yan, C., Hu, C., Zhao, H.: Diff-foley: synchronized video-to-audio synthesis with latent diffusion models. In: NeurIPS (2023)"},{"key":"16_CR39","doi-asserted-by":"crossref","unstructured":"Majumder, S., Al-Halah, Z., Grauman, K.: Learning spatial features from audio-visual correspondence in egocentric videos. In: CVPR (2024)","DOI":"10.1109\/CVPR52733.2024.02555"},{"key":"16_CR40","unstructured":"Mittal, H., Morgado, P., Jain, U., Gupta, A.: Learning state-aware visual representations from audible interactions. In: Oh, A.H., Agarwal, A., Belgrave, D., Cho, K. (eds.) Advances in Neural Information Processing Systems (2022). https:\/\/openreview.net\/forum?id=AhbTKBlM7X"},{"key":"16_CR41","unstructured":"Nichol, A., et al.: Glide: towards photorealistic image generation and editing with text-guided diffusion models. In: International Conference on Machine Learning (2021). https:\/\/api.semanticscholar.org\/CorpusID:245335086"},{"key":"16_CR42","doi-asserted-by":"crossref","unstructured":"Owens, A., Isola, P., McDermott, J., Torralba, A., Adelson, E.H., Freeman, W.T.: Visually indicated sounds. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.264"},{"key":"16_CR43","unstructured":"Popov, V., Vovk, I., Gogoryan, V., Sadekova, T., Kudinov, M.A.: Grad-tts: a diffusion probabilistic model for text-to-speech. In: International Conference on Machine Learning (2021). https:\/\/api.semanticscholar.org\/CorpusID:234483016"},{"key":"16_CR44","unstructured":"Ramazanova, M., Escorcia, V., Heilbron, F.C., Zhao, C., Ghanem, B.: Owl (observe, watch, listen): localizing actions in egocentric video via audiovisual temporal context (2022)"},{"key":"16_CR45","unstructured":"Saharia, C., et al.: Photorealistic text-to-image diffusion models with deep language understanding. ArXiv arxiv:2205.11487 (2022). https:\/\/api.semanticscholar.org\/CorpusID:248986576"},{"key":"16_CR46","unstructured":"Song, Y., Ermon, S.: Generative modeling by estimating gradients of the data distribution. In: Neural Information Processing Systems (2019). https:\/\/api.semanticscholar.org\/CorpusID:196470871"},{"key":"16_CR47","unstructured":"Soomro, K., Zamir, A.R., Shah, M.: Ucf101: a dataset of 101 human actions classes from videos in the wild. CoRR (2012)"},{"key":"16_CR48","doi-asserted-by":"crossref","unstructured":"Su, K., Qian, K., Shlizerman, E., Torralba, A., Gan, C.: Physics-driven diffusion models for impact sound synthesis from videos. In: 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 9749\u20139759 (2023). https:\/\/api.semanticscholar.org\/CorpusID:257805229","DOI":"10.1109\/CVPR52729.2023.00940"},{"key":"16_CR49","unstructured":"Wang, D., Chen, J.: Supervised speech separation based on deep learning: an overview. arxiv (201)"},{"key":"16_CR50","doi-asserted-by":"crossref","unstructured":"Wu*, Y., Chen*, K., Zhang*, T., Hui*, Y., Berg-Kirkpatrick, T., Dubnov, S.: Large-scale contrastive language-audio pretraining with feature fusion and keyword-to-caption augmentation. In: IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP (2023)","DOI":"10.1109\/ICASSP49357.2023.10095969"},{"key":"16_CR51","doi-asserted-by":"crossref","unstructured":"Yang, D., et al.: Diffsound: discrete diffusion model for text-to-sound generation. IEEE\/ACM Trans. Audio Speech Lang. Process. 31, 1720\u20131733 (2022). https:\/\/api.semanticscholar.org\/CorpusID:250698823","DOI":"10.1109\/TASLP.2023.3268730"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72897-6_16","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T23:19:25Z","timestamp":1733095165000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72897-6_16"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,2]]},"ISBN":["9783031728969","9783031728976"],"references-count":51,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72897-6_16","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,12,2]]},"assertion":[{"value":"2 December 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}