{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,13]],"date-time":"2026-04-13T16:04:17Z","timestamp":1776096257028,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":54,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,4,25]],"date-time":"2025-04-25T00:00:00Z","timestamp":1745539200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,4,26]]},"DOI":"10.1145\/3706599.3719849","type":"proceedings-article","created":{"date-parts":[[2025,4,23]],"date-time":"2025-04-23T20:20:42Z","timestamp":1745439642000},"page":"1-9","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Scene-to-Audio: Distant Scene Sonification for Blind and Low Vision People"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1350-9095","authenticated-orcid":false,"given":"Chitralekha","family":"Gupta","sequence":"first","affiliation":[{"name":"Augmented Human Lab, School of Computing, National University of Singapore, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1430-8770","authenticated-orcid":false,"given":"Ashwin","family":"Ram","sequence":"additional","affiliation":[{"name":"Saarland Informatics Campus, Saarland University, Saarbr\u00fccken, Saarland, Germany"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-6675-3459","authenticated-orcid":false,"given":"Shreyas","family":"Sridhar","sequence":"additional","affiliation":[{"name":"School of Computing, National University of Singapore, Singapore, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0768-1019","authenticated-orcid":false,"given":"Christophe","family":"Jouffrais","sequence":"additional","affiliation":[{"name":"IPAL, CNRS, Singapore, Singapore and IRIT, Univ of Toulouse, Toulouse, France"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7441-5493","authenticated-orcid":false,"given":"Suranga","family":"Nanayakkara","sequence":"additional","affiliation":[{"name":"Augmented Human Lab, School of Computing, National University of Singapore, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,4,25]]},"reference":[{"key":"e_1_3_3_3_2_2","doi-asserted-by":"publisher","DOI":"10.21785\/icad2019.029"},{"key":"e_1_3_3_3_3_2","doi-asserted-by":"publisher","DOI":"10.1145\/3410530.3414332"},{"key":"e_1_3_3_3_4_2","doi-asserted-by":"publisher","DOI":"10.1145\/3373625.3417001"},{"key":"e_1_3_3_3_5_2","doi-asserted-by":"publisher","DOI":"10.1145\/3290607.3313008"},{"key":"e_1_3_3_3_6_2","doi-asserted-by":"publisher","DOI":"10.4324\/9780240825007"},{"key":"e_1_3_3_3_7_2","doi-asserted-by":"crossref","unstructured":"Paul\u00a0D Bolls and Annie Lang. 2003. I saw it on the radio: The allocation of attention to high-imagery radio advertisements. Media psychology 5 1 (2003) 33\u201355.","DOI":"10.1207\/S1532785XMEP0501_2"},{"key":"e_1_3_3_3_8_2","doi-asserted-by":"crossref","unstructured":"Virginia Braun and Victoria Clarke. 2021. One size fits all? What counts as quality practice in (reflexive) thematic analysis? Qualitative research in psychology 18 3 (2021) 328\u2013352.","DOI":"10.1080\/14780887.2020.1769238"},{"key":"e_1_3_3_3_9_2","volume-title":"Auditory scene analysis: The perceptual organization of sound","author":"Bregman Albert\u00a0S","year":"1994","unstructured":"Albert\u00a0S Bregman. 1994. Auditory scene analysis: The perceptual organization of sound. MIT press, Cambridge, MA."},{"key":"e_1_3_3_3_10_2","doi-asserted-by":"publisher","DOI":"10.1145\/3643834.3661556"},{"key":"e_1_3_3_3_11_2","doi-asserted-by":"publisher","DOI":"10.7312\/chio18588"},{"key":"e_1_3_3_3_12_2","doi-asserted-by":"publisher","DOI":"10.1145\/3382507.3418874"},{"key":"e_1_3_3_3_13_2","volume-title":"Understanding radio - 2nd Edition","author":"Crisell Andrew","year":"1994","unstructured":"Andrew Crisell. 1994. Understanding radio - 2nd Edition. Routledge, New York."},{"key":"e_1_3_3_3_14_2","doi-asserted-by":"crossref","unstructured":"Ga\u00ebl Dubus and Roberto Bresin. 2013. A systematic review of mapping strategies for the sonification of physical quantities. PloS one 8 12 (2013) e82491.","DOI":"10.1371\/journal.pone.0082491"},{"key":"e_1_3_3_3_15_2","doi-asserted-by":"crossref","unstructured":"Louise Fryer. 2010. Audio description as audio drama\u2013a practitioner\u2019s point of view. Perspectives: Studies in Translatology 18 3 (2010) 205\u2013213.","DOI":"10.1080\/0907676X.2010.485681"},{"key":"e_1_3_3_3_16_2","doi-asserted-by":"crossref","unstructured":"William\u00a0W Gaver. 1987. Auditory icons: Using sound in computer interfaces. ACM SIGCHI Bulletin 19 1 (1987) 74.","DOI":"10.1145\/28189.1044809"},{"key":"e_1_3_3_3_17_2","doi-asserted-by":"crossref","unstructured":"William\u00a0W Gaver. 1993. What in the world do we hear?: An ecological approach to auditory event perception. Ecological psychology 5 1 (1993) 1\u201329.","DOI":"10.1207\/s15326969eco0501_1"},{"key":"e_1_3_3_3_18_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3642211"},{"key":"e_1_3_3_3_19_2","doi-asserted-by":"crossref","unstructured":"Melanie\u00a0C Green and Timothy\u00a0C Brock. 2000. The role of transportation in the persuasiveness of public narratives. Journal of personality and social psychology 79 5 (2000) 701.","DOI":"10.1037\/\/0022-3514.79.5.701"},{"key":"e_1_3_3_3_20_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10096328"},{"key":"e_1_3_3_3_21_2","doi-asserted-by":"publisher","unstructured":"Chitralekha Gupta Shreyas Sridhar Denys\u00a0J.C. Matthies Christophe Jouffrais and Suranga Nanayakkara. 2024. SonicVista: Towards Creating Awareness of Distant Scenes through Sonification. Proc. ACM Interact. Mob. Wearable Ubiquitous Technol. 8 2 Article 76 (may 2024) 32\u00a0pages. 10.1145\/3659609","DOI":"10.1145\/3659609"},{"key":"e_1_3_3_3_22_2","doi-asserted-by":"crossref","unstructured":"Vicki\u00a0L Hanson Anna Cavender and Shari Trewin. 2015. Writing about accessibility. Interactions 22 6 (2015) 62\u201365.","DOI":"10.1145\/2828432"},{"key":"e_1_3_3_3_23_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-77343-626"},{"key":"e_1_3_3_3_24_2","doi-asserted-by":"crossref","unstructured":"Achim H\u00e4ttich and Martina Schweizer. 2020. I hear what you see: Effects of audio description used in a cinema on immersion and enjoyment in blind and visually impaired people. British Journal of Visual Impairment 38 3 (2020) 284\u2013298.","DOI":"10.1177\/0264619620911429"},{"key":"e_1_3_3_3_25_2","doi-asserted-by":"publisher","DOI":"10.1145\/3544549.3585610"},{"key":"e_1_3_3_3_26_2","doi-asserted-by":"crossref","unstructured":"Zihao Ji Weijian Hu Ze Wang Kailun Yang and Kaiwei Wang. 2021. Seeing through events: Real-time moving object sonification for visually impaired people using event-based camera. Sensors 21 10 (2021) 3558.","DOI":"10.3390\/s21103558"},{"key":"e_1_3_3_3_27_2","doi-asserted-by":"publisher","DOI":"10.1145\/3597638.3608381"},{"key":"e_1_3_3_3_28_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49660.2025.10890164"},{"key":"e_1_3_3_3_29_2","doi-asserted-by":"publisher","unstructured":"Purnima Kamath Chitralekha Gupta Lonce Wyse and Suranga Nanayakkara. 2024. Example-Based Framework for Perceptually Guided Audio Texture Generation. IEEE\/ACM Trans. Audio Speech and Lang. Proc. 32 (April 2024) 2555\u20132565. 10.1109\/TASLP.2024.3393741","DOI":"10.1109\/TASLP.2024.3393741"},{"key":"e_1_3_3_3_30_2","unstructured":"Felix Kreuk Gabriel Synnaeve Adam Polyak Uriel Singer Alexandre D\u2019efossez Jade Copet Devi Parikh Yaniv Taigman and Yossi Adi. 2022. AudioGen: Textually Guided Audio Generation. ArXiv abs\/2209.15352 (2022). https:\/\/api.semanticscholar.org\/CorpusID:252668761"},{"key":"e_1_3_3_3_31_2","volume-title":"The effects of noise on man","author":"Kryter Karl\u00a0D","year":"1970","unstructured":"Karl\u00a0D Kryter. 1970. The effects of noise on man. Academic Press, Elsevier, New York."},{"key":"e_1_3_3_3_32_2","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3580941"},{"key":"e_1_3_3_3_33_2","series-title":"Proceedings of Machine Learning Research","first-page":"21450","volume-title":"Proceedings of the 40th International Conference on Machine Learning","volume":"202","author":"Liu Haohe","year":"2023","unstructured":"Haohe Liu, Zehua Chen, Yi Yuan, Xinhao Mei, Xubo Liu, Danilo Mandic, Wenwu Wang, and Mark\u00a0D Plumbley. 2023. AudioLDM: Text-to-Audio Generation with Latent Diffusion Models. In Proceedings of the 40th International Conference on Machine Learning(Proceedings of Machine Learning Research, Vol.\u00a0202), Andreas Krause, Emma Brunskill, Kyunghyun Cho, Barbara Engelhardt, Sivan Sabato, and Jonathan Scarlett (Eds.). PMLR, Cambridge MA, USA, 21450\u201321474. https:\/\/proceedings.mlr.press\/v202\/liu23f.html"},{"key":"e_1_3_3_3_34_2","doi-asserted-by":"publisher","DOI":"10.1145\/3411764.3445233"},{"key":"e_1_3_3_3_35_2","doi-asserted-by":"publisher","unstructured":"Mariana\u00a0Julieta Lopez Gavin Kearney and Krisztian Hofstadter. 2021. Enhancing audio description: inclusive cinematic experiences through sound design. Journal of Audiovisual Translation 4 1 (2021) 157\u2013182. 10.47476\/jat.v4i1.2021.154","DOI":"10.47476\/jat.v4i1.2021.154"},{"key":"e_1_3_3_3_36_2","volume-title":"Motivation and personality","author":"Maslow Abraham\u00a0H","year":"1954","unstructured":"Abraham\u00a0H Maslow. 1954. Motivation and personality. Harper & Row, New York."},{"key":"e_1_3_3_3_37_2","doi-asserted-by":"crossref","unstructured":"Keenan\u00a0R May Brianna\u00a0J Tomlinson Xiaomeng Ma Phillip Roberts and Bruce\u00a0N Walker. 2020. Spotlights and soundscapes: On the design of mixed reality auditory environments for persons with visual impairment. ACM Transactions on Accessible Computing (TACCESS) 13 2 (2020) 1\u201347.","DOI":"10.1145\/3378576"},{"key":"e_1_3_3_3_38_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3642177"},{"key":"e_1_3_3_3_39_2","unstructured":"Saul McLeod. 2007 [Last accessed 1 March 2025]. Maslow\u2019s hierarchy of needs. Simply psychology https:\/\/www.simplypsychology.org\/maslow.html."},{"key":"e_1_3_3_3_40_2","doi-asserted-by":"publisher","DOI":"10.1007\/3-540-57207-4_21"},{"key":"e_1_3_3_3_41_2","volume-title":"An Introduction to the Psychology of Hearing: Sixth Edition","author":"Moore Brian","year":"2013","unstructured":"Brian Moore. 2013. An Introduction to the Psychology of Hearing: Sixth Edition. Brill, Leiden, The Netherlands. https:\/\/brill.com\/view\/title\/24210"},{"key":"e_1_3_3_3_42_2","doi-asserted-by":"publisher","DOI":"10.4324\/9781315647517"},{"key":"e_1_3_3_3_43_2","doi-asserted-by":"crossref","unstructured":"Aude Oliva and Antonio Torralba. 2001. Modeling the shape of the scene: A holistic representation of the spatial envelope. International journal of computer vision 42 (2001) 145\u2013175.","DOI":"10.1023\/A:1011139631724"},{"key":"e_1_3_3_3_44_2","first-page":"8748","volume-title":"International conference on machine learning","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et\u00a0al. 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PMLR, Cambridge MA, USA, 8748\u20138763."},{"key":"e_1_3_3_3_45_2","doi-asserted-by":"publisher","DOI":"10.4324\/9781351015356-2"},{"key":"e_1_3_3_3_46_2","doi-asserted-by":"publisher","DOI":"10.1145\/3210825.3213565"},{"key":"e_1_3_3_3_47_2","doi-asserted-by":"publisher","DOI":"10.1145\/3517428.3544813"},{"key":"e_1_3_3_3_48_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10096023"},{"key":"e_1_3_3_3_49_2","doi-asserted-by":"publisher","DOI":"10.1145\/3434074.3447186"},{"key":"e_1_3_3_3_50_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613905.3650927"},{"key":"e_1_3_3_3_51_2","doi-asserted-by":"publisher","DOI":"10.1145\/3313831.3376886"},{"key":"e_1_3_3_3_52_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3642466"},{"key":"e_1_3_3_3_53_2","unstructured":"Zeyu Xie Xuenan Xu Zhizheng Wu and Mengyue Wu. 2024. PicoAudio: Enabling Precise Timestamp and Frequency Controllability of Audio Events in Text-to-audio Generation. arxiv:https:\/\/arXiv.org\/abs\/2407.02869\u00a0[cs.SD] https:\/\/arxiv.org\/abs\/2407.02869"},{"key":"e_1_3_3_3_54_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02537"},{"key":"e_1_3_3_3_55_2","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300341"}],"event":{"name":"CHI EA '25: Extended Abstracts of the CHI Conference on Human Factors in Computing Systems","location":"Yokohama Japan","acronym":"CHI EA '25","sponsor":["SIGCHI ACM Special Interest Group on Computer-Human Interaction"]},"container-title":["Proceedings of the Extended Abstracts of the CHI Conference on Human Factors in Computing Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3706599.3719849","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3706599.3719849","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:18:35Z","timestamp":1750295915000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3706599.3719849"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,4,25]]},"references-count":54,"alternative-id":["10.1145\/3706599.3719849","10.1145\/3706599"],"URL":"https:\/\/doi.org\/10.1145\/3706599.3719849","relation":{},"subject":[],"published":{"date-parts":[[2025,4,25]]},"assertion":[{"value":"2025-04-25","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}