{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T07:58:44Z","timestamp":1776931124083,"version":"3.51.2"},"publisher-location":"New York, NY, USA","reference-count":73,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,4,13]]},"DOI":"10.1145\/3772318.3791655","type":"proceedings-article","created":{"date-parts":[[2026,4,13]],"date-time":"2026-04-13T09:47:11Z","timestamp":1776073631000},"page":"1-23","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Beyond Descriptions: A Generative Scene2Audio Framework for Blind and Low-Vision Users to Experience Vista Landscapes"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1350-9095","authenticated-orcid":false,"given":"Chitralekha","family":"Gupta","sequence":"first","affiliation":[{"name":"Augmented Human Lab, Dept. of Computer Science, National University of Singapore, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-4054-3547","authenticated-orcid":false,"given":"Jing","family":"Peng","sequence":"additional","affiliation":[{"name":"Augmented Human Lab, National University of Singapore, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1430-8770","authenticated-orcid":false,"given":"Ashwin","family":"Ram","sequence":"additional","affiliation":[{"name":"Saarland University, Saarland Informatics Campus, Saarbr\u00fccken, Germany"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-6675-3459","authenticated-orcid":false,"given":"Shreyas","family":"Sridhar","sequence":"additional","affiliation":[{"name":"School of Computing, National University of Singapore, Singapore, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0768-1019","authenticated-orcid":false,"given":"Christophe","family":"Jouffrais","sequence":"additional","affiliation":[{"name":"IRIT, CNRS, Toulouse, France and IPAL, CNRS, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7441-5493","authenticated-orcid":false,"given":"Suranga","family":"Nanayakkara","sequence":"additional","affiliation":[{"name":"Augmented Human Lab, School of Computing, National University of Singapore, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2026,4,13]]},"reference":[{"key":"e_1_3_3_2_2_2","doi-asserted-by":"publisher","DOI":"10.21785\/icad2019.029"},{"key":"e_1_3_3_2_3_2","doi-asserted-by":"publisher","DOI":"10.1145\/3410530.3414332"},{"key":"e_1_3_3_2_4_2","doi-asserted-by":"publisher","DOI":"10.1145\/3373625.3417001"},{"key":"e_1_3_3_2_5_2","doi-asserted-by":"publisher","DOI":"10.1145\/3290607.3313008"},{"key":"e_1_3_3_2_6_2","doi-asserted-by":"publisher","DOI":"10.4324\/9780080491103"},{"key":"e_1_3_3_2_7_2","doi-asserted-by":"crossref","unstructured":"Daniel\u00a0Ellis Berlyne. 1954. A theory of human curiosity. British Journal of Psychology (1954) 180\u2013191.","DOI":"10.1111\/j.2044-8295.1954.tb01243.x"},{"key":"e_1_3_3_2_8_2","doi-asserted-by":"crossref","unstructured":"Roger Boldu Denys\u00a0JC Matthies Haimo Zhang and Suranga Nanayakkara. 2020. AiSee: an assistive wearable device to support visually impaired grocery shoppers. Proceedings of the ACM on Interactive Mobile Wearable and Ubiquitous Technologies 4 4 (2020) 1\u201325.","DOI":"10.1145\/3432196"},{"key":"e_1_3_3_2_9_2","doi-asserted-by":"crossref","unstructured":"Paul\u00a0D Bolls and Annie Lang. 2003. I saw it on the radio: The allocation of attention to high-imagery radio advertisements. Media psychology 5 1 (2003) 33\u201355.","DOI":"10.1207\/S1532785XMEP0501_2"},{"key":"e_1_3_3_2_10_2","doi-asserted-by":"crossref","unstructured":"Virginia Braun and Victoria Clarke. 2021. One size fits all? What counts as quality practice in (reflexive) thematic analysis? Qualitative research in psychology 18 3 (2021) 328\u2013352.","DOI":"10.1080\/14780887.2020.1769238"},{"key":"e_1_3_3_2_11_2","volume-title":"Auditory scene analysis: The perceptual organization of sound","author":"Bregman Albert\u00a0S","year":"1994","unstructured":"Albert\u00a0S Bregman. 1994. Auditory scene analysis: The perceptual organization of sound. MIT press."},{"key":"e_1_3_3_2_12_2","doi-asserted-by":"publisher","DOI":"10.1145\/3313831.3376749"},{"key":"e_1_3_3_2_13_2","doi-asserted-by":"publisher","DOI":"10.7312\/chio18588"},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"publisher","DOI":"10.1145\/3382507.3418874"},{"key":"e_1_3_3_2_15_2","volume-title":"Understanding radio","author":"Crisell Andrew","year":"1994","unstructured":"Andrew Crisell. 1994. Understanding radio. Routledge."},{"key":"e_1_3_3_2_16_2","volume-title":"Flow: The psychology of optimal experience","author":"Czikszentmihalyi Mihaly","year":"1990","unstructured":"Mihaly Czikszentmihalyi. 1990. Flow: The psychology of optimal experience. New York: Harper & Row."},{"key":"e_1_3_3_2_17_2","doi-asserted-by":"crossref","unstructured":"Ga\u00ebl Dubus and Roberto Bresin. 2013. A systematic review of mapping strategies for the sonification of physical quantities. PloS one 8 12 (2013) e82491.","DOI":"10.1371\/journal.pone.0082491"},{"key":"e_1_3_3_2_18_2","doi-asserted-by":"crossref","unstructured":"Louise Fryer. 2010. Audio description as audio drama\u2013a practitioner\u2019s point of view. Perspectives: Studies in Translatology 18 3 (2010) 205\u2013213.","DOI":"10.1080\/0907676X.2010.485681"},{"key":"e_1_3_3_2_19_2","doi-asserted-by":"crossref","unstructured":"William\u00a0W Gaver. 1987. Auditory icons: Using sound in computer interfaces. ACM SIGCHI Bulletin 19 1 (1987) 74.","DOI":"10.1145\/28189.1044809"},{"key":"e_1_3_3_2_20_2","doi-asserted-by":"crossref","unstructured":"William\u00a0W Gaver. 1993. What in the world do we hear?: An ecological approach to auditory event perception. Ecological psychology 5 1 (1993) 1\u201329.","DOI":"10.1207\/s15326969eco0501_1"},{"key":"e_1_3_3_2_21_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952261"},{"key":"e_1_3_3_2_22_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3642211"},{"key":"e_1_3_3_2_23_2","doi-asserted-by":"crossref","unstructured":"Melanie\u00a0C Green and Timothy\u00a0C Brock. 2000. The role of transportation in the persuasiveness of public narratives. Journal of personality and social psychology 79 5 (2000) 701.","DOI":"10.1037\/0022-3514.79.5.701"},{"key":"e_1_3_3_2_24_2","unstructured":"Nielsen\u00a0Norman Group. 2012. How Many Test Users in a Usability Study? https:\/\/www.nngroup.com\/articles\/how-many-test-users\/. [Online; accessed 23-Jan-2024]."},{"key":"e_1_3_3_2_25_2","unstructured":"Nielsen\u00a0Norman Group. 2021. How to Conduct Usability Studies for Accessibility. https:\/\/media.nngroup.com\/media\/reports\/free\/How_to_Conduct_Usability_Studies_for_Accessibility.pdf. [Online; accessed 23-Jan-2024]."},{"key":"e_1_3_3_2_26_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10096328"},{"key":"e_1_3_3_2_27_2","doi-asserted-by":"publisher","unstructured":"Chitralekha Gupta Shreyas Sridhar Denys\u00a0J.C. Matthies Christophe Jouffrais and Suranga Nanayakkara. 2024. SonicVista: Towards Creating Awareness of Distant Scenes through Sonification. Proc. ACM Interact. Mob. Wearable Ubiquitous Technol. 8 2 Article 76 (may 2024) 32\u00a0pages. 10.1145\/3659609","DOI":"10.1145\/3659609"},{"key":"e_1_3_3_2_28_2","doi-asserted-by":"crossref","unstructured":"Vicki\u00a0L Hanson Anna Cavender and Shari Trewin. 2015. Writing about accessibility. Interactions 22 6 (2015) 62\u201365.","DOI":"10.1145\/2828432"},{"key":"e_1_3_3_2_29_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-77343-6_26"},{"key":"e_1_3_3_2_30_2","doi-asserted-by":"publisher","DOI":"10.1016\/S0166-4115(08)62386-9"},{"key":"e_1_3_3_2_31_2","doi-asserted-by":"crossref","unstructured":"Achim H\u00e4ttich and Martina Schweizer. 2020. I hear what you see: Effects of audio description used in a cinema on immersion and enjoyment in blind and visually impaired people. British Journal of Visual Impairment 38 3 (2020) 284\u2013298.","DOI":"10.1177\/0264619620911429"},{"key":"e_1_3_3_2_32_2","doi-asserted-by":"publisher","DOI":"10.1145\/3544549.3585610"},{"key":"e_1_3_3_2_33_2","doi-asserted-by":"publisher","DOI":"10.1145\/2049536.2049573"},{"key":"e_1_3_3_2_34_2","doi-asserted-by":"crossref","unstructured":"Zihao Ji Weijian Hu Ze Wang Kailun Yang and Kaiwei Wang. 2021. Seeing through events: Real-time moving object sonification for visually impaired people using event-based camera. Sensors 21 10 (2021) 3558.","DOI":"10.3390\/s21103558"},{"key":"e_1_3_3_2_35_2","doi-asserted-by":"publisher","DOI":"10.1145\/3597638.3608381"},{"key":"e_1_3_3_2_36_2","unstructured":"Purnima Kamath Chitralekha Gupta and Suranga Nanayakkara. 2024. MorphFader: Enabling Fine-grained Controllable Morphing with Text-to-Audio Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2408.07260 (2024)."},{"key":"e_1_3_3_2_37_2","doi-asserted-by":"crossref","unstructured":"Purnima Kamath Chitralekha Gupta Lonce Wyse and Suranga Nanayakkara. 2024. Example-Based Framework for Perceptually Guided Audio Texture Generation. IEEE\/ACM Transactions on Audio Speech and Language Processing (2024).","DOI":"10.1109\/TASLP.2024.3393741"},{"key":"e_1_3_3_2_38_2","doi-asserted-by":"crossref","unstructured":"Emine\u00a0Merve Kaya and Mounya Elhilali. 2017. Modelling auditory attention. Philosophical Transactions of the Royal Society B: Biological Sciences 372 1714 (2017) 20160101.","DOI":"10.1098\/rstb.2016.0101"},{"key":"e_1_3_3_2_39_2","unstructured":"Felix Kreuk Gabriel Synnaeve Adam Polyak Uriel Singer Alexandre D\u00e9fossez Jade Copet Devi Parikh Yaniv Taigman and Yossi Adi. 2022. Audiogen: Textually guided audio generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2209.15352 (2022)."},{"key":"e_1_3_3_2_40_2","volume-title":"The effects of noise on man","author":"Kryter Karl\u00a0D","year":"2013","unstructured":"Karl\u00a0D Kryter. 2013. The effects of noise on man. Elsevier."},{"key":"e_1_3_3_2_41_2","doi-asserted-by":"crossref","unstructured":"Alina Kuznetsova Hassan Rom Neil Alldrin Jasper Uijlings Ivan Krasin Jordi Pont-Tuset Shahab Kamali Stefan Popov Matteo Malloci Alexander Kolesnikov Tom Duerig and Vittorio Ferrari. 2020. The Open Images Dataset V4: Unified image classification object detection and visual relationship detection at scale. IJCV (2020).","DOI":"10.1007\/s11263-020-01316-z"},{"key":"e_1_3_3_2_42_2","doi-asserted-by":"crossref","unstructured":"J\u00a0Richard Landis and Gary\u00a0G Koch. 1977. The measurement of observer agreement for categorical data. biometrics (1977) 159\u2013174.","DOI":"10.2307\/2529310"},{"key":"e_1_3_3_2_43_2","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3580941"},{"key":"e_1_3_3_2_44_2","unstructured":"Haohe Liu Zehua Chen Yi Yuan Xinhao Mei Xubo Liu Danilo Mandic Wenwu Wang and Mark\u00a0D Plumbley. 2023. Audioldm: Text-to-audio generation with latent diffusion models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2301.12503 (2023)."},{"key":"e_1_3_3_2_45_2","doi-asserted-by":"publisher","DOI":"10.1145\/3411764.3445233"},{"key":"e_1_3_3_2_46_2","unstructured":"Mariana\u00a0Julieta Lopez Gavin Kearney and Krisztian Hofstadter. 2021. Enhancing audio description: inclusive cinematic experiences through sound design. Journal of Audiovisual Translation (2021) 157\u2013182."},{"key":"e_1_3_3_2_47_2","volume-title":"Motivation and personality","author":"Maslow Abraham\u00a0H","year":"1975","unstructured":"Abraham\u00a0H Maslow. 1975. Motivation and personality. Harper & Row."},{"key":"e_1_3_3_2_48_2","doi-asserted-by":"crossref","unstructured":"Keenan\u00a0R May Brianna\u00a0J Tomlinson Xiaomeng Ma Phillip Roberts and Bruce\u00a0N Walker. 2020. Spotlights and soundscapes: On the design of mixed reality auditory environments for persons with visual impairment. ACM Transactions on Accessible Computing (TACCESS) 13 2 (2020) 1\u201347.","DOI":"10.1145\/3378576"},{"key":"e_1_3_3_2_49_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3642177"},{"key":"e_1_3_3_2_50_2","doi-asserted-by":"crossref","unstructured":"Mary\u00a0L McHugh. 2012. Interrater reliability: the kappa statistic. Biochemia medica 22 3 (2012) 276\u2013282.","DOI":"10.11613\/BM.2012.031"},{"key":"e_1_3_3_2_51_2","unstructured":"Saul McLeod. 2007. Maslow\u2019s hierarchy of needs. Simply psychology1 (2007) 1\u201318."},{"key":"e_1_3_3_2_52_2","doi-asserted-by":"publisher","unstructured":"Miguel Melo Guilherme Gon\u00e7alves jos\u00e9 Vasconcelos-Raposo and Maximino Bessa. 2023. How Much Presence is Enough? Qualitative Scales for Interpreting the Igroup Presence Questionnaire Score. IEEE Access 11 (2023) 24675\u201324685. 10.1109\/ACCESS.2023.3254892","DOI":"10.1109\/ACCESS.2023.3254892"},{"key":"e_1_3_3_2_53_2","doi-asserted-by":"publisher","DOI":"10.1007\/3-540-57207-4_21"},{"key":"e_1_3_3_2_54_2","volume-title":"An introduction to the psychology of hearing","author":"Moore Brian\u00a0CJ","year":"2012","unstructured":"Brian\u00a0CJ Moore. 2012. An introduction to the psychology of hearing. Brill."},{"key":"e_1_3_3_2_55_2","doi-asserted-by":"publisher","DOI":"10.1145\/3173574.3173633"},{"key":"e_1_3_3_2_56_2","doi-asserted-by":"publisher","DOI":"10.4324\/9781315647517"},{"key":"e_1_3_3_2_57_2","doi-asserted-by":"crossref","unstructured":"Aude Oliva and Antonio Torralba. 2001. Modeling the shape of the scene: A holistic representation of the spatial envelope. International journal of computer vision 42 (2001) 145\u2013175.","DOI":"10.1023\/A:1011139631724"},{"key":"e_1_3_3_2_58_2","doi-asserted-by":"crossref","unstructured":"Robert\u00a0F Potter. 2006. Made you listen: The effects of production effects on automatic attention to short radio promotional announcements. Journal of promotion management 12 2 (2006) 35\u201348.","DOI":"10.1300\/J057v12n02_04"},{"key":"e_1_3_3_2_59_2","doi-asserted-by":"crossref","unstructured":"Thomas Potter Zoran Cvetkovi\u0107 and Enzo De\u00a0Sena. 2022. On the relative importance of visual and spatial audio rendering on vr immersion. Frontiers in Signal Processing 2 (2022) 904866.","DOI":"10.3389\/frsip.2022.904866"},{"key":"e_1_3_3_2_60_2","first-page":"8748","volume-title":"International conference on machine learning","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et\u00a0al. 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PMLR, 8748\u20138763."},{"key":"e_1_3_3_2_61_2","doi-asserted-by":"crossref","unstructured":"Kyle Rector Keith Salmon Dan Thornton Neel Joshi and Meredith\u00a0Ringel Morris. 2017. Eyes-free art: Exploring proxemic audio interfaces for blind and low vision art engagement. Proceedings of the ACM on Interactive Mobile Wearable and Ubiquitous Technologies 1 3 (2017) 1\u201321.","DOI":"10.1145\/3130958"},{"key":"e_1_3_3_2_62_2","doi-asserted-by":"crossref","unstructured":"Emma Rodero. 2012. See it on a radio story: sound effects and shots to evoked imagery and attention on audio fiction. Communication research 39 4 (2012) 458\u2013479.","DOI":"10.1177\/0093650210386947"},{"key":"e_1_3_3_2_63_2","doi-asserted-by":"crossref","unstructured":"Emma Rodero. 2019. The spark orientation effect for improving attention and recall. Communication Research 46 7 (2019) 965\u2013985.","DOI":"10.1177\/0093650215609085"},{"key":"e_1_3_3_2_64_2","doi-asserted-by":"publisher","DOI":"10.4324\/9781351015356-2"},{"key":"e_1_3_3_2_65_2","doi-asserted-by":"publisher","DOI":"10.1145\/1978942.1979268"},{"key":"e_1_3_3_2_66_2","doi-asserted-by":"publisher","DOI":"10.1145\/3210825.3213565"},{"key":"e_1_3_3_2_67_2","doi-asserted-by":"publisher","DOI":"10.1145\/3517428.3544813"},{"key":"e_1_3_3_2_68_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10096023"},{"key":"e_1_3_3_2_69_2","doi-asserted-by":"publisher","DOI":"10.1145\/3434074.3447186"},{"key":"e_1_3_3_2_70_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613905.3650927"},{"key":"e_1_3_3_2_71_2","doi-asserted-by":"publisher","DOI":"10.1145\/3313831.3376886"},{"key":"e_1_3_3_2_72_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3642466"},{"key":"e_1_3_3_2_73_2","unstructured":"Zeyu Xie Xuenan Xu Zhizheng Wu and Mengyue Wu. 2024. PicoAudio: Enabling Precise Timestamp and Frequency Controllability of Audio Events in Text-to-audio Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2407.02869 (2024)."},{"key":"e_1_3_3_2_74_2","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300341"}],"event":{"name":"CHI 2026: CHI Conference on Human Factors in Computing Systems","location":"Barcelona Spain","acronym":"CHI '26","sponsor":["SIGCHI ACM Special Interest Group on Computer-Human Interaction"]},"container-title":["Proceedings of the 2026 CHI Conference on Human Factors in Computing Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3772318.3791655","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,15]],"date-time":"2026-04-15T10:31:31Z","timestamp":1776249091000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3772318.3791655"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,13]]},"references-count":73,"alternative-id":["10.1145\/3772318.3791655","10.1145\/3772318"],"URL":"https:\/\/doi.org\/10.1145\/3772318.3791655","relation":{},"subject":[],"published":{"date-parts":[[2026,4,13]]},"assertion":[{"value":"2026-04-13","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}