{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,8]],"date-time":"2026-06-08T23:34:16Z","timestamp":1780961656904,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":69,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,5,11]],"date-time":"2024-05-11T00:00:00Z","timestamp":1715385600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,5,11]]},"DOI":"10.1145\/3613904.3642162","type":"proceedings-article","created":{"date-parts":[[2024,5,11]],"date-time":"2024-05-11T08:39:12Z","timestamp":1715416752000},"page":"1-19","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":11,"title":["Unspoken Sound: Identifying Trends in Non-Speech Audio Captioning on YouTube"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-4692-8261","authenticated-orcid":false,"given":"Lloyd","family":"May","sequence":"first","affiliation":[{"name":"Center for Computer Research in Music and Acoustics (CCRMA), Stanford University, United States"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8612-1898","authenticated-orcid":false,"given":"Keita","family":"Ohshiro","sequence":"additional","affiliation":[{"name":"Department of Informatics, New Jersey Institute of Technology, United States"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3481-497X","authenticated-orcid":false,"given":"Khang","family":"Dang","sequence":"additional","affiliation":[{"name":"Informatics, New Jersey Institute of Technology, United States"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9761-3564","authenticated-orcid":false,"given":"Sripathi","family":"Sridhar","sequence":"additional","affiliation":[{"name":"Sound Interaction and Computing Lab, New Jersey Institute of Technology, United States"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-6616-2579","authenticated-orcid":false,"given":"Jhanvi","family":"Pai","sequence":"additional","affiliation":[{"name":"Sound Interaction and Computing Lab, New Jersey Institute of Technology, United States"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4506-6639","authenticated-orcid":false,"given":"Magdalena","family":"Fuentes","sequence":"additional","affiliation":[{"name":"MARL-IDM, New York University, United States"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4971-2004","authenticated-orcid":false,"given":"Sooyeon","family":"Lee","sequence":"additional","affiliation":[{"name":"Informatics\/Ying Wu College of Computing, New Jersey Institute of Technology, United States"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5908-390X","authenticated-orcid":false,"given":"Mark","family":"Cartwright","sequence":"additional","affiliation":[{"name":"New York University, United States"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,5,11]]},"reference":[{"key":"e_1_3_3_3_1_1","unstructured":"2021. Closed Captioning of Internet Video Programming. https:\/\/www.fcc.gov\/consumers\/guides\/captioning-internet-video-programming"},{"key":"e_1_3_3_3_2_1","unstructured":"2021. Closed Captioning on Television. https:\/\/www.fcc.gov\/consumers\/guides\/closed-captioning-television"},{"key":"e_1_3_3_3_3_1","unstructured":"2023. YouDescribe - Audio Description for Youtube Videos. https:\/\/www.youdescribe.org\/."},{"key":"e_1_3_3_3_6_1","unstructured":"Akhter Al\u00a0Amin. 2020. Audio-Visual Caption Evaluation Metric for People who are Deaf and Hard of Hearing. (2020)."},{"key":"e_1_3_3_3_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3517428.3544808"},{"key":"e_1_3_3_3_8_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-78095-1_15"},{"key":"e_1_3_3_3_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3430263.3452429"},{"key":"e_1_3_3_3_10_1","volume-title":"International Conference on Human-Computer Interaction. Springer, 202\u2013220","author":"Amin Akhter\u00a0Al","year":"2021","unstructured":"Akhter\u00a0Al Amin, Saad Hassan, and Matt Huenerfauth. 2021. Effect of occlusion on deaf and hard of hearing users\u2019 perception of captioned video quality. In International Conference on Human-Computer Interaction. Springer, 202\u2013220."},{"key":"e_1_3_3_3_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3491102.3517681"},{"key":"e_1_3_3_3_12_1","volume-title":"Deaf and Hard of Hearing Viewers\u2019 Preference for Speaker Identifier Type in Live TV Programming. In International Conference on Human-Computer Interaction. Springer, 200\u2013211","author":"Amin Akher\u00a0Al","year":"2022","unstructured":"Akher\u00a0Al Amin, Joseph Mendis, Raja Kushalnagar, Christian Vogler, Sooyeon Lee, and Matt Huenerfauth. 2022. Deaf and Hard of Hearing Viewers\u2019 Preference for Speaker Identifier Type in Live TV Programming. In International Conference on Human-Computer Interaction. Springer, 200\u2013211."},{"key":"e_1_3_3_3_13_1","volume-title":"Caption Accuracy Metrics Project Research into Automated Error Ranking of Real-time Captions in Live Television News Programs","author":"Apone Tom","year":"2011","unstructured":"Tom Apone, Brad Botkin, Marcia Brooks, and Larry Goldberg. 2011. Caption Accuracy Metrics Project Research into Automated Error Ranking of Real-time Captions in Live Television News Programs. The Carl and Ruth Shapiro Family National Center for Accessible Media, Boston (2011)."},{"key":"e_1_3_3_3_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3290607.3312921"},{"key":"e_1_3_3_3_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3290607.3312921"},{"key":"e_1_3_3_3_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/2745197.2745204"},{"key":"e_1_3_3_3_17_1","volume-title":"Web content accessibility guidelines (WCAG) 2.0","author":"Caldwell Ben","year":"2008","unstructured":"Ben Caldwell, Michael Cooper, Loretta\u00a0Guarino Reid, Gregg Vanderheiden, Wendy Chisholm, John Slatin, and Jason White. 2008. Web content accessibility guidelines (WCAG) 2.0. WWW Consortium (W3C) 290 (2008), 1\u201334."},{"key":"e_1_3_3_3_18_1","unstructured":"Sourish Chaudhuri. 2017. Adding sound effect information to YouTube captions. https:\/\/ai.googleblog.com\/2017\/03\/adding-sound-effect-information-to.html"},{"key":"e_1_3_3_3_19_1","volume-title":"Twenty-First Century Communications and Video Accessibility Act of","author":"Congres U.S.","year":"2010","unstructured":"U.S. Congres. 2010. Twenty-First Century Communications and Video Accessibility Act of 2010."},{"key":"e_1_3_3_3_20_1","volume-title":"Television Decoder Circuitry Act of","author":"Congress U.S.","year":"1990","unstructured":"U.S. Congress. 1990. Television Decoder Circuitry Act of 1990. https:\/\/www.congress.gov\/bill\/101st-congress\/senate-bill\/1974 Pub. L. No. 101-431."},{"key":"e_1_3_3_3_21_1","volume-title":"Telecommunications Act of","author":"Congress U.S.","year":"1996","unstructured":"U.S. Congress. 1996. Telecommunications Act of 1996."},{"key":"e_1_3_3_3_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3581511"},{"key":"e_1_3_3_3_23_1","volume-title":"Closed captioning: Subtitling, stenography, and the digital convergence of text with television","author":"Downey J","unstructured":"Gregory\u00a0J Downey. 2008. Closed captioning: Subtitling, stenography, and the digital convergence of text with television. JHU Press."},{"key":"e_1_3_3_3_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/WASPAA.2017.8170058"},{"key":"e_1_3_3_3_25_1","unstructured":"Meryl\u00a0K Evans. 2019. Here\u2019s how automatic captions earned their nickname.https:\/\/www.youtube.com\/watch?v=N7MfajxyWDY"},{"key":"e_1_3_3_3_26_1","doi-asserted-by":"publisher","DOI":"10.1037\/e578132012-014"},{"key":"e_1_3_3_3_27_1","unstructured":"Andrew Gallagher Terrance McCartney Zhonghua Xi and Sourish Chaudhuri. 2017. Captions based on speaker identification. (2017)."},{"key":"e_1_3_3_3_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3411764.3445509"},{"key":"e_1_3_3_3_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3234695.3241023"},{"key":"e_1_3_3_3_30_1","unstructured":"Ken Harrenstien. 2009. Automatic captions in YouTube. https:\/\/googleblog.blogspot.com\/2009\/11\/automatic-captions-in-youtube.html"},{"key":"e_1_3_3_3_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3544549.3585880"},{"key":"e_1_3_3_3_32_1","unstructured":"Shawn Henry. 2022. Captions\/Subtitles. https:\/\/www.w3.org\/WAI\/media\/av\/captions\/"},{"key":"e_1_3_3_3_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/2632111"},{"key":"e_1_3_3_3_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3581494"},{"key":"e_1_3_3_3_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300324"},{"key":"e_1_3_3_3_36_1","doi-asserted-by":"crossref","unstructured":"Bo Jiang Sijiang Liu Liping He Weimin Wu Hongli Chen and Yunfei Shen. 2017. Subtitle positioning for e-learning videos based on rough gaze estimation and saliency detection. In SIGGRAPH Asia 2017 Posters. 1\u20132.","DOI":"10.1145\/3145690.3145735"},{"key":"e_1_3_3_3_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3325862"},{"key":"e_1_3_3_3_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/2982142.2982164"},{"key":"e_1_3_3_3_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3597638.3608425"},{"key":"e_1_3_3_3_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3581130"},{"key":"e_1_3_3_3_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3313831.3376266"},{"key":"e_1_3_3_3_42_1","volume-title":"RTTD-ID: Tracked captions with multiple speakers for deaf students. arXiv preprint arXiv:1909.08172","author":"Kushalnagar Raja","year":"2019","unstructured":"Raja Kushalnagar, Gary Behm, Kevin Wolfe, Peter Yeung, Becca Dingman, Shareef Ali, Abraham Glasser, and Claire Ryan. 2019. RTTD-ID: Tracked captions with multiple speakers for deaf students. arXiv preprint arXiv:1909.08172 (2019)."},{"key":"e_1_3_3_3_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/2661334.2661381"},{"key":"e_1_3_3_3_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/2501988.2502057"},{"key":"e_1_3_3_3_45_1","doi-asserted-by":"crossref","first-page":"11","DOI":"10.1145\/1279540.1279551","article-title":"Emotive captioning","volume":"5","author":"Lee G","year":"2007","unstructured":"Daniel\u00a0G Lee, Deborah\u00a0I Fels, and John\u00a0Patrick Udo. 2007. Emotive captioning. Computers in Entertainment (CIE) 5, 2 (2007), 11.","journal-title":"Computers in Entertainment (CIE)"},{"key":"e_1_3_3_3_46_1","unstructured":"Elisa Lewis. 2018. 7 ways captions and transcripts improve video SEO. https:\/\/www.3playmedia.com\/blog\/7-ways-video-transcripts-captions-improve-seo\/"},{"key":"e_1_3_3_3_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/3512922"},{"key":"e_1_3_3_3_48_1","unstructured":"F.\u00a0Wai ling Ho-Ching Jennifer Mankoff and James\u00a0A. Landay. 2002. From Data to Display: the Design and Evaluation of a Peripheral Sound Display for the Deaf. https:\/\/api.semanticscholar.org\/CorpusID:10800692"},{"key":"e_1_3_3_3_49_1","volume-title":"Visually-aware audio captioning with adaptive audio-visual attention. arXiv preprint arXiv:2210.16428","author":"Liu Xubo","year":"2022","unstructured":"Xubo Liu, Qiushi Huang, Xinhao Mei, Haohe Liu, Qiuqiang Kong, Jianyuan Sun, Shengchen Li, Tom Ko, Yu Zhang, Lilian\u00a0H Tang, 2022. Visually-aware audio captioning with adaptive audio-visual attention. arXiv preprint arXiv:2210.16428 (2022)."},{"key":"e_1_3_3_3_50_1","unstructured":"Kim Lyons. 2020. YouTube is ending its community captions feature and deaf creators aren\u2019t happy about it. https:\/\/www.theverge.com\/2020\/7\/31\/21349401\/youtube-community-captions-deaf-creators-accessibility-google"},{"key":"e_1_3_3_3_51_1","doi-asserted-by":"publisher","DOI":"10.1080\/01449290600636488"},{"key":"e_1_3_3_3_52_1","doi-asserted-by":"publisher","DOI":"10.1145\/3597638.3608398"},{"key":"e_1_3_3_3_53_1","unstructured":"Described Media\u00a0Program and Captioned. 2017. Captioning key - sound effects and music. https:\/\/dcmp.org\/learn\/602-captioning-key\u2014sound-effects-and-music"},{"key":"e_1_3_3_3_54_1","volume-title":"Automated audio captioning: an overview of recent progress and new challenges. EURASIP journal on audio, speech, and music processing","author":"Mei Xinhao","year":"2022","unstructured":"Xinhao Mei, Xubo Liu, Mark\u00a0D Plumbley, and Wenwu Wang. 2022. Automated audio captioning: an overview of recent progress and new challenges. EURASIP journal on audio, speech, and music processing 2022, 1 (2022), 1\u201318."},{"key":"e_1_3_3_3_55_1","volume-title":"The Global","author":"Murphy Andrea","year":"2000","unstructured":"Andrea Murphy and Hank Tucker. 2023. The Global 2000. https:\/\/www.forbes.com\/lists\/global2000"},{"key":"e_1_3_3_3_56_1","volume-title":"Proceedings of the 2013 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies. 201\u2013210","author":"Naim Iftekhar","year":"2013","unstructured":"Iftekhar Naim, Daniel Gildea, Walter Lasecki, and Jeffrey\u00a0P Bigham. 2013. Text alignment for real-time crowd captioning. In Proceedings of the 2013 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies. 201\u2013210."},{"key":"e_1_3_3_3_57_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSS.2020.2972399"},{"key":"e_1_3_3_3_58_1","doi-asserted-by":"publisher","DOI":"10.1145\/2507065.2507100"},{"key":"e_1_3_3_3_59_1","unstructured":"Ellie Parfitt. 2016. Auto-generated captions are often wrong. https:\/\/www.hearinglikeme.com\/nomorecraptions\/"},{"key":"e_1_3_3_3_60_1","doi-asserted-by":"publisher","DOI":"10.1145\/3379337.3415864"},{"key":"e_1_3_3_3_61_1","volume-title":"Dancing with words: Using animated text for captioning. Intl. Journal of Human\u2013Computer Interaction 24, 5","author":"Rashid Raisa","year":"2008","unstructured":"Raisa Rashid, Quoc Vy, Richard Hunt, and Deborah\u00a0I Fels. 2008. Dancing with words: Using animated text for captioning. Intl. Journal of Human\u2013Computer Interaction 24, 5 (2008), 505\u2013519."},{"key":"e_1_3_3_3_62_1","volume-title":"Accuracy rate in live subtitling: The NER model. Audiovisual translation in a global context: Mapping an ever-changing landscape","author":"Romero-Fresco Pablo","year":"2015","unstructured":"Pablo Romero-Fresco and Juan\u00a0Mart\u00ednez P\u00e9rez. 2015. Accuracy rate in live subtitling: The NER model. Audiovisual translation in a global context: Mapping an ever-changing landscape (2015), 28\u201350."},{"key":"e_1_3_3_3_63_1","doi-asserted-by":"crossref","unstructured":"James Sandford. 2015. The impact of subtitle display rate on enjoyment under normal television viewing conditions. (2015).","DOI":"10.1049\/ibc.2015.0018"},{"key":"e_1_3_3_3_64_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-03658-3_110"},{"key":"e_1_3_3_3_65_1","doi-asserted-by":"publisher","DOI":"10.1145\/1969289.1969318"},{"key":"e_1_3_3_3_66_1","doi-asserted-by":"publisher","DOI":"10.1145\/2982142.2982205"},{"key":"e_1_3_3_3_67_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2016.2613641"},{"key":"e_1_3_3_3_68_1","unstructured":"Tom Wheeler Rosenworcel Clyburn Pai and O\u2019Rielly. 2014. Federal Communications Commission FCC 14-12 before the federal...https:\/\/docs.fcc.gov\/public\/attachments\/fcc-14-12a1.pdf"},{"key":"e_1_3_3_3_69_1","doi-asserted-by":"publisher","DOI":"10.1145\/1978942.1978963"},{"key":"e_1_3_3_3_70_1","unstructured":"YouTube. 2023. YouTube for Press. https:\/\/blog.youtube\/press\/"},{"key":"e_1_3_3_3_71_1","volume-title":"Reading Sounds","author":"Zdenek Sean","unstructured":"Sean Zdenek. 2015. Reading sounds. In Reading Sounds. University of Chicago Press."}],"event":{"name":"CHI '24: CHI Conference on Human Factors in Computing Systems","location":"Honolulu HI USA","acronym":"CHI '24","sponsor":["SIGCHI ACM Special Interest Group on Computer-Human Interaction","SIGACCESS ACM Special Interest Group on Accessible Computing"]},"container-title":["Proceedings of the CHI Conference on Human Factors in Computing Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3613904.3642162","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3613904.3642162","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T23:56:42Z","timestamp":1750291002000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3613904.3642162"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,5,11]]},"references-count":69,"alternative-id":["10.1145\/3613904.3642162","10.1145\/3613904"],"URL":"https:\/\/doi.org\/10.1145\/3613904.3642162","relation":{},"subject":[],"published":{"date-parts":[[2024,5,11]]},"assertion":[{"value":"2024-05-11","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}