{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,12]],"date-time":"2026-06-12T23:43:00Z","timestamp":1781307780507,"version":"3.54.1"},"reference-count":41,"publisher":"IEEE","license":[{"start":{"date-parts":[[2023,10,22]],"date-time":"2023-10-22T00:00:00Z","timestamp":1697932800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,10,22]],"date-time":"2023-10-22T00:00:00Z","timestamp":1697932800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023,10,22]]},"DOI":"10.1109\/waspaa58266.2023.10248189","type":"proceedings-article","created":{"date-parts":[[2023,9,15]],"date-time":"2023-09-15T17:31:37Z","timestamp":1694799097000},"page":"1-5","source":"Crossref","is-referenced-by-count":15,"title":["Yet Another Generative Model for Room Impulse Response Estimation"],"prefix":"10.1109","author":[{"given":"Sungho","family":"Lee","sequence":"first","affiliation":[{"name":"Department of Intelligence and Information"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hyeong-Seok","family":"Choi","sequence":"additional","affiliation":[{"name":"Supertone, Inc"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kyogu","family":"Lee","sequence":"additional","affiliation":[{"name":"Department of Intelligence and Information"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref13","article-title":"Neural codec language models are zero-shot text to speech synthesizers","author":"wang","year":"2023"},{"key":"ref35","article-title":"CSTR VCTK corpus: English multi-speaker corpus for CSTR voice cloning toolkit","author":"veaux","year":"2017"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01123"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/APSIPA.2013.6694316"},{"key":"ref15","article-title":"Neural discrete representation learning","volume":"30","author":"van den oord","year":"2017","journal-title":"Adv in NeurIPS"},{"key":"ref37","year":"2014","journal-title":"Method for the Subjective Assessment of Intermediate Quality Level of Audio Systems"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/TSP52935.2021.9522648"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1992.226080"},{"key":"ref31","first-page":"882","article-title":"A multi-angle, multi-distance dataset of microphone impulse responses","volume":"70","author":"hern\u00e1ndez","year":"2022","journal-title":"JAES"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/ICDSP.2009.5201259"},{"key":"ref11","article-title":"Audiolm: a language modeling approach to audio generation","author":"borsos","year":"2022"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2014.2379648"},{"key":"ref10","article-title":"Attention is all you need","author":"vaswani","year":"2017","journal-title":"NeurIPS"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9052970"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/WASPAA52581.2021.9632680"},{"key":"ref1","article-title":"Acoustics &#x2014; Measurements of room acoustic parameters &#x2014; Part 2: Reverberation time in ordinary rooms","year":"2008"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1982.1171604"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.3390\/app7050483"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01268"},{"key":"ref38","article-title":"Decoupled weight decay regularization","author":"loshchilov","year":"2017"},{"key":"ref19","article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","author":"dosovitskiy","year":"2021","journal-title":"ICLRE"},{"key":"ref18","article-title":"High fidelity neural audio compression","author":"d\u00e9fossez","year":"2022"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747603"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2019.2917582"},{"key":"ref26","article-title":"Open database of spatial room impulse responses at detmold university of music","volume":"149","author":"amengual gari","year":"2020","journal-title":"AES Conv"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2010.5496083"},{"key":"ref20","first-page":"10 524","article-title":"On layer normalization in the transformer architecture","author":"xiong","year":"2020","journal-title":"ICML"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414038"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746610"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/TASSP.1984.1164317"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1186\/s13636-023-00284-9"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.3390\/acoustics4030047"},{"key":"ref29","first-page":"225","article-title":"Sound scene data collection in real acoustical environments","volume":"20","author":"nakamura","year":"1999","journal-title":"ASJ"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9415122"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2016.2577502"},{"key":"ref9","first-page":"495","article-title":"Soundstream: An end-to-end neural audio codec","volume":"30","author":"zeghidour","year":"2021","journal-title":"IEEE\/ACM TASLP"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10094770"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2022.3193298"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.1612524113"},{"key":"ref5","first-page":"1270","article-title":"Openair: An interactive auralization web resource and database","volume":"2","author":"shelley","year":"2010","journal-title":"129th AES Conv"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054701"}],"event":{"name":"2023 IEEE Workshop on Applications of Signal Processing to Audio and Acoustics (WASPAA)","location":"New Paltz, NY, USA","start":{"date-parts":[[2023,10,22]]},"end":{"date-parts":[[2023,10,25]]}},"container-title":["2023 IEEE Workshop on Applications of Signal Processing to Audio and Acoustics (WASPAA)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/10248019\/10248047\/10248189.pdf?arnumber=10248189","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,10,2]],"date-time":"2023-10-02T17:41:08Z","timestamp":1696268468000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10248189\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,22]]},"references-count":41,"URL":"https:\/\/doi.org\/10.1109\/waspaa58266.2023.10248189","relation":{},"subject":[],"published":{"date-parts":[[2023,10,22]]}}}