{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,20]],"date-time":"2026-04-20T12:53:16Z","timestamp":1776689596010,"version":"3.51.2"},"publisher-location":"New York, NY, USA","reference-count":59,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,3,27]],"date-time":"2023-03-27T00:00:00Z","timestamp":1679875200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-sa\/4.0\/"}],"funder":[{"DOI":"10.13039\/100009950","name":"Ministry of Education - Singapore","doi-asserted-by":"publisher","award":["Learning Generative Recurrent Neural Networks"],"award-info":[{"award-number":["Learning Generative Recurrent Neural Networks"]}],"id":[{"id":"10.13039\/100009950","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,3,27]]},"DOI":"10.1145\/3581641.3584083","type":"proceedings-article","created":{"date-parts":[[2023,3,27]],"date-time":"2023-03-27T16:16:52Z","timestamp":1679933812000},"page":"621-632","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":5,"title":["Evaluating Descriptive Quality of AI-Generated Audio Using Image-Schemas"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0351-6574","authenticated-orcid":false,"given":"Purnima","family":"Kamath","sequence":"first","affiliation":[{"name":"Augmented Human Lab, National University of Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0295-0897","authenticated-orcid":false,"given":"Zhuoyao","family":"Li","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1350-9095","authenticated-orcid":false,"given":"Chitralekha","family":"Gupta","sequence":"additional","affiliation":[{"name":"Augmented Human Lab, National University of Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8127-1157","authenticated-orcid":false,"given":"Kokil","family":"Jaidka","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7441-5493","authenticated-orcid":false,"given":"Suranga","family":"Nanayakkara","sequence":"additional","affiliation":[{"name":"Augmented Human Lab, National University of Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9200-1048","authenticated-orcid":false,"given":"Lonce","family":"Wyse","sequence":"additional","affiliation":[{"name":"Music Technology Group, University Pompeu Fabra, Spain"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,3,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682598"},{"key":"e_1_3_2_1_2_1","unstructured":"Shane Barratt and Rishi Sharma. 2018. A Note on the Inception Score. ArXiv abs\/1801.01973(2018)."},{"key":"e_1_3_2_1_3_1","volume-title":"Demystifying MMD GANs. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=r1lUOzWCW","author":"Bi\u0144kowski Miko\u0142aj","year":"2018","unstructured":"Miko\u0142aj Bi\u0144kowski, Danica\u00a0J Sutherland, Michael Arbel, and Arthur Gretton. 2018. Demystifying MMD GANs. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=r1lUOzWCW"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.cviu.2018.10.009"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/2858036.2858198"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.2307\/3090681"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300522"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462153"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7471749"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3134664"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3491102.3501819"},{"key":"e_1_3_2_1_12_1","volume-title":"jsPsych: A JavaScript library for creating behavioral experiments in a Web browser. Behavior research methods 47, 1","author":"De\u00a0Leeuw R","year":"2015","unstructured":"Joshua\u00a0R De\u00a0Leeuw. 2015. jsPsych: A JavaScript library for creating behavioral experiments in a Web browser. Behavior research methods 47, 1 (2015), 1\u201312."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300599"},{"key":"e_1_3_2_1_14_1","volume-title":"Adversarial Audio Synthesis. In International Conference on Learning Representations. ICLR","author":"Donahue Chris","year":"2019","unstructured":"Chris Donahue, Julian McAuley, and Miller Puckette. 2019. Adversarial Audio Synthesis. In International Conference on Learning Representations. ICLR, New Orleans, Louisiana, United States. https:\/\/openreview.net\/forum?id=ByMVTsR5KQ"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"crossref","unstructured":"Peter\u00a0W Donhauser and Denise Klein. 2022. Audio-Tokens: a toolbox for rating sorting and comparing audio samples in the browser. Behavior research methods(2022).","DOI":"10.31234\/osf.io\/3j58q"},{"key":"e_1_3_2_1_16_1","volume-title":"GANSynth: Adversarial Neural Audio Synthesis. In International Conference on Learning Representations. ICLR","author":"Engel Jesse","year":"2019","unstructured":"Jesse Engel, Kumar\u00a0Krishna Agrawal, Shuo Chen, Ishaan Gulrajani, Chris Donahue, and Adam Roberts. 2019. GANSynth: Adversarial Neural Audio Synthesis. In International Conference on Learning Representations. ICLR, New Orleans, Louisiana, United States. https:\/\/openreview.net\/forum?id=H1xQVn09FX"},{"key":"e_1_3_2_1_17_1","volume-title":"Proceedings of the 34th International Conference on Machine Learning -","volume":"1077","author":"Engel Jesse","year":"2017","unstructured":"Jesse Engel, Cinjon Resnick, Adam Roberts, Sander Dieleman, Mohammad Norouzi, Douglas Eck, and Karen Simonyan. 2017. Neural Audio Synthesis of Musical Notes with WaveNet Autoencoders. In Proceedings of the 34th International Conference on Machine Learning - Volume 70(ICML\u201917). JMLR.org, Sydney, NSW, Australia, 1068\u20131077."},{"key":"e_1_3_2_1_18_1","unstructured":"Philippe Esling Adrien Bitton 2018. Generative timbre spaces: regularizing variational auto-encoders with perceptual metrics. arXiv preprint arXiv:1805.08501(2018)."},{"key":"e_1_3_2_1_19_1","volume-title":"Bridging Audio Analysis, Perception and Synthesis with Perceptually-regularized Variational Timbre Spaces","author":"Esling Philippe","unstructured":"Philippe Esling, Axel Chemla-Romeu-Santos, and Adrien Bitton. 2018. Bridging Audio Analysis, Perception and Synthesis with Perceptually-regularized Variational Timbre Spaces. In International Society for Music Information Retrieval. ISMIR, Paris, France, 175\u2013181."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3078714.3078715"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1207\/s15326969eco0501_1"},{"key":"e_1_3_2_1_22_1","volume-title":"NIPS 2016 Tutorial: Generative Adversarial Networks. ArXiv abs\/1701","author":"Goodfellow Ian","year":"2017","unstructured":"Ian Goodfellow. 2017. NIPS 2016 Tutorial: Generative Adversarial Networks. ArXiv abs\/1701.00160(2017)."},{"key":"e_1_3_2_1_23_1","unstructured":"Chitralekha Gupta Yize Wei Zequn Gong Purnima Kamath Zhuoyao Li and Lonce Wyse. 2022. Parameter Sensitivity of Deep-Feature based Evaluation Metrics for Audio Textures. In ISMIR."},{"key":"e_1_3_2_1_24_1","volume-title":"Advances in Neural Information Processing Systems, I.\u00a0Guyon, U.\u00a0Von Luxburg, S.\u00a0Bengio, H.\u00a0Wallach, R.\u00a0Fergus, S.\u00a0Vishwanathan, and R.\u00a0Garnett (Eds.). Vol.\u00a030. Curran Associates","author":"Heusel Martin","year":"2017","unstructured":"Martin Heusel, Hubert Ramsauer, Thomas Unterthiner, Bernhard Nessler, and Sepp Hochreiter. 2017. GANs Trained by a Two Time-Scale Update Rule Converge to a Local Nash Equilibrium. In Advances in Neural Information Processing Systems, I.\u00a0Guyon, U.\u00a0Von Luxburg, S.\u00a0Bengio, H.\u00a0Wallach, R.\u00a0Fergus, S.\u00a0Vishwanathan, and R.\u00a0Garnett (Eds.). Vol.\u00a030. Curran Associates, Inc., Long Beach, USA. https:\/\/proceedings.neurips.cc\/paper\/2017\/file\/8a1d694707eb0fefe65871369074926d-Paper.pdf"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/3411763.3443447"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2006.883259"},{"key":"e_1_3_2_1_27_1","volume-title":"Demographics of mechanical turk. CeDER Working Papers 10, 1","author":"Ipeirotis G","year":"2010","unstructured":"Panagiotis\u00a0G Ipeirotis. 2010. Demographics of mechanical turk. CeDER Working Papers 10, 1 (2010). http:\/\/hdl.handle.net\/2451\/29585"},{"key":"e_1_3_2_1_28_1","unstructured":"ITU. 2014. Recommendation ITU-R BS.1534-2: Method for the subjective assessment of intermediate quality level of audio systems. In ITU BS Series. Radiocommunication sector of ITU Int. 1\u201336."},{"key":"e_1_3_2_1_29_1","volume-title":"Web Audio Conference. Georgia Tech","author":"Jillings Nicholas","year":"2016","unstructured":"Nicholas Jillings, Brecht De\u00a0Man, David Moffat, and Joshua Reiss. 2016. Web Audio Evaluation Tool: A framework for subjective assessment of audio. In Web Audio Conference. Georgia Tech, Atlanta, USA."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","unstructured":"Mark Johnson. 2005. The philosophical significance of image schemas. 15\u201334. https:\/\/doi.org\/10.1515\/9783110197532.1.15","DOI":"10.1515\/9783110197532.1.15"},{"key":"e_1_3_2_1_31_1","volume-title":"The body in the mind: The bodily basis of meaning, imagination, and reason","author":"Johnson Mark","unstructured":"Mark Johnson. 2013. The body in the mind: The bodily basis of meaning, imagination, and reason. University of Chicago press."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/3308560.3317081"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2219"},{"key":"e_1_3_2_1_34_1","volume-title":"Metaphors we live by","author":"Lakoff George","unstructured":"George Lakoff and Mark Johnson. 2008. Metaphors we live by. University of Chicago press."},{"key":"e_1_3_2_1_35_1","volume-title":"Crowdsourcing Music Similarity Judgments using Mechanical Turk","author":"Lee Jin\u00a0Ha","unstructured":"Jin\u00a0Ha Lee. 2010. Crowdsourcing Music Similarity Judgments using Mechanical Turk.. In International Society for Music Information Retrieval. ISMIR, Utrecht, Netherlands, 183\u2013188."},{"key":"e_1_3_2_1_36_1","volume-title":"Learning disentangled representations of timbre and pitch for musical instrument sounds using gaussian mixture variational autoencoders","author":"Luo Yin-Jyun","year":"2019","unstructured":"Yin-Jyun Luo, Kat Agres, and Dorien Herremans. 2019. Learning disentangled representations of timbre and pitch for musical instrument sounds using gaussian mixture variational autoencoders. International Society of Music Information Retrieval (ISMIR) (2019)."},{"key":"e_1_3_2_1_37_1","volume-title":"Learning tags that vary within a song","author":"Mandel I","unstructured":"Michael\u00a0I Mandel, Douglas Eck, and Yoshua Bengio. 2010. Learning tags that vary within a song. In International Society for Music Information Retrieval. ISMIR, Utrecht, Netherlands, 399\u2013404."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/2531602.2531663"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neuron.2011.06.032"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.cub.2018.03.049"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3290607.3313054"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.5555\/3524938.3525603"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/3465336.3475109"},{"key":"e_1_3_2_1_44_1","volume-title":"Proceedings of the 12th International Conference on Music Perception and Cognition. School of Music Studies","author":"Oh Jieun","year":"2012","unstructured":"Jieun Oh and Ge Wang. 2012. Evaluating crowdsourcing through amazon mechanical turk as a technique for conducting music perception experiments. In Proceedings of the 12th International Conference on Music Perception and Cognition. School of Music Studies, Aristotle University of Thessaloniki, Greece, 1\u20136."},{"key":"e_1_3_2_1_45_1","unstructured":"Brian O\u2019Reilly. 2008. Brian O\u2019Reilly\u2019s Electroacoustic compositions and noise music. https:\/\/vimeo.com\/dendriform"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.5555\/285582.285601"},{"key":"e_1_3_2_1_47_1","volume-title":"Advances in Neural Information Processing Systems, S.\u00a0Bengio, H.\u00a0Wallach, H.\u00a0Larochelle, K.\u00a0Grauman, N.\u00a0Cesa-Bianchi, and R.\u00a0Garnett (Eds.). Vol.\u00a031. Curran Associates","author":"Sajjadi Mehdi","year":"2018","unstructured":"Mehdi S.\u00a0M. Sajjadi, Olivier Bachem, Mario Lucic, Olivier Bousquet, and Sylvain Gelly. 2018. Assessing Generative Models via Precision and Recall. In Advances in Neural Information Processing Systems, S.\u00a0Bengio, H.\u00a0Wallach, H.\u00a0Larochelle, K.\u00a0Grauman, N.\u00a0Cesa-Bianchi, and R.\u00a0Garnett (Eds.). Vol.\u00a031. Curran Associates, Inc., Montreal, Canada. https:\/\/proceedings.neurips.cc\/paper\/2018\/file\/f7696a9b362ac5a51c3dc8f098b73923-Paper.pdf"},{"key":"e_1_3_2_1_48_1","volume-title":"Advances in Neural Information Processing Systems, D.\u00a0Lee, M.\u00a0Sugiyama, U.\u00a0Luxburg, I.\u00a0Guyon, and R.\u00a0Garnett (Eds.). Vol.\u00a029. Curran Associates","author":"Salimans Tim","year":"2016","unstructured":"Tim Salimans, Ian Goodfellow, Wojciech Zaremba, Vicki Cheung, Alec Radford, Xi Chen, and Xi Chen. 2016. Improved Techniques for Training GANs. In Advances in Neural Information Processing Systems, D.\u00a0Lee, M.\u00a0Sugiyama, U.\u00a0Luxburg, I.\u00a0Guyon, and R.\u00a0Garnett (Eds.). Vol.\u00a029. Curran Associates, Inc., Barcelona, Spain. https:\/\/proceedings.neurips.cc\/paper\/2016\/file\/8a3363abe792db2d8761d6403605aeb7-Paper.pdf"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1609\/hcomp.v9i1.18944"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.5334\/jors.187"},{"key":"e_1_3_2_1_51_1","volume-title":"ISMIR, Vol.\u00a0104. Citeseer","author":"Speck A","unstructured":"Jacquelin\u00a0A Speck, Erik\u00a0M Schmidt, Brandon\u00a0G Morton, and Youngmoo\u00a0E Kim. 2011. A Comparative Study of Collaborative vs. Traditional Musical Mood Annotation.. In ISMIR, Vol.\u00a0104. Citeseer, ISMIR, Miami, USA, 549\u2013554."},{"key":"e_1_3_2_1_52_1","first-page":"3","article-title":"PEAQ-The ITU standard for objective measurement of perceived audio quality","volume":"48","author":"Thiede Thilo","year":"2000","unstructured":"Thilo Thiede, William\u00a0C Treurniet, Roland Bitto, Christian Schmidmer, Thomas Sporer, John\u00a0G Beerends, and Catherine Colomes. 2000. PEAQ-The ITU standard for objective measurement of perceived audio quality. Journal of the Audio Engineering Society 48, 1\/2 (2000), 3\u201329.","journal-title":"Journal of the Audio Engineering Society"},{"key":"e_1_3_2_1_53_1","volume-title":"Crowdsourcing Preference Judgments for Evaluation of Music Similarity Tasks. ACM SIGIR Workshop on Crowdsourcing for Search Evaluation (01","author":"Urbano Juli\u00e1n","year":"2010","unstructured":"Juli\u00e1n Urbano, Jorge Morato, M\u00f3nica Marrero, and Diego Mart\u00edn. 2010. Crowdsourcing Preference Judgments for Evaluation of Music Similarity Tasks. ACM SIGIR Workshop on Crowdsourcing for Search Evaluation (01 2010), 9\u201316."},{"key":"e_1_3_2_1_54_1","unstructured":"A\u00e4ron van\u00a0den Oord Sander Dieleman Heiga Zen Karen Simonyan Oriol Vinyals Alexander Graves Nal Kalchbrenner Andrew Senior and Koray Kavukcuoglu. 2016. WaveNet: A Generative Model for Raw Audio. In Arxiv. Arxiv Virtual. https:\/\/arxiv.org\/abs\/1609.03499"},{"key":"e_1_3_2_1_55_1","volume-title":"What can the language of musicians tell us about music interaction design?Computer Music Journal 34, 4","author":"Wilkie Katie","year":"2010","unstructured":"Katie Wilkie, Simon Holland, and Paul Mulholland. 2010. What can the language of musicians tell us about music interaction design?Computer Music Journal 34, 4 (2010), 34\u201348."},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1609\/hcomp.v5i1.13317"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-03789-4_20"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.5281\/zenodo.1178191"},{"key":"e_1_3_2_1_59_1","volume-title":"Syntex: parametric audio texture datasets for conditional training of instrumental interfaces.International Conference on New Interfaces for Musical Expression (16 4","author":"Wyse Lonce","year":"2022","unstructured":"Lonce Wyse and Prashanth\u00a0Thattai Ravikumar. 2022. Syntex: parametric audio texture datasets for conditional training of instrumental interfaces.International Conference on New Interfaces for Musical Expression (16 4 2022). https:\/\/nime.pubpub.org\/pub\/0nl57935 https:\/\/nime.pubpub.org\/pub\/0nl57935."}],"event":{"name":"IUI '23: 28th International Conference on Intelligent User Interfaces","location":"Sydney NSW Australia","acronym":"IUI '23","sponsor":["SIGAI ACM Special Interest Group on Artificial Intelligence","SIGCHI ACM Special Interest Group on Computer-Human Interaction"]},"container-title":["Proceedings of the 28th International Conference on Intelligent User Interfaces"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581641.3584083","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3581641.3584083","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T16:36:21Z","timestamp":1750178181000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581641.3584083"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,3,27]]},"references-count":59,"alternative-id":["10.1145\/3581641.3584083","10.1145\/3581641"],"URL":"https:\/\/doi.org\/10.1145\/3581641.3584083","relation":{},"subject":[],"published":{"date-parts":[[2023,3,27]]},"assertion":[{"value":"2023-03-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}