{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,13]],"date-time":"2025-12-13T07:23:08Z","timestamp":1765610588681,"version":"3.48.0"},"publisher-location":"New York, NY, USA","reference-count":43,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,6,30]]},"DOI":"10.1145\/3771594.3771597","type":"proceedings-article","created":{"date-parts":[[2025,12,13]],"date-time":"2025-12-13T06:52:24Z","timestamp":1765608744000},"page":"17-28","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Sonic Scribbles \u2013 Constructing Sketch Classes from Visual Associations of the Mental Model for Audio"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9268-4854","authenticated-orcid":false,"given":"Lars","family":"Engeln","sequence":"first","affiliation":[{"name":"Technische Universit\u00e4t Dresden, Dresden, Germany"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-1116-8032","authenticated-orcid":false,"given":"Rainer","family":"Groh","sequence":"additional","affiliation":[{"name":"Technische Universit\u00e4t Dresden, Dresden, Germany"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,12,12]]},"reference":[{"key":"e_1_3_3_2_2_2","doi-asserted-by":"publisher","unstructured":"Mohammad Adeli Jean Rouat and St\u00e9phane Molotchnikoff. 2014. Audiovisual correspondence between musical timbre and visual shapes. Frontiers in Human Neuroscience 8 (May 2014). 10.3389\/fnhum.2014.00352","DOI":"10.3389\/fnhum.2014.00352"},{"key":"e_1_3_3_2_3_2","first-page":"122","volume-title":"ISMIR","author":"Andersen Kristina","year":"2016","unstructured":"Kristina Andersen and Peter Knees. 2016. Conversations with Expert Users in Music Retrieval and Research Challenges for Creative MIR.. In ISMIR. 122\u2013128. https:\/\/research.tue.nl\/en\/publications\/conversations-with-expert-users-in-music-retrieval-and-research-c"},{"key":"e_1_3_3_2_4_2","doi-asserted-by":"crossref","unstructured":"Andrey Anikin and Niklas Johansson. 2019. Implicit associations between individual properties of color and sound. Attention Perception & Psychophysics 81 (2019) 764\u2013777.","DOI":"10.3758\/s13414-018-01639-7"},{"key":"e_1_3_3_2_5_2","doi-asserted-by":"publisher","unstructured":"Tadas Baltru\u0161aitis Chaitanya Ahuja and Louis-Philippe Morency. 2019. Multimodal Machine Learning: A Survey and Taxonomy. IEEE Transactions on Pattern Analysis and Machine Intelligence 41 2 (2019) 423\u2013443. 10.1109\/TPAMI.2018.2798607","DOI":"10.1109\/TPAMI.2018.2798607"},{"key":"e_1_3_3_2_6_2","doi-asserted-by":"crossref","unstructured":"Marcia\u00a0J Bates. 1989. The design of browsing and berrypicking techniques for the online search interface. Online review 13 5 (1989) 407\u2013424.","DOI":"10.1108\/eb024320"},{"key":"e_1_3_3_2_7_2","volume-title":"Britto A, Gouyon F, Dixon S, editors. 14th Conference of the International Society for Music Information Retrieval (ISMIR); 2013 Nov 4-8; Curitiba, Brazil.[place unknown]: ISMIR; 2013. p. 493-8.","author":"Bogdanov Dmitry","year":"2013","unstructured":"Dmitry Bogdanov, Nicolas Wack, Emilia G\u00f3mez\u00a0Guti\u00e9rrez, Sankalp Gulati, Herrera Boyer, Oscar Mayor, Gerard Roma\u00a0Trepat, Justin Salamon, Jos\u00e9\u00a0Ricardo Zapata\u00a0Gonz\u00e1lez, Xavier Serra, et\u00a0al. 2013. Essentia: An audio analysis library for music information retrieval. In Britto A, Gouyon F, Dixon S, editors. 14th Conference of the International Society for Music Information Retrieval (ISMIR); 2013 Nov 4-8; Curitiba, Brazil.[place unknown]: ISMIR; 2013. p. 493-8. International Society for Music Information Retrieval (ISMIR)."},{"key":"e_1_3_3_2_8_2","doi-asserted-by":"crossref","unstructured":"Jos\u00e9\u00a0Luis Caivano. 1994. Color and sound: Physical and psychophysical relations. Color Research & Application 19 2 (1994) 126\u2013133.","DOI":"10.1111\/j.1520-6378.1994.tb00072.x"},{"key":"e_1_3_3_2_9_2","doi-asserted-by":"crossref","unstructured":"Gemma\u00a0A Calvert Peter\u00a0C Hansen Susan\u00a0D Iversen and Michael\u00a0J Brammer. 2001. Detection of audio-visual integration sites in humans by application of electrophysiological criteria to the BOLD effect. Neuroimage 14 2 (2001) 427\u2013438.","DOI":"10.1006\/nimg.2001.0812"},{"key":"e_1_3_3_2_10_2","doi-asserted-by":"publisher","unstructured":"Charles\u00a0P. Davis Hannah\u00a0M. Morrow and Gary Lupyan. 2019. What Does a Horgous Look Like? Nonsense Words Elicit Meaningful Drawings. Cognitive Science 43 10 (oct 2019). 10.1111\/cogs.12791","DOI":"10.1111\/cogs.12791"},{"key":"e_1_3_3_2_11_2","doi-asserted-by":"crossref","unstructured":"Marie Delacre Dani\u00ebl Lakens and Christophe Leys. 2017. Why psychologists should by default use Welch\u2019s t-test instead of Student\u2019s t-test. International Review of Social Psychology 30 1 (2017).","DOI":"10.5334\/irsp.82"},{"key":"e_1_3_3_2_12_2","unstructured":"Lars Engeln. 2023. Die Verbildlichung von Klangstrukturen im Kontext der Entwicklung von Werkzeugen f\u00fcr die Medienproduktion. Ph.\u00a0D. Dissertation. Technische Universit\u00e4t Dresden. https:\/\/nbn-resolving.org\/urn:nbn:de:bsz:14-qucosa2-872048"},{"key":"e_1_3_3_2_13_2","doi-asserted-by":"publisher","DOI":"10.1145\/3356590.3356606"},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"publisher","unstructured":"Lars Engeln and Rainer Groh. 2020. CoHEARence of audible shapes\u2014a qualitative user study for coherent visual audio design with resynthesized shapes. Personal and Ubiquitous Computing (2020) 1\u201311. 10.1007\/s00779-020-01392-5","DOI":"10.1007\/s00779-020-01392-5"},{"key":"e_1_3_3_2_15_2","doi-asserted-by":"publisher","DOI":"10.1145\/3478384.3478423"},{"key":"e_1_3_3_2_16_2","doi-asserted-by":"publisher","unstructured":"K.\u00a0K. Evans and A. Treisman. 2010. Natural cross-modal mappings between visual and auditory features. Journal of Vision 10 1 (jan 2010) 6\u20136. 10.1167\/10.1.6","DOI":"10.1167\/10.1.6"},{"key":"e_1_3_3_2_17_2","doi-asserted-by":"publisher","unstructured":"Eduardo Fonseca Manoj Plakal Frederic Font Daniel P.\u00a0W. Ellis Xavier Favory Jordi Pons and Xavier Serra. 2018. General-purpose Tagging of Freesound Audio with AudioSet Labels: Task Description Dataset and Baseline. 10.48550\/ARXIV.1807.09902","DOI":"10.48550\/ARXIV.1807.09902"},{"key":"e_1_3_3_2_18_2","doi-asserted-by":"publisher","DOI":"10.1145\/2502081.2502245"},{"key":"e_1_3_3_2_19_2","unstructured":"Konstantinos Giannakis. 2001. Sound Mosaics A Graphical User Interface For Sound Synthesis Based On Auditory-Visual Associations. (2001)."},{"key":"e_1_3_3_2_20_2","doi-asserted-by":"crossref","unstructured":"Kostas Giannakis. 2006. A comparative evaluation of auditory-visual mappings for sound visualisation. Organised Sound 11 3 (2006) 297\u2013307.","DOI":"10.1017\/S1355771806001531"},{"key":"e_1_3_3_2_21_2","unstructured":"Kostas Giannakis and Matt Smith. 2001. Imaging soundscapes: Identifying cognitive associations between auditory and visual dimensions. Musical Imagery (2001) 161\u2013179."},{"key":"e_1_3_3_2_22_2","unstructured":"Rainer Groh. 2017. An Iconography of Interaction."},{"key":"e_1_3_3_2_23_2","volume-title":"Workshop Farbbildverarbeitung (FarbBV2005), Gesellschaft zur F\u00f6rderung angewandter Informatik eV (GFaI), Arbeitsgemeinschaft industrieller Forschungsvereinigung \u201eOtto von Guericke \u201ceV (AiF OvG), Berlin","author":"Groh Rainer","year":"2005","unstructured":"Rainer Groh and Ingmar\u00a0S Franke. 2005. Farbperspektive im Kontext von Navigation durch virtuelle Welten, Artikel zu den theoretischen Grundlagen der Interfacegestaltung-11. In Workshop Farbbildverarbeitung (FarbBV2005), Gesellschaft zur F\u00f6rderung angewandter Informatik eV (GFaI), Arbeitsgemeinschaft industrieller Forschungsvereinigung \u201eOtto von Guericke \u201ceV (AiF OvG), Berlin, Vol.\u00a06."},{"key":"e_1_3_3_2_24_2","doi-asserted-by":"publisher","DOI":"10.1145\/3531073.3534491"},{"key":"e_1_3_3_2_25_2","doi-asserted-by":"publisher","DOI":"10.1145\/2911996.2912021"},{"key":"e_1_3_3_2_26_2","doi-asserted-by":"publisher","unstructured":"Peter Knees and Markus Schedl. 2013. A Survey of Music Similarity and Recommendation from Music Context Data. ACM Trans. Multimedia Comput. Commun. Appl. 10 1 Article 2 (dec 2013) 21\u00a0pages. 10.1145\/2542205.2542206","DOI":"10.1145\/2542205.2542206"},{"key":"e_1_3_3_2_27_2","doi-asserted-by":"crossref","unstructured":"Pengfei Li Yin Zhang and Bin Zhang. 2022. Understanding query combination behavior in exploratory searches. Applied Sciences 12 2 (2022) 706.","DOI":"10.3390\/app12020706"},{"key":"e_1_3_3_2_28_2","unstructured":"Sebastian L\u00f6bbers and George Fazekas. 2021. Sketching Sounds: Using sound-shape associations to build a sketch-based sound synthesiser. Digital Music Research Network (DMRN+ 16) (2021)."},{"key":"e_1_3_3_2_29_2","unstructured":"Sebastian L\u00f6bbers and Gy\u00f6rgy Fazekas. 2022. Seeing Sounds Hearing Shapes: a gamified study to evaluate sound-sketches. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2205.08866 (2022)."},{"key":"e_1_3_3_2_30_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-29956-8_11"},{"key":"e_1_3_3_2_31_2","doi-asserted-by":"publisher","DOI":"10.1145\/3356590.3356634"},{"key":"e_1_3_3_2_32_2","doi-asserted-by":"publisher","DOI":"10.1201\/b17511"},{"key":"e_1_3_3_2_33_2","doi-asserted-by":"publisher","DOI":"10.1145\/223904.223911"},{"key":"e_1_3_3_2_34_2","unstructured":"Vilayanur\u00a0S Ramachandran and Edward\u00a0M Hubbard. 2001. Synaesthesia\u2013a window into perception thought and language. Journal of consciousness studies 8 12 (2001) 3\u201334."},{"key":"e_1_3_3_2_35_2","doi-asserted-by":"crossref","unstructured":"Reijo Savolainen. 2018. Berrypicking and information foraging: Comparison of two theoretical frameworks for studying exploratory search. Journal of Information Science 44 5 (2018) 580\u2013593.","DOI":"10.1177\/0165551517713168"},{"key":"e_1_3_3_2_36_2","doi-asserted-by":"crossref","unstructured":"Daniele Sch\u00f6n S\u00f8lvi Ystad Richard Kronland-Martinet and Mireille Besson. 2010. The evocative power of sounds: Conceptual priming between words and nonverbal sounds. Journal of cognitive neuroscience 22 5 (2010) 1026\u20131035.","DOI":"10.1162\/jocn.2009.21302"},{"key":"e_1_3_3_2_37_2","doi-asserted-by":"crossref","unstructured":"Charles\u00a0E Schroeder and John Foxe. 2005. Multisensory contributions to low-level unisensory processing. Current opinion in neurobiology 15 4 (2005) 454\u2013458.","DOI":"10.1016\/j.conb.2005.06.008"},{"key":"e_1_3_3_2_38_2","first-page":"183","volume-title":"Audio Mostly Conference","author":"Sei\u00e7a Mariana","year":"2020","unstructured":"Mariana Sei\u00e7a, Lic\u00ednio Roque, Pedro Martins, and F\u00a0Am\u00edlcar Cardoso. 2020. Contrasts and similarities between two audio research communities in evaluating auditory artefacts.. In Audio Mostly Conference. 183\u2013190."},{"key":"e_1_3_3_2_39_2","doi-asserted-by":"publisher","unstructured":"Carlos Velasco Xiaoang Wan Klemens Knoeferle Xi Zhou Alejandro Salgado-Montejo and Charles Spence. 2015. Searching for flavor labels in food products: the influence of color-flavor congruence and association strength. Frontiers in Psychology 6 (2015). 10.3389\/fpsyg.2015.00301","DOI":"10.3389\/fpsyg.2015.00301"},{"key":"e_1_3_3_2_40_2","doi-asserted-by":"publisher","unstructured":"Carlos Velasco Andy\u00a0T. Woods Ophelia Deroy and Charles Spence. 2015. Hedonic mediation of the crossmodal correspondence between taste and shape. Food Quality and Preference 41 (2015) 151\u2013158. 10.1016\/j.foodqual.2014.11.010","DOI":"10.1016\/j.foodqual.2014.11.010"},{"key":"e_1_3_3_2_41_2","doi-asserted-by":"crossref","unstructured":"Shams Watkins Ladan Shams Sachiyo Tanaka J-D Haynes and Geraint Rees. 2006. Sound alters activity in human V1 in association with illusory visual perception. Neuroimage 31 3 (2006) 1247\u20131256.","DOI":"10.1016\/j.neuroimage.2006.01.016"},{"key":"e_1_3_3_2_42_2","doi-asserted-by":"publisher","unstructured":"Baixi Xing Kejun Zhang Lekai Zhang Xinda Wu Jian Dou and Shouqian Sun. 2019. Image\u2013Music Synesthesia-Aware Learning Based on Emotional Similarity Recognition. IEEE Access 7 (2019) 136378\u2013136390. 10.1109\/access.2019.2942073","DOI":"10.1109\/access.2019.2942073"},{"key":"e_1_3_3_2_43_2","doi-asserted-by":"crossref","unstructured":"Dongchao Yang Jianwei Yu Helin Wang Wen Wang Chao Weng Yuexian Zou and Dong Yu. 2023. Diffsound: Discrete diffusion model for text-to-sound generation. IEEE\/ACM Transactions on Audio Speech and Language Processing (2023).","DOI":"10.1109\/TASLP.2023.3268730"},{"key":"e_1_3_3_2_44_2","doi-asserted-by":"publisher","DOI":"10.1007\/3-540-46027-6"}],"event":{"name":"AM '25: 20th International Audio Mostly Conference","acronym":"AM '25","location":"Coimbra Portugal"},"container-title":["Proceedings of the 20th International Audio Mostly Conference"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3771594.3771597","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,13]],"date-time":"2025-12-13T07:17:45Z","timestamp":1765610265000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3771594.3771597"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,30]]},"references-count":43,"alternative-id":["10.1145\/3771594.3771597","10.1145\/3771594"],"URL":"https:\/\/doi.org\/10.1145\/3771594.3771597","relation":{},"subject":[],"published":{"date-parts":[[2025,6,30]]},"assertion":[{"value":"2025-12-12","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}