{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T16:54:54Z","timestamp":1772816094083,"version":"3.50.1"},"reference-count":37,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2025,12,11]],"date-time":"2025-12-11T00:00:00Z","timestamp":1765411200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2026,1,14]],"date-time":"2026-01-14T00:00:00Z","timestamp":1768348800000},"content-version":"vor","delay-in-days":34,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"funder":[{"DOI":"10.13039\/501100007601","name":"Horizon 2020","doi-asserted-by":"publisher","award":["101019375"],"award-info":[{"award-number":["101019375"]}],"id":[{"id":"10.13039\/501100007601","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J. Audio Speech Music Process."],"DOI":"10.1186\/s13636-025-00428-z","type":"journal-article","created":{"date-parts":[[2025,12,11]],"date-time":"2025-12-11T07:16:51Z","timestamp":1765437411000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Sound and music biases in deep music transcription models: a systematic analysis"],"prefix":"10.1186","volume":"2026","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0723-3010","authenticated-orcid":false,"given":"Luk\u00e1\u0161\u00a0Samuel","family":"Mart\u00e1k","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Patricia","family":"Hu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Gerhard","family":"Widmer","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,12,11]]},"reference":[{"key":"428_CR1","doi-asserted-by":"crossref","unstructured":"A. Ycart, L. Liu, E. Benetos, M. Pearce, Investigating the Perceptual Validity of Evaluation Metrics for Automatic Piano Music Transcription. Trans. Int. Soc. Music Inf. Retr.\u00a03(1), 68\u201381(2020). https:\/\/transactions.ismir.net\/articles\/10.5334\/tismir.57","DOI":"10.5334\/tismir.57"},{"key":"428_CR2","unstructured":"C. Hawthorne, A. Stasyuk, A. Roberts, I. Simon, C.A. Huang, S. Dieleman, E. Elsen, J.H. Engel, D. Eck, in 7th International Conference on Learning Representations, (ICLR 2019), New Orleans, LA, USA, May 6-9, 2019. Enabling Factorized Piano Music Modeling and Generation with the MAESTRO Dataset (OpenReview.net, 2019). https:\/\/openreview.net\/forum?id=r1lYRjC9F7\u00a0Accessed 4 Feb 2019"},{"key":"428_CR3","unstructured":"P. Hu, L.S. Mart\u00e1k, C. Cancino-Chac\u00f3n, G. Widmer, in Proceedings of the 25th International Society for Music Information Retrieval Conference, ISMIR 2024, San Francisco, CA, USA and online, November 10-14, 2024. Towards Musically Informed Evaluation of Piano Transcription Models,\u00a0International Society for Music Information Retrieval. (2024)"},{"key":"428_CR4","doi-asserted-by":"crossref","unstructured":"S. B\u00f6ck, M. Schedl, in 2012 IEEE international conference on acoustics, speech and signal processing (ICASSP). Polyphonic Piano Note Transcription with Recurrent Neural Networks (IEEE,\u00a0Piscataway, NJ, 2012), pp. 121\u2013124","DOI":"10.1109\/ICASSP.2012.6287832"},{"issue":"5","key":"428_CR5","doi-asserted-by":"publisher","first-page":"927","DOI":"10.1109\/TASLP.2016.2533858","volume":"24","author":"S Sigtia","year":"2016","unstructured":"S. Sigtia, E. Benetos, S. Dixon, An End-to-End Neural Network for Polyphonic Piano Music Transcription. IEEE ACM Trans. Audio Speech Lang. Process. 24(5), 927\u2013936 (2016). https:\/\/doi.org\/10.1109\/TASLP.2016.2533858","journal-title":"IEEE ACM Trans. Audio Speech Lang. Process."},{"key":"428_CR6","unstructured":"R. Kelz, M. Dorfer, F. Korzeniowski, S. B\u00f6ck, A. Arzt, G. Widmer, in Proceedings of the 17th International Society for Music Information Retrieval Conference, ISMIR 2016, New York City, United States, August 7-11, 2016. On the Potential of Simple Framewise Approaches to Piano Transcription,\u00a0International Society for Music Information Retrieval. (2016), pp. 475\u2013481"},{"key":"428_CR7","unstructured":"C. Hawthorne, E. Elsen, J. Song, A. Roberts, I. Simon, C. Raffel, J.H. Engel, S. Oore, D. Eck, in Proceedings of the 19th International Society for Music Information Retrieval Conference, ISMIR 2018, Paris, France, September 23-27, 2018, ed. by E. G\u00f3mez, X. Hu, E. Humphrey, E. Benetos. Onsets and Frames: Dual-Objective Piano Transcription (2018), pp. 50\u201357.\u00a0http:\/\/ismir2018.ircam.fr\/doc\/pdfs\/19_Paper.pdf\u00a0Accessed 15 Aug 2018"},{"key":"428_CR8","unstructured":"T. Kwon, D. Jeong, J. Nam, in The 21th International Society for Music Information Retrieval Conference (ISMIR). Polyphonic Piano Transcription Using Autoregressive Multi-State Note Model (International Society for Music Information Retrieval, 2020)"},{"key":"428_CR9","doi-asserted-by":"publisher","first-page":"3707","DOI":"10.1109\/TASLP.2021.3121991","volume":"29","author":"Q Kong","year":"2021","unstructured":"Q. Kong, B. Li, X. Song, Y. Wan, Y. Wang, High-resolution piano transcription with pedals by regressing onset and offset times. IEEE\/ACM Trans. Audio Speech Lang. Process. 29, 3707\u20133717 (2021). https:\/\/doi.org\/10.1109\/TASLP.2021.3121991","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"428_CR10","unstructured":"C. Hawthorne, I. Simon, R. Swavely, E. Manilow, J.H. Engel, in Proceedings of the 22nd International Society for Music Information Retrieval Conference, ISMIR 2021, Online, November 7-12, 2021, ed. by J.H. Lee, A. Lerch, Z. Duan, J. Nam, P. Rao, P. van Kranenburg, A. Srinivasamurthy. Sequence-to-Sequence Piano Transcription with Transformers (2021), pp. 246\u2013253. https:\/\/archives.ismir.net\/ismir2021\/paper\/000030.pdf\u00a0Accessed 23 Feb 2022"},{"key":"428_CR11","unstructured":"K. Toyama, T. Akama, Y. Ikemiya, Y. Takida, W. Liao, Y. Mitsufuji, in Ismir 2023 Hybrid Conference. Automatic Piano Transcription With Hierarchical Frequency-Time Transformer,\u00a0International Society for Music Information Retrieval. (2023)"},{"key":"428_CR12","unstructured":"M. Bay, A.F. Ehmann, J.S. Downie, in Proceedings of the 10th International Society for Music Information Retrieval Conference, ISMIR 2009, Kobe International Conference Center, Kobe, Japan, October 26-30, 2009. Evaluation of Multiple-F0 Estimation and Tracking Systems,\u00a0International Society for Music Information Retrieval. (2009), pp. 315\u2013320"},{"key":"428_CR13","unstructured":"M. M\u00fcller, V. Konz, W. Bogler, V. Arifi-M\u00fcller, in Proceedings of the International Society for Music Information Retrieval Conference (ISMIR): late breaking session. Saarland Music Data (SMD),\u00a0International Society for Music Information Retrieval. (2011)"},{"issue":"6","key":"428_CR14","doi-asserted-by":"publisher","first-page":"1643","DOI":"10.1109\/TASL.2009.2038819","volume":"18","author":"V Emiya","year":"2010","unstructured":"V. Emiya, R. Badeau, B. David, Multipitch estimation of piano sounds using a new probabilistic spectral smoothness principle. IEEE Trans. Audio Speech Lang. Process. 18(6), 1643\u20131654 (2010)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"428_CR15","unstructured":"V. Emiya, N. Bertin, B. David, R. Badeau, MAPS - A piano database for multipitch estimation and automatic transcription of music.\u00a011 (2010).\u00a0https:\/\/inria.hal.science\/inria-00544155"},{"key":"428_CR16","unstructured":"R. Kelz, G. Widmer, in Audio Engineering Society Conference: 2017 AES International Conference on Semantic Audio. An Experimental Analysis of the Entanglement Problem in Neural-Network-based Music Transcription Systems (Audio Engineering Society,\u00a0New York, 2017)"},{"key":"428_CR17","doi-asserted-by":"publisher","unstructured":"L.S. Mart\u00e1k, R. Kelz, G. Widmer, Balancing Bias and Performance in Polyphonic Piano Transcription Systems. Front. Signal Process. 2 (2022). https:\/\/doi.org\/10.3389\/frsip.2022.975932","DOI":"10.3389\/frsip.2022.975932"},{"key":"428_CR18","doi-asserted-by":"publisher","unstructured":"D. Edwards, S. Dixon, E. Benetos, A. Maezawa, Y. Kusaka, A Data-Driven Analysis of Robust Automatic Piano Transcription. IEEE Signal Process. Lett. PP(8), 1\u20135 (2024). https:\/\/doi.org\/10.1109\/LSP.2024.3363646","DOI":"10.1109\/LSP.2024.3363646"},{"key":"428_CR19","first-page":"71703","volume":"36","author":"D Teney","year":"2023","unstructured":"D. Teney, Y. Lin, S.J. Oh, E. Abbasnejad, ID and OOD Performance Are Sometimes Inversely Correlated on Real-world Datasets. Adv. Neural. Inf. Process. Syst. 36, 71703\u201371722 (2023)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"428_CR20","doi-asserted-by":"publisher","unstructured":"M. Taenzer, S.I. Mimilakis, J. Abe\u00dfer. SMD-synth: A synthesized variant of the SMD MIDI-Audio Piano Music subset (2021). https:\/\/doi.org\/10.5281\/zenodo.4637908","DOI":"10.5281\/zenodo.4637908"},{"key":"428_CR21","unstructured":"J. Abe\u00dfer, F. Bittner, M. Richter, M.G. Rodriguez, H. Lukashevich, in Proc. Digital Music Research Network One-day Workshop (DMRN+ 16). A Benchmark Dataset to Study Microphone Mismatch Conditions for Piano Multipitch Estimation on Mobile Devices (Centre for Digital Music (C4DM), Queen Mary University of London:\u00a0London, 2021)"},{"key":"428_CR22","doi-asserted-by":"crossref","unstructured":"L.N. Ferreira, L.H. Lelis, J. Whitehead, in Proceedings of the 16th AAAI Conference on Artificial Intelligence and Interactive Digital Entertainment. Computer-Generated Music for Tabletop Role-Playing Games, AIIDE\u201920 (The AAAI Press:\u00a0Washington, DC, 2020)","DOI":"10.1609\/aiide.v16i1.7408"},{"key":"428_CR23","unstructured":"J.P. Gardner, S. Durand, D. Stoller, R.M. Bittner, LLark: A Multimodal Instruction-Following Language Model for Music,\u00a0in\u00a0Proceedings of the 41st International Conference on Machine Learning.\u00a0235,\u00a015037\u201315082 (PMLR,\u00a0Vienna, 2023).\u00a0https:\/\/proceedings.mlr.press\/v235\/gardner24a.html"},{"key":"428_CR24","doi-asserted-by":"publisher","unstructured":"Y. Ma, A. \u00d8land, A. Ragni, B.M. Del Sette, C. Saitis, C. Donahue, C. Lin, C. Plachouras, E. Benetos, E. Shatri et al., Foundation Models for Music: A Survey.\u00a0CoRR.\u00a0abs\/2408.14340 (2024).\u00a0https:\/\/doi.org\/10.48550\/arXiv.2408.14340","DOI":"10.48550\/arXiv.2408.14340"},{"key":"428_CR25","doi-asserted-by":"publisher","first-page":"269","DOI":"10.1007\/BF00419657","volume":"56","author":"BH Repp","year":"1994","unstructured":"B.H. Repp, Relational invariance of expressive microstructure across global tempo changes in music performance: an exploratory study. Psychol. Res. 56, 269\u2013284 (1994)","journal-title":"Psychol. Res."},{"issue":"4","key":"428_CR26","doi-asserted-by":"publisher","first-page":"285","DOI":"10.1007\/BF00419658","volume":"56","author":"P Desain","year":"1994","unstructured":"P. Desain, H. Honing, Does expressive timing in music performance scale proportionally with tempo? Psychol. Res. 56(4), 285\u2013292 (1994)","journal-title":"Psychol. Res."},{"key":"428_CR27","doi-asserted-by":"publisher","unstructured":"M. Bernays, C. Traube, B. Gingras, W. Goebl, I. Spcl, Investigating pianists\u2019 individuality in the performance of five timbral nuances through patterns of articulation, touch, dynamics, and pedaling (2014). https:\/\/doi.org\/10.3389\/fpsyg.2014.00157","DOI":"10.3389\/fpsyg.2014.00157"},{"key":"428_CR28","unstructured":"C. Cancino-Chac\u00f3n, M. Grachten, A Computational Study of the Role of Tonal Tension in Expressive Piano Performance (2018). https:\/\/arxiv.org\/pdf\/1807.01080\u00a0Accessed 11 Mar 2024"},{"key":"428_CR29","unstructured":"D. Herremans, E. Chew, in The Annual Meeting of the Cognitive Science Society. Towards emotion based music generation: A tonal tension model based on the spiral array (2019), pp. 52\u201353. https:\/\/hal.science\/hal-03277753\/document\u00a010 Feb 2024"},{"key":"428_CR30","doi-asserted-by":"crossref","unstructured":"E. Chew, Playing with the edge: Tipping points and the role of tonality. Music Percept. Interdiscip. J. 33(3), 344\u2013366 (2016). Accessed 10 Feb 2024","DOI":"10.1525\/mp.2016.33.3.344"},{"key":"428_CR31","unstructured":"R.B. Dannenberg, in International Computer Music Conference (ICMC). The Interpretation of MIDI Velocity (2006), pp. 193\u2013196"},{"key":"428_CR32","unstructured":"F. Fabbri, A Theory of Musical Genres: Two Applications. Pop. Music Perspect. 52\u201381. Seminal critique of genre rigidity in musicology.\u00a0(International Association for the Study of Popular Music (IASPM):\u00a0Gothenburg, 1982)"},{"key":"428_CR33","doi-asserted-by":"crossref","unstructured":"D. Li, Y. Zang, Q. Kong, in 2025 IEEE international conference on acoustics, speech and signal processing (ICASSP). Piano Transcription by Hierarchical Language Modeling with Pretrained Roll-based Encoders (Nature Publishing Group:\u00a0London, 2025), p. 1--5","DOI":"10.1109\/ICASSP49660.2025.10890508"},{"issue":"6755","key":"428_CR34","doi-asserted-by":"publisher","first-page":"788","DOI":"10.1038\/44565","volume":"401","author":"DD Lee","year":"1999","unstructured":"D.D. Lee, H.S. Seung, Learning the parts of objects by non-negative matrix factorization. Nature 401(6755), 788\u2013791 (1999)","journal-title":"Nature"},{"key":"428_CR35","doi-asserted-by":"crossref","unstructured":"E.G. Tabak, E. Vanden-Eijnden, Density estimation by dual ascent of the log-likelihood. Commun. Math. Sci. 8(1), 217\u2013233 (2010)","DOI":"10.4310\/CMS.2010.v8.n1.a11"},{"key":"428_CR36","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1155\/2007\/48317","volume":"2007","author":"GE Poliner","year":"2006","unstructured":"G.E. Poliner, D.P. Ellis, A discriminative model for polyphonic piano transcription. EURASIP J. Adv. Signal Process. 2007, 1\u20139 (2006)","journal-title":"EURASIP J. Adv. Signal Process."},{"issue":"1","key":"428_CR37","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1080\/09298210902928495","volume":"38","author":"D Temperley","year":"2009","unstructured":"D. Temperley, A unified probabilistic model for polyphonic music analysis. J. New Music Res. 38(1), 3\u201318 (2009). https:\/\/doi.org\/10.1080\/09298210902928495","journal-title":"J. New Music Res."}],"container-title":["Journal on Audio, Speech, and Music Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1186\/s13636-025-00428-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1186\/s13636-025-00428-z","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1186\/s13636-025-00428-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T10:45:10Z","timestamp":1772793910000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1186\/s13636-025-00428-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,11]]},"references-count":37,"journal-issue":{"issue":"1","published-online":{"date-parts":[[2026,12]]}},"alternative-id":["428"],"URL":"https:\/\/doi.org\/10.1186\/s13636-025-00428-z","relation":{},"ISSN":["3091-4523"],"issn-type":[{"value":"3091-4523","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,12,11]]},"assertion":[{"value":"18 June 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"30 September 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 December 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 March 2026","order":6,"name":"change_date","label":"Change Date","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"Update","order":7,"name":"change_type","label":"Change Type","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"Formatting was corrected in the pdf.","order":8,"name":"change_details","label":"Change Details","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Not applicable.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval and consent to participate"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}},{"value":"The authors declare that they have no competing interests.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}],"article-number":"5"}}