{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T11:23:54Z","timestamp":1777634634326,"version":"3.51.4"},"reference-count":77,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"National Key Research and Development Program of China","award":["2021YFC3340803"],"award-info":[{"award-number":["2021YFC3340803"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61877060"],"award-info":[{"award-number":["61877060"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE\/ACM Trans. Audio Speech Lang. Process."],"published-print":{"date-parts":[[2023]]},"DOI":"10.1109\/taslp.2022.3221005","type":"journal-article","created":{"date-parts":[[2022,11,10]],"date-time":"2022-11-10T20:37:31Z","timestamp":1668112651000},"page":"229-241","source":"Crossref","is-referenced-by-count":8,"title":["MusicYOLO: A Vision-Based Framework for Automatic Singing Transcription"],"prefix":"10.1109","volume":"31","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7037-9800","authenticated-orcid":false,"given":"Xianke","family":"Wang","sequence":"first","affiliation":[{"name":"Hubei Key Laboratory of Smart Internet Technology, Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4885-8922","authenticated-orcid":false,"given":"Bowen","family":"Tian","sequence":"additional","affiliation":[{"name":"Hubei Key Laboratory of Smart Internet Technology, Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Weiming","family":"Yang","sequence":"additional","affiliation":[{"name":"Hubei Key Laboratory of Smart Internet Technology, Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4705-7189","authenticated-orcid":false,"given":"Wei","family":"Xu","sequence":"additional","affiliation":[{"name":"Hubei Key Laboratory of Smart Internet Technology, Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wenqing","family":"Cheng","sequence":"additional","affiliation":[{"name":"Hubei Key Laboratory of Smart Internet Technology, Huazhong University of Science and Technology, Wuhan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.91"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2016.2577031"},{"key":"ref71","article-title":"Darknet: Open source neural networks in c","author":"redmon","year":"2016"},{"key":"ref70","article-title":"Yolox: Exceeding yolo series in 2021","author":"ge","year":"2021"},{"key":"ref76","first-page":"367","article-title":"mir_eval: A transparent implementation of common MIR metrics","author":"raffel","year":"0","journal-title":"Proc Int Soc Music Inf Retrieval Conf"},{"key":"ref77","first-page":"900","article-title":"Hierarchical classification networks for singing voice segmentation and transcription","author":"fu","year":"0","journal-title":"Proc Int Soc Music Inf Retrieval Conf"},{"key":"ref74","first-page":"567","article-title":"Evaluation framework for automatic singing transcription","author":"molina","year":"0","journal-title":"Proc Int Soc Music Inf Retrieval Conf"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2017.2778423"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2010.2042119"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-69568-4_29"},{"key":"ref33","first-page":"737","article-title":"Singing voice melody transcription using deep neural networks","author":"rigaud","year":"0","journal-title":"Proc Int Soc Music Inf Retrieval Conf"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/CISCE52179.2021.9445941"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2004.1326812"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1162\/COMJ_a_00180"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1017\/ATSIP.2021.4"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.2996095"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952166"},{"key":"ref34","article-title":"Melody extraction from polyphonic music signals","author":"salamon","year":"2013"},{"key":"ref60","first-page":"85","article-title":"DCASE 2017 challenge setup: Tasks, datasets and baseline system","author":"mesaros","year":"0","journal-title":"Proc Detection Classification Acoust Scenes Events Workshop"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952264"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1145\/2964284.2964310"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01049"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461329"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.21105\/joss.02154"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2013.2295918"},{"key":"ref65","first-page":"234","article-title":"U-Net: Convolutional networks for biomedical image segmentation","author":"ronneberger","year":"0","journal-title":"Proc Int Conf Med Image Comput Comput - Assist Interv"},{"key":"ref66","first-page":"3","article-title":"Constant-Q transform toolbox for music processing","author":"sch\u00f6rkhuber","year":"0","journal-title":"Proc Proc Sound Music Comput Conf"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.2982285"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2010-488"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2010.2100380"},{"key":"ref69","first-page":"18","article-title":"librosa: Audio and music signal analysis in python","author":"mcfee","year":"0","journal-title":"Proc 14th Python Sci Conf"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2014.6853672"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414601"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1121\/1.396427"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2004.1325920"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178034"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/TAU.1968.1161986"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2005-335"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952173"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2014.2363410"},{"key":"ref50","first-page":"2375","article-title":"Multi-microphone fusion for detection of speech and acoustic events in smart spaces","author":"giannoulis","year":"0","journal-title":"Proc Eur Signal Process Conf"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2016.2530401"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-2323"},{"key":"ref58","first-page":"128","article-title":"Audio event detection and classification using extended R-FCN approach","author":"wang","year":"0","journal-title":"Proc Detection Classification Acoust Scenes Events Workshop"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/WASPAA.2015.7336889"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1007\/978-0-387-93808-0_40"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-69568-4_29"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-69568-4"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.23919\/EUSIPCO.2017.8081709"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/iCECE.2010.78"},{"key":"ref10","article-title":"Modeling music: Studies of music transcription, music perception and music production","author":"elowsson","year":"2018"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2014.6854953"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2014.2367814"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.23919\/EUSIPCO.2017.8081482"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-240"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6639185"},{"key":"ref15","first-page":"525","article-title":"Comparison of pitch trackers for real-time guitar effects","author":"knesebeck","year":"0","journal-title":"Proc 13th Int Conf Digit Audio Effects"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/TASSP.1976.1162765"},{"key":"ref17","first-page":"97","article-title":"Accurate short-term analysis of the fundamental frequency and the harmonics-to-noise ratio of a sampled sound","volume":"17","author":"boersma","year":"1993","journal-title":"Proc Inst Phonetic Sci"},{"key":"ref18","first-page":"518","article-title":"A robust algorithm for pitch tracking (RAPT)","volume":"495","author":"talkin","year":"1995","journal-title":"Speech Coding and Synthesis"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1121\/1.2951592"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2014.6853678"},{"key":"ref3","first-page":"23","article-title":"Computer-aided melody note transcription using the Tony software: Accuracy and efficiency","author":"mauch","year":"0","journal-title":"Proc Int Conf Technol Music Notation Representation"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2014.2331102"},{"key":"ref5","first-page":"301","article-title":"Signal processing for melody transcription","author":"mcnab","year":"0","journal-title":"Proc 19th Aust Comput Sci Conf"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2016.2531284"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1121\/1.1458024"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/ICME.2015.7177439"},{"key":"ref9","first-page":"1","article-title":"Note onset detection based on harmonic cepstrum regularity","author":"heo","year":"0","journal-title":"Proc IEEE Int Conf Multimedia Expo"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN.2015.7280624"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1016\/j.patrec.2013.02.015"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2011.5947153"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472917"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2012.2226160"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2015.2467964"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-123"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2015.2389618"}],"container-title":["IEEE\/ACM Transactions on Audio, Speech, and Language Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6570655\/9970249\/09944977.pdf?arnumber=9944977","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,12,26]],"date-time":"2022-12-26T19:26:35Z","timestamp":1672082795000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9944977\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023]]},"references-count":77,"URL":"https:\/\/doi.org\/10.1109\/taslp.2022.3221005","relation":{},"ISSN":["2329-9290","2329-9304"],"issn-type":[{"value":"2329-9290","type":"print"},{"value":"2329-9304","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023]]}}}