{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,6]],"date-time":"2025-12-06T16:47:57Z","timestamp":1765039677678,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":42,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,6,7]],"date-time":"2023-06-07T00:00:00Z","timestamp":1686096000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc\/4.0\/"}],"funder":[{"DOI":"10.13039\/501100000781","name":"European Research Council","doi-asserted-by":"publisher","award":["101019375"],"award-info":[{"award-number":["101019375"]}],"id":[{"id":"10.13039\/501100000781","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,6,7]]},"DOI":"10.1145\/3587819.3590968","type":"proceedings-article","created":{"date-parts":[[2023,6,8]],"date-time":"2023-06-08T17:24:19Z","timestamp":1686245059000},"page":"239-248","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["Self-Supervised Contrastive Learning for Robust Audio-Sheet Music Retrieval Systems"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-1344-3463","authenticated-orcid":false,"given":"Luis","family":"Carvalho","sequence":"first","affiliation":[{"name":"Johannes Kepler University, Linz, Austria"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-9947-6783","authenticated-orcid":false,"given":"Tobias","family":"Wash\u00fcttl","sequence":"additional","affiliation":[{"name":"Johannes Kepler University, Linz, Austria"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3531-1282","authenticated-orcid":false,"given":"Gerhard","family":"Widmer","sequence":"additional","affiliation":[{"name":"Johannes Kepler University, Linz, Austria"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,6,8]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","unstructured":"Abien Fred Agarap. 2018. Deep Learning using Rectified Linear Units (ReLU). 10.48550\/ARXIV.1803.08375","DOI":"10.48550\/ARXIV.1803.08375"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2021.3135192"},{"key":"e_1_3_2_1_3_1","volume-title":"Proceedings of the Conference on Prestigious Applications of Intelligent Systems (PAIS). Prague, Czechia.","author":"Arzt Andreas","year":"2014","unstructured":"Andreas Arzt, Sebastian B\u00f6ck, Sebastian Flossmann, Harald Frostel, Martin Gasser, Cynthia C.S. Liem, and Gerhard Widmer. 2014. The Piano Music Companion. In Proceedings of the Conference on Prestigious Applications of Intelligent Systems (PAIS). Prague, Czechia."},{"key":"e_1_3_2_1_4_1","volume-title":"Proceedings of the International Society for Music Information Retrieval Conference (ISMIR)","author":"Arzt Andreas","year":"2012","unstructured":"Andreas Arzt, Sebastian B\u00f6ck, and Gerhard Widmer. 2012. Fast Identification of Piece and Score Position via Symbolic Fingerprinting. In Proceedings of the International Society for Music Information Retrieval Conference (ISMIR). Porto, Portugal, 433--438."},{"key":"e_1_3_2_1_5_1","volume-title":"In Proceedings of the 18th European Conference on Artificial Intelligence (ECAI)","author":"Arzt Andreas","year":"2008","unstructured":"Andreas Arzt, Gerhard Widmer, and Simon Dixon. 2008. Automatic Page Turning for Musicians via Real-Time Machine Listening. In In Proceedings of the 18th European Conference on Artificial Intelligence (ECAI). Patras, Greece, 241--245."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7471681"},{"key":"e_1_3_2_1_7_1","volume-title":"Proceedings of the International Society for Music Information Retrieval Conference (ISMIR)","author":"Balke Stefan","year":"2019","unstructured":"Stefan Balke, Matthias Dorfer, Luis Carvalho, Andreas Arzt, and Gerhard Widmer. 2019. Learning Soft-Attention Models for Tempo-invariant Audio-Sheet Music Retrieval. In Proceedings of the International Society for Music Information Retrieval Conference (ISMIR). Delft, Netherlands, 216--222."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2012.6287832"},{"key":"e_1_3_2_1_9_1","volume-title":"Jan Haji\u010d Jr., and Alexander Pacha","author":"Calvo-Zaragoza Jorge","year":"2021","unstructured":"Jorge Calvo-Zaragoza, Jan Haji\u010d Jr., and Alexander Pacha. 2021. Understanding Optical Music Recognition. Comput. Surveys 53, 4 (2021)."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10994-017-5631-y"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.5555\/3524938.3525087"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2005.202"},{"key":"e_1_3_2_1_13_1","unstructured":"Djork-Arn\u00e9 Clevert Thomas Unterthiner and Sepp Hochreiter. 2016. Fast and Accurate Deep Network Learning by Exponential Linear Units (ELUs). In International Conferen1ce on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_14_1","volume-title":"Proceedings of the International Society for Music Information Retrieval Conference (ISMIR)","author":"Dorfer Matthias","year":"2017","unstructured":"Matthias Dorfer, Andreas Arzt, and Gerhard Widmer. 2017. Learning Audio-Sheet Music Correspondences for Score Identification and Offline Alignment. In Proceedings of the International Society for Music Information Retrieval Conference (ISMIR). Suzhou, China, 115--122."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.5334\/tismir.12"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1007\/s13735-018-0151-5"},{"key":"e_1_3_2_1_17_1","volume-title":"Martin Riedmiller, and Thomas Brox.","author":"Dosovitskiy Alexey","year":"2014","unstructured":"Alexey Dosovitskiy, Jost Tobias Springenberg, Martin Riedmiller, and Thomas Brox. 2014. Discriminative Unsupervised Feature Learning with Convolutional Neural Networks. In Advances in Neural Information Processing Systems, Vol. 27."},{"key":"e_1_3_2_1_18_1","volume-title":"Proceedings of the International Conference on Music Information Retrieval (ISMIR)","author":"Fremerey Christian","year":"2009","unstructured":"Christian Fremerey, Michael Clausen, Sebastian Ewert, and Meinard M\u00fcller. 2009. Sheet Music-Audio Identification. In Proceedings of the International Conference on Music Information Retrieval (ISMIR). Kobe, Japan, 645--650."},{"key":"e_1_3_2_1_19_1","volume-title":"Enabling Factorized Piano Music Modeling and Generation with the MAESTRO Dataset. In International Conference on Learning Representations.","author":"Hawthorne Curtis","year":"2019","unstructured":"Curtis Hawthorne, Andriy Stasyuk, Adam Roberts, Ian Simon, Cheng-Zhi Anna Huang, Sander Dieleman, Erich Elsen, Jesse Engel, and Douglas Eck. 2019. Enabling Factorized Piano Music Modeling and Generation with the MAESTRO Dataset. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.123"},{"key":"e_1_3_2_1_21_1","volume-title":"Real-Time Music Following in Score Sheet Images via Multi-Resolution Prediction. Frontiers in Computer Science 3","author":"Henkel Florian","year":"2021","unstructured":"Florian Henkel and Gerhard Widmer. 2021. Real-Time Music Following in Score Sheet Images via Multi-Resolution Prediction. Frontiers in Computer Science 3 (2021)."},{"key":"e_1_3_2_1_22_1","volume-title":"Proceedings of the 32nd International Conference on International Conference on Machine Learning (ICML)","author":"Ioffe Sergey","year":"2015","unstructured":"Sergey Ioffe and Christian Szegedy. 2015. Batch Normalization: Accelerating Deep Network Training by Reducing Internal Covariate Shift. In Proceedings of the 32nd International Conference on International Conference on Machine Learning (ICML). Lille, France, 448--456."},{"key":"e_1_3_2_1_23_1","volume-title":"Proceedings of the International Society for Music Information Retrieval Conference (ISMIR)","author":"Izmirli \u00d6zg\u00fcr","year":"2012","unstructured":"\u00d6zg\u00fcr Izmirli and Gyanendra Sharma. 2012. Bridging Printed Music and Audio Through Alignment Using a Mid-level Score Representation. In Proceedings of the International Society for Music Information Retrieval Conference (ISMIR). Porto, Portugal, 61--66. http:\/\/ismir2012.ismir.net\/event\/papers\/061-ismir-2012.pdf"},{"volume-title":"Adam: A Method for Stochastic Optimization. In International Conference on Learning Representations (ICLR).","author":"Diederik","key":"e_1_3_2_1_24_1","unstructured":"Diederik P. Kingma and Jimmy Ba. 2015. Adam: A Method for Stochastic Optimization. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_25_1","volume-title":"Zemel","author":"Kiros Ryan","year":"2014","unstructured":"Ryan Kiros, Ruslan Salakhutdinov, and Richard S. Zemel. 2014. Unifying Visual-Semantic Embeddings with Multimodal Neural Language Models. arXiv preprint (arXiv:1411.2539) (2014)."},{"key":"e_1_3_2_1_26_1","volume-title":"Hinton","author":"Krizhevsky Alex","year":"2012","unstructured":"Alex Krizhevsky, Ilya Sutskever, and Geoffrey E. Hinton. 2012. ImageNet Classification with Deep Convolutional Neural Networks. In Advances in Neural Information Processing Systems. 1097--1105."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-86198-8_5"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2018.2868887"},{"key":"e_1_3_2_1_29_1","volume-title":"Le","author":"Park Daniel S.","year":"2019","unstructured":"Daniel S. Park, William Chan, Yu Zhang, Chung-Cheng Chiu, Barret Zoph, Ekin D. Cubuk, and Quoc V. Le. 2019. SpecAugment: A Simple Data Augmentation Method for Automatic Speech Recognition. In Proceedings of the Annual Conference of the International Speech Communication Association (INTERSPEECH). 2613--2617."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/566654.566636"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2017.2657381"},{"key":"e_1_3_2_1_32_1","volume-title":"Proceedings of the International Society for Music Information Retrieval Conference (ISMIR). 121--126","author":"Schl\u00fcter Jan","year":"2015","unstructured":"Jan Schl\u00fcter and Thomas Grill. 2015. Exploring Data Augmentation for Improved Singing Voice Detection with Neural Networks. In Proceedings of the International Society for Music Information Retrieval Conference (ISMIR). 121--126."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1186\/s40537-019-0197-0"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2016.2533858"},{"key":"e_1_3_2_1_35_1","volume-title":"Best Practices for Convolutional Neural Networks. International Conference on Document Analysis and Recognition (ICDAR) 3","author":"Simard Patrice","year":"2003","unstructured":"Patrice Simard, Dave Steinkraus, and John Platt. 2003. Best Practices for Convolutional Neural Networks. International Conference on Document Analysis and Recognition (ICDAR) 3 (2003), 958--962."},{"key":"e_1_3_2_1_36_1","volume-title":"Very Deep Convolutional Networks for Large-Scale Image Recognition. In International Conference on Learning Representations (ICLR).","author":"Simonyan Karen","year":"2015","unstructured":"Karen Simonyan and Andrew Zisserman. 2015. Very Deep Convolutional Networks for Large-Scale Image Recognition. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_37_1","unstructured":"Kihyuk Sohn. 2016. Improved Deep Metric Learning with Multi-class N-pair Loss Objective. In Advances in Neural Information Processing Systems. 1857--1865."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-805"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053815"},{"key":"e_1_3_2_1_40_1","volume-title":"Proceedings of the International Society for Music Information Retrieval Conference (ISMIR)","author":"van der Wel Eelco","year":"2017","unstructured":"Eelco van der Wel and Karen Ullrich. 2017. Optical Music Recognition with Convolutional Sequence-to-Sequence Models. In Proceedings of the International Society for Music Information Retrieval Conference (ISMIR). Suzhou, China, 731--737."},{"volume-title":"Proceedings of the International Society for Music Information Retrieval Conference (ISMIR). 916--923","author":"Yang Daniel","key":"e_1_3_2_1_41_1","unstructured":"Daniel Yang, Thitaree Tanprasert, Teerapat Jenrungrot, Mengyi Shan, and Timothy J. Tsai. 2019. MIDI Passage Retrieval Using Cell Phone Pictures of Sheet Music. In Proceedings of the International Society for Music Information Retrieval Conference (ISMIR). 916--923."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683609"}],"event":{"name":"MMSys '23: 14th Conference on ACM Multimedia Systems","sponsor":["SIGMM ACM Special Interest Group on Multimedia","SIGCOMM ACM Special Interest Group on Data Communication","SIGMOBILE ACM Special Interest Group on Mobility of Systems, Users, Data and Computing"],"location":"Vancouver BC Canada","acronym":"MMSys '23"},"container-title":["Proceedings of the 14th ACM Multimedia Systems Conference"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3587819.3590968","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3587819.3590968","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T18:08:01Z","timestamp":1750183681000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3587819.3590968"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,6,7]]},"references-count":42,"alternative-id":["10.1145\/3587819.3590968","10.1145\/3587819"],"URL":"https:\/\/doi.org\/10.1145\/3587819.3590968","relation":{},"subject":[],"published":{"date-parts":[[2023,6,7]]},"assertion":[{"value":"2023-06-08","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}