{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,5]],"date-time":"2025-10-05T00:25:07Z","timestamp":1759623907162,"version":"build-2065373602"},"reference-count":19,"publisher":"Institute of Electronics, Information and Communications Engineers (IEICE)","issue":"10","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEICE Trans. Inf. &amp; Syst."],"published-print":{"date-parts":[[2025,10,1]]},"DOI":"10.1587\/transinf.2024edl8096","type":"journal-article","created":{"date-parts":[[2025,4,6]],"date-time":"2025-04-06T18:06:39Z","timestamp":1743962799000},"page":"1287-1291","source":"Crossref","is-referenced-by-count":0,"title":["Spectral-Domain Augmentation for Cover Song Identification"],"prefix":"10.1587","volume":"E108.D","author":[{"given":"Jinsoo","family":"SEO","sequence":"first","affiliation":[{"name":"Department of Electrical Engineering, Gangneung-Wonju National University"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"532","reference":[{"key":"1","doi-asserted-by":"publisher","unstructured":"[1] F. Yesiler, G. Doras, R.M. Bittner, C.J. Tralie, and J. Serr\u00e0, \u201cAudio-based musical version identification: Elements and challenges,\u201d IEEE Signal Process. Mag., vol.38, no.6, pp.115-136, Aug. 2021. 10.1109\/msp.2021.3105941","DOI":"10.1109\/MSP.2021.3105941"},{"key":"2","doi-asserted-by":"crossref","unstructured":"[2] F. Yesiler, J. Serra, and E. Gomez, \u201cAccurate and scalable version identification using musically-motivated embeddings,\u201d Proc. ICASSP-2020, pp.21-25, 2020. 10.1109\/icassp40776.2020.9053793","DOI":"10.1109\/ICASSP40776.2020.9053793"},{"key":"3","doi-asserted-by":"crossref","unstructured":"[3] Z. Yu, X. Xu, X. Chen, and D. Yang, \u201cLearning a representation for cover song identification using convolutional neural network,\u201d Proc. ICASSP-2020, pp.541-545, 2020. 10.1109\/icassp40776.2020.9053839","DOI":"10.1109\/ICASSP40776.2020.9053839"},{"key":"4","doi-asserted-by":"crossref","unstructured":"[4] X. Du, K. Chen, Z. Wang, B. Zhu, and Z. Ma, \u201cBytecover2: Towards dimensionality reduction of latent embedding for efficient cover song identification,\u201d Proc. ICASSP-2022, pp.616-620, 2022. 10.1109\/icassp43922.2022.9747630","DOI":"10.1109\/ICASSP43922.2022.9747630"},{"key":"5","doi-asserted-by":"publisher","unstructured":"[5] J. Seo, J. Kim, and H. Kim, \u201cCQTXNet: A modified Xception network with attention modules for cover song identification,\u201d IEICE Trans. Inf. &amp; Syst., vol.E107-D, no.1, pp.49-52, 2024. 10.1587\/transinf.2023mul0003","DOI":"10.1587\/transinf.2023MUL0003"},{"key":"6","doi-asserted-by":"publisher","unstructured":"[6] C. Shorten and T.M. Khoshgoftaar, \u201cA survey on image data augmentation for deep learning,\u201d Journal of big data, vol.6, no.1, pp.1-48, 2019. 10.1186\/s40537-019-0197-0","DOI":"10.1186\/s40537-019-0197-0"},{"key":"7","unstructured":"[7] S. Yun, D. Han, S. Oh, S. Chun, J. Choe, and Y. Yoo, \u201cCutmix: Regularization strategy to train strong classifiers with localizable features,\u201d Proc. CVPR-2018, pp.4510-4520, 2018."},{"key":"8","doi-asserted-by":"crossref","unstructured":"[8] J. Salamon and J. Bello, \u201cHybrid spectrogram and waveform source separation,\u201d IEEE Signal Process. Lett., vol.24, no.3, pp.279-283, March 2017.","DOI":"10.1109\/LSP.2017.2657381"},{"key":"9","doi-asserted-by":"crossref","unstructured":"[9] D.S. Park, W. Chan, Y. Zhang, C.-C. Chiu, B. Zoph, E.D. Cubuk, and Q.V. Le, \u201cSpecaugment: A simple data augmentation method for automatic speech recognition,\u201d Proc. Interspeech-2019, pp.2613-2617, 2019. 10.21437\/interspeech.2019-2680","DOI":"10.21437\/Interspeech.2019-2680"},{"key":"10","doi-asserted-by":"crossref","unstructured":"[10] X. Song, Z. Wu, Y. Huang, D. Su, and H. Meng, \u201cSpecSwap: A simple data augmentation method for end-to-end speech recognition,\u201d Proc. Interspeech-2020, pp.581-585, 2020. 10.21437\/interspeech.2020-2275","DOI":"10.21437\/Interspeech.2020-2275"},{"key":"11","doi-asserted-by":"crossref","unstructured":"[11] A. Cohen-Hadria, A. Roebel, and G. Peeters, \u201cImproving singing voice separation using deep u-net and wave-u-net with data augmentation,\u201d Proc. EUSIPCO-2019, pp.1-5, 2019. 10.23919\/eusipco.2019.8902810","DOI":"10.23919\/EUSIPCO.2019.8902810"},{"key":"12","unstructured":"[12] A. Defossez, N. Usunier, L. Bottou, and F. Bach, \u201cMusic source separation in the waveform domain,\u201d arXiv preprint arXiv:1911.13254, pp.1-16, 2019."},{"key":"13","unstructured":"[13] H. Zhang, M. Cisse, Y. Dauphin, and D. Lopez-Paz, \u201cMixup: Beyond empirical risk minimization,\u201d Proc. ICLR-2018, 2018."},{"key":"14","unstructured":"[14] Y. Tokozume, Y. Ushiku, and T. Harada, \u201cLearning from between-class examples for deep sound recognition,\u201d Proc. ICLR-2018, 2018."},{"key":"15","unstructured":"[15] A. Defossez, \u201cHybrid spectrogram and waveform source separation,\u201d Proc. MDX-2021, pp.1-11, 2021."},{"key":"16","doi-asserted-by":"crossref","unstructured":"[16] S. Wang, J. Rohdin, O. Plchot, L. Burget, K. Yu, and J. Cernocky, \u201cInvestigation of specaugment for deep speaker embedding learning,\u201d Proc. ICASSP-2020, pp.7139-7143, 2020. 10.1109\/icassp40776.2020.9053481","DOI":"10.1109\/ICASSP40776.2020.9053481"},{"key":"17","doi-asserted-by":"crossref","unstructured":"[17] X. Xu, X. Chen, and D. Yang, \u201cKey-invariant convolutional neural network toward efficient cover song identification,\u201d Proc. ICME-2018, pp.1-6, 2018. 10.1109\/icme.2018.8486531","DOI":"10.1109\/ICME.2018.8486531"},{"key":"18","doi-asserted-by":"crossref","unstructured":"[18] D.P.W. Ellis and G.E. Poliner, \u201cIdentifying \u2018cover songs\u2019 with chroma features and dynamic programming beat tracking,\u201d Proc. ICASSP-2007, pp.IV-1429-IV-1432, 2007. 10.1109\/icassp.2007.367348","DOI":"10.1109\/ICASSP.2007.367348"},{"key":"19","doi-asserted-by":"crossref","unstructured":"[19] B. McFee, C. Raffel, D. Liang, D. Ellis, M. McVicar, E. Battenberg, and O. Nieto, \u201clibrosa: Audio and music signal analysis in python,\u201d Proc. 14th python in science conference, pp.18-24, 2015. 10.25080\/majora-7b98e3ed-003","DOI":"10.25080\/Majora-7b98e3ed-003"}],"container-title":["IEICE Transactions on Information and Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E108.D\/10\/E108.D_2024EDL8096\/_pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,4]],"date-time":"2025-10-04T03:26:58Z","timestamp":1759548418000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E108.D\/10\/E108.D_2024EDL8096\/_article"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,1]]},"references-count":19,"journal-issue":{"issue":"10","published-print":{"date-parts":[[2025]]}},"URL":"https:\/\/doi.org\/10.1587\/transinf.2024edl8096","relation":{},"ISSN":["0916-8532","1745-1361"],"issn-type":[{"type":"print","value":"0916-8532"},{"type":"electronic","value":"1745-1361"}],"subject":[],"published":{"date-parts":[[2025,10,1]]},"article-number":"2024EDL8096"}}