{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T09:04:29Z","timestamp":1765357469548,"version":"3.44.0"},"publisher-location":"New York, NY, USA","reference-count":36,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,26]],"date-time":"2023-10-26T00:00:00Z","timestamp":1698278400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Institute of Information & communications Technology Planning & Evaluation (IITP) grant funded by the Korea government","award":["2022-0-00641, 2021-0-01343"],"award-info":[{"award-number":["2022-0-00641, 2021-0-01343"]}]},{"name":"Culture Sports and Tourism R&D Program through the Korea Create Content Agency grant funded by the Ministry of Culture Sports","award":["R2022020066"],"award-info":[{"award-number":["R2022020066"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,26]]},"DOI":"10.1145\/3581783.3612269","type":"proceedings-article","created":{"date-parts":[[2023,10,27]],"date-time":"2023-10-27T07:27:30Z","timestamp":1698391650000},"page":"2362-2370","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["Exploiting Time-Frequency Conformers for Music Audio Enhancement"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-8073-3754","authenticated-orcid":false,"given":"Yunkee","family":"Chae","sequence":"first","affiliation":[{"name":"Seoul National University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-4468-0367","authenticated-orcid":false,"given":"Junghyun","family":"Koo","sequence":"additional","affiliation":[{"name":"Seoul National University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7421-9529","authenticated-orcid":false,"given":"Sungho","family":"Lee","sequence":"additional","affiliation":[{"name":"Seoul National University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4210-0312","authenticated-orcid":false,"given":"Kyogu","family":"Lee","sequence":"additional","affiliation":[{"name":"Seoul National University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"CMGAN: Conformer-Based Metric-GAN for Monaural Speech Enhancement. arXiv preprint arXiv:2209.11112","author":"Abdulatif Sherif","year":"2022","unstructured":"Sherif Abdulatif, Ruizhe Cao, and Bin Yang. 2022. CMGAN: Conformer-Based Metric-GAN for Monaural Speech Enhancement. arXiv preprint arXiv:2209.11112 (2022)."},{"key":"e_1_3_2_1_2_1","volume-title":"Phase-aware speech enhancement with deep complex u-net. arXiv preprint arXiv:1903.03107","author":"Choi Hyeong-Seok","year":"2019","unstructured":"Hyeong-Seok Choi, Jang-Hyun Kim, Jaesung Huh, Adrian Kim, Jung-Woo Ha, and Kyogu Lee. 2019. Phase-aware speech enhancement with deep complex u-net. arXiv preprint arXiv:1903.03107 (2019)."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414852"},{"key":"e_1_3_2_1_4_1","volume-title":"Music source separation in the waveform domain. arXiv preprint arXiv:1911.13254","author":"D\u00e9fossez Alexandre","year":"2019","unstructured":"Alexandre D\u00e9fossez, Nicolas Usunier, L\u00e9on Bottou, and Francis Bach. 2019. Music source separation in the waveform domain. arXiv preprint arXiv:1911.13254 (2019)."},{"key":"e_1_3_2_1_5_1","volume-title":"ICASSP 2022 deep noise suppression challenge. In ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 9271--9275","author":"Dubey Harishchandra","year":"2022","unstructured":"Harishchandra Dubey, Vishak Gopal, Ross Cutler, Ashkan Aazami, Sergiy Matusevych, Sebastian Braun, Sefik Emre Eskimez, Manthan Thakker, Takuya Yoshioka, Hannes Gamper, et al. 2022. ICASSP 2022 deep noise suppression challenge. In ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 9271--9275."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/WASPAA.2015.7336912"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746020"},{"key":"e_1_3_2_1_8_1","volume-title":"Conformer: Convolution-augmented transformer for speech recognition. arXiv preprint arXiv:2005.08100","author":"Gulati Anmol","year":"2020","unstructured":"Anmol Gulati, James Qin, Chung-Cheng Chiu, Niki Parmar, Yu Zhang, Jiahui Yu, Wei Han, Shibo Wang, Zhengdong Zhang, Yonghui Wu, et al. 2020. Conformer: Convolution-augmented transformer for speech recognition. arXiv preprint arXiv:2005.08100 (2020)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.123"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.21105\/joss.02154"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2537"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2007.911054"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.632"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747454"},{"key":"e_1_3_2_1_15_1","volume-title":"Fr\u00e9chet Audio Distance: A Metric for Evaluating Music Enhancement Algorithms. arXiv preprint arXiv:1812.08466","author":"Kilgour Kevin","year":"2018","unstructured":"Kevin Kilgour, Mauricio Zuluaga, Dominik Roblek, and Matthew Sharifi. 2018. Fr\u00e9chet Audio Distance: A Metric for Evaluating Music Enhancement Algorithms. arXiv preprint arXiv:1812.08466 (2018)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-2207"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/WASPAA52581.2021.9632794"},{"key":"e_1_3_2_1_18_1","first-page":"17022","article-title":"Hifi-gan: Generative adversarial networks for efficient and high fidelity speech synthesis","volume":"33","author":"Kong Jungil","year":"2020","unstructured":"Jungil Kong, Jaehyeon Kim, and Jaekyoung Bae. 2020a. Hifi-gan: Generative adversarial networks for efficient and high fidelity speech synthesis. Advances in Neural Information Processing Systems (NeurIPS), Vol. 33 (2020), 17022--17033.","journal-title":"Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"e_1_3_2_1_19_1","volume-title":"Diffwave: A versatile diffusion model for audio synthesis. arXiv preprint arXiv:2009.09761","author":"Kong Zhifeng","year":"2020","unstructured":"Zhifeng Kong, Wei Ping, Jiaji Huang, Kexin Zhao, and Bryan Catanzaro. 2020b. Diffwave: A versatile diffusion model for audio synthesis. arXiv preprint arXiv:2009.09761 (2020)."},{"key":"e_1_3_2_1_20_1","volume-title":"In Proceedings of the International conference on machine learning (ICML). PMLR, 1558--1566","author":"Lindbo Larsen Anders Boesen","year":"2016","unstructured":"Anders Boesen Lindbo Larsen, S\u00f8ren Kaae S\u00f8nderby, Hugo Larochelle, and Ole Winther. 2016. Autoencoding beyond pixels using a learned similarity metric. In In Proceedings of the International conference on machine learning (ICML). PMLR, 1558--1566."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","unstructured":"Vincent Lostanlen Carmine-Emanuele Cella Rachel Bittner and Slim Essid. 2018. Medley-solos-DB: a cross-collection dataset for musical instrument recognition. https:\/\/doi.org\/10.5281\/zenodo.1344103","DOI":"10.5281\/zenodo.1344103"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.304"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054536"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1428"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.5281\/zenodo.1117372"},{"key":"e_1_3_2_1_26_1","volume-title":"Proceedings of The 23rd International Society for Music Information Retrieval Conference (ISMIR)","author":"Schaffer Noah","year":"2022","unstructured":"Noah Schaffer, Boaz Cogan, Ethan Manilow, Max Morrison, Prem Seetharaman, and Bryan Pardo. 2022. Music Separation Enhancement with Generative Modeling. In Proceedings of The 23rd International Society for Music Information Retrieval Conference (ISMIR) (2022), 772--780."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.207"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-93764-9_28"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.21105\/joss.01667"},{"key":"e_1_3_2_1_30_1","volume-title":"In Proceedings of the 29th International Conference on International Joint Conferences on Artificial Intelligence (IJCAI). 3816--3822","author":"Tang Chuanxin","year":"2021","unstructured":"Chuanxin Tang, Chong Luo, Zhiyuan Zhao, Wenxuan Xie, and Wenjun Zeng. 2021. Joint time-frequency and time domain learning for speech enhancement. In In Proceedings of the 29th International Conference on International Joint Conferences on Artificial Intelligence (IJCAI). 3816--3822."},{"key":"e_1_3_2_1_31_1","volume-title":"Instance normalization: The missing ingredient for fast stylization. arXiv preprint arXiv:1607.08022","author":"Ulyanov Dmitry","year":"2016","unstructured":"Dmitry Ulyanov, Andrea Vedaldi, and Victor Lempitsky. 2016. Instance normalization: The missing ingredient for fast stylization. arXiv preprint arXiv:1607.08022 (2016)."},{"key":"e_1_3_2_1_32_1","volume-title":"Advances in Neural Information Processing Systems (NeurIPS)","volume":"30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in Neural Information Processing Systems (NeurIPS), Vol. 30 (2017)."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053795"},{"key":"e_1_3_2_1_34_1","volume-title":"2021 Asia-Pacific Signal and Information Processing Association Annual Summit and Conference (APSIPA ASC). IEEE, 523--529","author":"Yu Guochen","year":"2021","unstructured":"Guochen Yu, Yutian Wang, Chengshi Zheng, Hui Wang, and Qin Zhang. 2021. Cyclegan-based non-parallel speech enhancement with an adaptive attention-in-attention mechanism. In 2021 Asia-Pacific Signal and Information Processing Association Annual Summit and Conference (APSIPA ASC). IEEE, 523--529."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746610"},{"key":"e_1_3_2_1_36_1","volume-title":"Noisy-Reverberant Speech Enhancement Using DenseUNet with Time-Frequency Attention. In In Proceedings of Interspeech","author":"Zhao Yan","year":"2020","unstructured":"Yan Zhao and DeLiang Wang. 2020. Noisy-Reverberant Speech Enhancement Using DenseUNet with Time-Frequency Attention. In In Proceedings of Interspeech 2020. 3261--3265."}],"event":{"name":"MM '23: The 31st ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Ottawa ON Canada","acronym":"MM '23"},"container-title":["Proceedings of the 31st ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612269","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3581783.3612269","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T00:03:28Z","timestamp":1755821008000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612269"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,26]]},"references-count":36,"alternative-id":["10.1145\/3581783.3612269","10.1145\/3581783"],"URL":"https:\/\/doi.org\/10.1145\/3581783.3612269","relation":{},"subject":[],"published":{"date-parts":[[2023,10,26]]},"assertion":[{"value":"2023-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}