{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T05:04:15Z","timestamp":1750309455883,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":36,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,12,3]],"date-time":"2024-12-03T00:00:00Z","timestamp":1733184000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,12,3]]},"DOI":"10.1145\/3696409.3700259","type":"proceedings-article","created":{"date-parts":[[2024,12,28]],"date-time":"2024-12-28T09:55:23Z","timestamp":1735379723000},"page":"1-7","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Mix-fine-tune: An Alternate Fine-tuning Strategy for Domain Adaptation and Generalization of Low-resource ASR"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-5799-5977","authenticated-orcid":false,"given":"Chengxi","family":"Lei","sequence":"first","affiliation":[{"name":"Massey university, Auckland, NZ"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9032-6781","authenticated-orcid":false,"given":"Satwinder Dr","family":"Singh","sequence":"additional","affiliation":[{"name":"University of Auckland, Auckland, NZ"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5776-9177","authenticated-orcid":false,"given":"Feng","family":"Hou","sequence":"additional","affiliation":[{"name":"Massey University, Auckland, NZ"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2899-9816","authenticated-orcid":false,"given":"Ruili","family":"Wang","sequence":"additional","affiliation":[{"name":"Massey University, Auckland, NZ"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,12,28]]},"reference":[{"key":"e_1_3_3_2_2_2","doi-asserted-by":"crossref","unstructured":"Arun Babu Changhan Wang Andros Tjandra Kushal Lakhotia Qiantong Xu Naman Goyal Kritika Singh Patrick Von\u00a0Platen Yatharth Saraf Juan Pino et\u00a0al. 2021. XLS-R: Self-supervised cross-lingual speech representation learning at scale. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2111.09296 (2021).","DOI":"10.21437\/Interspeech.2022-143"},{"key":"e_1_3_3_2_3_2","unstructured":"Alexei Baevski Wei-Ning Hsu Alexis Conneau and Michael Auli. 2021. Unsupervised speech recognition. Advances in Neural Information Processing Systems 34 (2021) 27826\u201327839."},{"key":"e_1_3_3_2_4_2","first-page":"1298","volume-title":"International Conference on Machine Learning","author":"Baevski Alexei","year":"2022","unstructured":"Alexei Baevski, Wei-Ning Hsu, Qiantong Xu, Arun Babu, Jiatao Gu, and Michael Auli. 2022. Data2vec: A general framework for self-supervised learning in speech, vision and language. In International Conference on Machine Learning. PMLR, 1298\u20131312."},{"key":"e_1_3_3_2_5_2","unstructured":"Alexei Baevski Steffen Schneider and Michael Auli. 2019. vq-wav2vec: Self-supervised learning of discrete speech representations. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1910.05453 (2019)."},{"key":"e_1_3_3_2_6_2","unstructured":"Alexei Baevski Yuhao Zhou Abdelrahman Mohamed and Michael Auli. 2020. wav2vec 2.0: A framework for self-supervised learning of speech representations. Advances in neural information processing systems 33 (2020) 12449\u201312460."},{"key":"e_1_3_3_2_7_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746038"},{"key":"e_1_3_3_2_8_2","doi-asserted-by":"crossref","unstructured":"Sanyuan Chen Chengyi Wang Zhengyang Chen Yu Wu Shujie Liu Zhuo Chen Jinyu Li Naoyuki Kanda Takuya Yoshioka Xiong Xiao et\u00a0al. 2022. WavLM: Large-scale self-supervised pre-training for full stack speech processing. IEEE Journal of Selected Topics in Signal Processing 16 6 (2022) 1505\u20131518.","DOI":"10.1109\/JSTSP.2022.3188113"},{"key":"e_1_3_3_2_9_2","doi-asserted-by":"crossref","unstructured":"Amandeep\u00a0Singh Dhanjal and Williamjeet Singh. 2024. A comprehensive survey on automatic speech recognition using neural networks. Multimedia Tools and Applications 83 8 (2024) 23367\u201323412.","DOI":"10.1007\/s11042-023-16438-y"},{"key":"e_1_3_3_2_10_2","doi-asserted-by":"publisher","DOI":"10.1145\/1143844.1143891"},{"key":"e_1_3_3_2_11_2","unstructured":"Hongyu Guo Yongyi Mao and Richong Zhang. 2019. Augmenting data with mixup for sentence classification: An empirical study. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1905.08941 (2019)."},{"key":"e_1_3_3_2_12_2","volume-title":"LREC","author":"Hofbauer Konrad","year":"2008","unstructured":"Konrad Hofbauer, Stefan Petrik, and Horst Hering. 2008. The ATCOSIM Corpus of Non-Prompted Clean Air Traffic Control Speech.. In LREC. Citeseer."},{"key":"e_1_3_3_2_13_2","doi-asserted-by":"crossref","unstructured":"Wei-Ning Hsu Benjamin Bolte Yao-Hung\u00a0Hubert Tsai Kushal Lakhotia Ruslan Salakhutdinov and Abdelrahman Mohamed. 2021. Hubert: Self-supervised speech representation learning by masked prediction of hidden units. IEEE\/ACM transactions on audio speech and language processing 29 (2021) 3451\u20133460.","DOI":"10.1109\/TASLP.2021.3122291"},{"key":"e_1_3_3_2_14_2","unstructured":"Wei-Ning Hsu Anuroop Sriram Alexei Baevski Tatiana Likhomanenko Qiantong Xu Vineel Pratap Jacob Kahn Ann Lee Ronan Collobert Gabriel Synnaeve et\u00a0al. 2021. Robust wav2vec 2.0: Analyzing domain shift in self-supervised pre-training. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2104.01027 (2021)."},{"key":"e_1_3_3_2_15_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9052942"},{"key":"e_1_3_3_2_16_2","first-page":"3301","volume-title":"Interspeech","author":"Kocour Martin","year":"2021","unstructured":"Martin Kocour, Karel Vesel\u1ef3, Alexander Blatt, Juan Zuluaga-Gomez, Igor Sz\u00f6ke, Jan Cernock\u1ef3, Dietrich Klakow, and Petr Motlicek. 2021. Boosting of Contextual Information in ASR for Air-Traffic Call-Sign Recognition.. In Interspeech. 3301\u20133305."},{"key":"e_1_3_3_2_17_2","doi-asserted-by":"crossref","unstructured":"Quentin Lhoest Albert\u00a0Villanova Del\u00a0Moral Yacine Jernite Abhishek Thakur Patrick Von\u00a0Platen Suraj Patil Julien Chaumond Mariama Drame Julien Plu Lewis Tunstall et\u00a0al. 2021. Datasets: A community library for natural language processing. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2109.02846 (2021).","DOI":"10.18653\/v1\/2021.emnlp-demo.21"},{"key":"e_1_3_3_2_18_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414483"},{"key":"e_1_3_3_2_19_2","unstructured":"Aaron van\u00a0den Oord Yazhe Li and Oriol Vinyals. 2018. Representation learning with contrastive predictive coding. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1807.03748 (2018)."},{"key":"e_1_3_3_2_20_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"e_1_3_3_2_21_2","doi-asserted-by":"crossref","unstructured":"Rohit Prabhavalkar Takaaki Hori Tara\u00a0N Sainath Ralf Schl\u00fcter and Shinji Watanabe. 2023. End-to-end speech recognition: A survey. IEEE\/ACM Transactions on Audio Speech and Language Processing (2023).","DOI":"10.1109\/TASLP.2023.3328283"},{"key":"e_1_3_3_2_22_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSPW59220.2023.10193184"},{"key":"e_1_3_3_2_23_2","doi-asserted-by":"crossref","unstructured":"Satwinder Singh Feng Hou and Ruili Wang. 2023. A novel self-training approach for low-resource speech recognition. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2308.05269 (2023).","DOI":"10.21437\/Interspeech.2023-540"},{"key":"e_1_3_3_2_24_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746899"},{"key":"e_1_3_3_2_25_2","unstructured":"Satwinder Singh Ruili Wang Feng Hou and Zhizhong Ma. 2023. Enhancing end-to-end automatic speech recognition for low-resource Punjabi language using synthesized datasets. Available at SSRN 4181844 (2023)."},{"key":"e_1_3_3_2_26_2","doi-asserted-by":"crossref","unstructured":"Lubo\u0161 \u0160m\u00eddl Jan \u0160vec Daniel Tihelka Jind\u0159ich Matou\u0161ek Jan Romportl and Pavel Ircing. 2019. Air traffic control communication (ATCC) speech corpora and their use for ASR and TTS development. Language Resources and Evaluation 53 (2019) 449\u2013464.","DOI":"10.1007\/s10579-019-09449-5"},{"key":"e_1_3_3_2_27_2","unstructured":"Nitish Srivastava Geoffrey Hinton Alex Krizhevsky Ilya Sutskever and Ruslan Salakhutdinov. 2014. Dropout: a simple way to prevent neural networks from overfitting. The journal of machine learning research 15 1 (2014) 1929\u20131958."},{"key":"e_1_3_3_2_28_2","unstructured":"Christian Szegedy Wojciech Zaremba Ilya Sutskever Joan Bruna Dumitru Erhan Ian Goodfellow and Rob Fergus. 2013. Intriguing properties of neural networks. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1312.6199 (2013)."},{"key":"e_1_3_3_2_29_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-21852-6_3"},{"key":"e_1_3_3_2_30_2","unstructured":"Vladimir\u00a0Naumovich Vapnik Vlamimir Vapnik et\u00a0al. 1998. Statistical learning theory. (1998)."},{"key":"e_1_3_3_2_31_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747022"},{"key":"e_1_3_3_2_32_2","first-page":"10937","volume-title":"International Conference on Machine Learning","author":"Wang Chengyi","year":"2021","unstructured":"Chengyi Wang, Yu Wu, Yao Qian, Kenichi Kumatani, Shujie Liu, Furu Wei, Michael Zeng, and Xuedong Huang. 2021. UniSpeech: Unified speech representation learning with labeled and unlabeled data. In International Conference on Machine Learning. PMLR, 10937\u201310947."},{"key":"e_1_3_3_2_33_2","doi-asserted-by":"crossref","unstructured":"Chiyuan Zhang Samy Bengio Moritz Hardt Benjamin Recht and Oriol Vinyals. 2021. Understanding deep learning (still) requires rethinking generalization. Commun. ACM 64 3 (2021) 107\u2013115.","DOI":"10.1145\/3446776"},{"key":"e_1_3_3_2_34_2","unstructured":"Hongyi Zhang Moustapha Cisse Yann\u00a0N Dauphin and David Lopez-Paz. 2017. mixup: Beyond empirical risk minimization. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1710.09412 (2017)."},{"key":"e_1_3_3_2_35_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-03335-4_31"},{"key":"e_1_3_3_2_36_2","doi-asserted-by":"crossref","unstructured":"Zi-Qiang Zhang Yan Song Ming-Hui Wu Xin Fang Ian McLoughlin and Li-Rong Dai. 2022. Cross-lingual self-training to learn multilingual representation for low-resource speech recognition. Circuits Systems and Signal Processing 41 12 (2022) 6827\u20136843.","DOI":"10.1007\/s00034-022-02075-7"},{"key":"e_1_3_3_2_37_2","unstructured":"Juan Zuluaga-Gomez Karel Vesel\u1ef3 Igor Sz\u00f6ke Alexander Blatt Petr Motlicek Martin Kocour Mickael Rigault Khalid Choukri Amrutha Prasad Seyyed\u00a0Saeed Sarfjoo et\u00a0al. 2022. ATCO2 corpus: A large-scale dataset for research on automatic speech recognition and natural language understanding of air traffic control communications. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2211.04054 (2022)."}],"event":{"name":"MMAsia '24: ACM Multimedia Asia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Auckland New Zealand","acronym":"MMAsia '24"},"container-title":["Proceedings of the 6th ACM International Conference on Multimedia in Asia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3696409.3700259","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3696409.3700259","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:10:16Z","timestamp":1750295416000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3696409.3700259"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,3]]},"references-count":36,"alternative-id":["10.1145\/3696409.3700259","10.1145\/3696409"],"URL":"https:\/\/doi.org\/10.1145\/3696409.3700259","relation":{},"subject":[],"published":{"date-parts":[[2024,12,3]]},"assertion":[{"value":"2024-12-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}