{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,6]],"date-time":"2025-12-06T08:10:16Z","timestamp":1765008616511,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":45,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62566045, 62366037"],"award-info":[{"award-number":["62566045, 62366037"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100004763","name":"Natural Science Foundation of Inner Mongolia","doi-asserted-by":"publisher","award":["2025MS06001"],"award-info":[{"award-number":["2025MS06001"]}],"id":[{"id":"10.13039\/501100004763","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Inner Mongolia Autonomous Region First-Class Discipline Research Special Project","award":["YLXKZX-ND-036"],"award-info":[{"award-number":["YLXKZX-ND-036"]}]},{"name":"Science and Technology Program of Inner Mongolia Autonomous Region","award":["2025KYPT0041, 2025KYPT0064"],"award-info":[{"award-number":["2025KYPT0041, 2025KYPT0064"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,12,9]]},"DOI":"10.1145\/3743093.3771018","type":"proceedings-article","created":{"date-parts":[[2025,12,6]],"date-time":"2025-12-06T08:06:16Z","timestamp":1765008376000},"page":"1-7","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Zero-Shot Speech Recognition from Text-Only Data through Synthesized Spectrogram Refinement Using Style Truncation and Contextual Alignment Loss"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-0802-9203","authenticated-orcid":false,"given":"Yuan","family":"Li","sequence":"first","affiliation":[{"name":"College of Computer Science, Inner Mongolia University, Huhhot, China; National &amp; Local Joint Engineering Research Center of Intelligent Information Processing Technology for Mongolian, Huhhot, China and Inner Mongolia Key Laboratory of Multilingual Artificial Intelligence Technology, Huhhot, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1647-1539","authenticated-orcid":false,"given":"Yonghe","family":"Wang","sequence":"additional","affiliation":[{"name":"College of Computer Science, Inner Mongolia University, Huhhot, China; National &amp; Local Joint Engineering Research Center of Intelligent Information Processing Technology for Mongolian, Huhhot, China and Inner Mongolia Key Laboratory of Multilingual Artificial Intelligence Technology, Huhhot, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-6287-2472","authenticated-orcid":false,"given":"Zhenjie","family":"Gao","sequence":"additional","affiliation":[{"name":"College of Computer Science, Inner Mongolia University, Huhhot, China; National &amp; Local Joint Engineering Research Center of Intelligent Information Processing Technology for Mongolian, Huhhot, China and Inner Mongolia Key Laboratory of Multilingual Artificial Intelligence Technology, Huhhot, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7312-1629","authenticated-orcid":false,"given":"Feilong","family":"Bao","sequence":"additional","affiliation":[{"name":"College of Computer Science, Inner Mongolia University, Huhhot, China; National &amp; Local Joint Engineering Research Center of Intelligent Information Processing Technology for Mongolian, Huhhot, China and Inner Mongolia Key Laboratory of Multilingual Artificial Intelligence Technology, Huhhot, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,12,6]]},"reference":[{"key":"e_1_3_3_2_2_2","unstructured":"Ankur Bapna Yu-An Chung and Na Wu. 2021. SLAM: A Unified Encoder for Speech and Language Modeling via Speech-Text Joint Pre-Training. ArXiv abs\/2110.10329 (2021)."},{"key":"e_1_3_3_2_3_2","doi-asserted-by":"crossref","unstructured":"Emanuele Bastianelli Andrea Vanzo Pawel Swietojanski et\u00a0al. 2020. SLURP: A spoken language understanding resource package. EMNLP (2020).","DOI":"10.18653\/v1\/2020.emnlp-main.588"},{"key":"e_1_3_3_2_4_2","volume-title":"INTERSPEECH","author":"Bataev Vladimir","year":"2023","unstructured":"Vladimir Bataev, Roman Korostik, Evgeny Shabalin, et\u00a0al. 2023. Text-only domain adaptation for end-to-end ASR using integrated text-to-mel-spectrogram generator. In INTERSPEECH."},{"key":"e_1_3_3_2_5_2","volume-title":"Interspeech","author":"Bataev Vladimir","year":"2023","unstructured":"Vladimir Bataev, Roman Korostik, Evgeny Shabalin, et\u00a0al. 2023. Text-only domain adaptation for end-to-end ASR using integrated text-to-mel-spectrogram generator. In Interspeech."},{"key":"e_1_3_3_2_6_2","doi-asserted-by":"crossref","unstructured":"Afsara Benazir and Felix\u00a0Xiaozhu Lin. 2025. Privacy-Preserving Edge Speech Understanding with Tiny Foundation Models. ArXiv abs\/2502.01649 (2025).","DOI":"10.1145\/3769102.3770609"},{"key":"e_1_3_3_2_7_2","doi-asserted-by":"crossref","unstructured":"Pierre Blanchard Desmond\u00a0John Higham and Nicholas\u00a0John Higham. 2020. Accurately computing the log-sum-exp and softmax functions. IMA J. Numer. Anal. (2020).","DOI":"10.1093\/imanum\/draa038"},{"key":"e_1_3_3_2_8_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-acl.682"},{"key":"e_1_3_3_2_9_2","doi-asserted-by":"crossref","unstructured":"Xiaoxu Cai Gaige Wang Jianwen Lou et\u00a0al. 2023. Perceptual loss guided Generative adversarial network for saliency detection. Inf. Sci. 654 (2023) 119625.","DOI":"10.1016\/j.ins.2023.119625"},{"key":"e_1_3_3_2_10_2","doi-asserted-by":"crossref","unstructured":"Giuseppe\u00a0Carlo Calafiore St\u00e9phane Gaubert and Corrado Possieri. 2019. A Universal Approximation Result for Difference of Log-Sum-Exp Neural Networks. IEEE Trans. Neural Netw. Learn. Syst. 31 (2019) 5603\u20135612.","DOI":"10.1109\/TNNLS.2020.2975051"},{"key":"e_1_3_3_2_11_2","volume-title":"Interspeech","author":"Chang Yanfeng","year":"2022","unstructured":"Yanfeng Chang and Yun-Nung Chen. 2022. Contrastive Learning for Improving ASR Robustness in Spoken Language Understanding. In Interspeech."},{"key":"e_1_3_3_2_12_2","unstructured":"Cheng Chen Yuchen Hu and Chao-Han\u00a0Huck Yang. 2023. HyPoradise: An Open Baseline for Generative Speech Recognition with Large Language Models. ArXiv abs\/2309.15701 (2023)."},{"key":"e_1_3_3_2_13_2","doi-asserted-by":"crossref","unstructured":"Samuele Cornell Jordan Darefsky Zhiyao Duan et\u00a0al. 2024. Generating Data with Text-to-Speech and Large-Language Models for Conversational Speech Recognition. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2408.09215 (2024).","DOI":"10.21437\/SynData4GenAI.2024-2"},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"crossref","unstructured":"Ye Du J Zhang Xin Fang et\u00a0al. 2023. A Semi-Supervised Complementary Joint Training Approach for Low-Resource Speech Recognition. IEEE\/ACM Trans. ASLP 31 (2023) 3908\u20133921.","DOI":"10.1109\/TASLP.2023.3313434"},{"key":"e_1_3_3_2_15_2","unstructured":"Tianyu Gao Xingcheng Yao and Danqi Chen. 2021. SimCSE: Simple Contrastive Learning of Sentence Embeddings. ArXiv abs\/2104.08821 (2021)."},{"key":"e_1_3_3_2_16_2","unstructured":"Xiangheng He Junjie Chen and Zixing Zhang. 2024. ProsodyFM: Unsupervised Phrasing and Intonation Control for Intelligible Speech Synthesis. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2412.11795 (2024)."},{"key":"e_1_3_3_2_17_2","doi-asserted-by":"crossref","unstructured":"Wei-Ning Hsu Benjamin Bolte Yao-Hung\u00a0Hubert Tsai et\u00a0al. 2021. HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units. IEEE\/ACM Trans. ASLP 29 (2021) 3451\u20133460.","DOI":"10.1109\/TASLP.2021.3122291"},{"key":"e_1_3_3_2_18_2","unstructured":"Jie Huang Wei Ping Peng Xu et\u00a0al. 2023. Raven: In-context learning with retrieval augmented encoder-decoder language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2308.07922 (2023)."},{"key":"e_1_3_3_2_19_2","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU51503.2021.9688010"},{"key":"e_1_3_3_2_20_2","doi-asserted-by":"crossref","unstructured":"Amir Hussein Shinji Watanabe and Ahmed Ali. 2022. Arabic speech recognition by end-to-end modular systems and human. Comput. Speech Lang. 71 (2022) 101272.","DOI":"10.1016\/j.csl.2021.101272"},{"key":"e_1_3_3_2_21_2","doi-asserted-by":"crossref","unstructured":"Jacob Kahn Morgane Rivi\u00e8re Weiyi Zheng et\u00a0al. 2020. Libri-Light: A Benchmark for ASR with Limited or No Supervision. ICASSP (2020) 7669\u20137673.","DOI":"10.1109\/ICASSP40776.2020.9052942"},{"key":"e_1_3_3_2_22_2","doi-asserted-by":"crossref","unstructured":"Priyabrata Karmakar Shyh\u00a0Wei Teng and Guojun Lu. 2024. Thank you for attention: a survey on attention-based artificial neural networks for automatic speech recognition. Intell. Syst. Appl. (2024) 200406.","DOI":"10.1016\/j.iswa.2024.200406"},{"key":"e_1_3_3_2_23_2","unstructured":"Adrian La\u2019ncucki. 2020. Fastpitch: Parallel Text-to-Speech with Pitch Prediction. ICASSP (2020) 6588\u20136592."},{"key":"e_1_3_3_2_24_2","unstructured":"Ryan Langman Ante Juki\u0107 and Dhawan. 2024. Spectral Codecs: Spectrogram-Based Audio Codecs for High Quality Speech Synthesis. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.05298 (2024)."},{"key":"e_1_3_3_2_25_2","doi-asserted-by":"crossref","unstructured":"Qing Lin Yongbin Liu Wen Wen and Zhihua Tao. 2021. Ensemble Making Few-Shot Learning Stronger. Data Intelligence 4 (2021) 529\u2013551.","DOI":"10.1162\/dint_a_00144"},{"key":"e_1_3_3_2_26_2","doi-asserted-by":"crossref","unstructured":"Yifan Liu Hao Chen Yu Chen Wei Yin and Chunhua Shen. 2021. Generic Perceptual Loss for Modeling Structured Output Dependencies. CVPR (2021) 5420\u20135428.","DOI":"10.1109\/CVPR46437.2021.00538"},{"key":"e_1_3_3_2_27_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414567"},{"key":"e_1_3_3_2_28_2","doi-asserted-by":"crossref","unstructured":"Naoki Makishima Satoshi Suzuki Atsushi Ando et\u00a0al. 2022. Speaker consistency loss and step-wise optimization for semi-supervised joint training of TTS and ASR using unpaired text data. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2207.04659 (2022).","DOI":"10.21437\/Interspeech.2022-304"},{"key":"e_1_3_3_2_29_2","volume-title":"Interspeech","author":"Makishima Naoki","year":"2022","unstructured":"Naoki Makishima, Satoshi Suzuki, Atsushi Ando, et\u00a0al. 2022. Speaker consistency loss and step-wise optimization for semi-supervised joint training of TTS and ASR using unpaired text data. In Interspeech."},{"key":"e_1_3_3_2_30_2","unstructured":"Zhenliang Ni Xinghao Chen Yingjie Zhai Yehui Tang and Yunhe Wang. 2024. Context-Guided Spatial Feature Reconstruction for Efficient Semantic Segmentation. ArXiv abs\/2405.06228 (2024)."},{"key":"e_1_3_3_2_31_2","doi-asserted-by":"crossref","unstructured":"Vassil Panayotov Guoguo Chen Daniel Povey and Sanjeev Khudanpur. 2015. Librispeech: An ASR corpus based on public domain audio books. ICASSP (2015) 5206\u20135210.","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"e_1_3_3_2_32_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10446998"},{"key":"e_1_3_3_2_33_2","doi-asserted-by":"crossref","unstructured":"Mohammad\u00a0Saeed Rad Behzad Bozorgtabar Urs-Viktor Marti et\u00a0al. 2019. SROBB: Targeted Perceptual Loss for Single Image Super-Resolution. ICCV (2019) 2710\u20132719.","DOI":"10.1109\/ICCV.2019.00280"},{"key":"e_1_3_3_2_34_2","unstructured":"Paul K et\u00a0al. Rubenstein. 2023. Audiopalm: A large language model that can speak and listen. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2306.12925 (2023)."},{"key":"e_1_3_3_2_35_2","volume-title":"International Conference on Machine Learning","author":"Sauer Axel","year":"2023","unstructured":"Axel Sauer, Tero Karras, Samuli Laine, et\u00a0al. 2023. StyleGAN-T: Unlocking the Power of GANs for Fast Large-Scale Text-to-Image Synthesis. In International Conference on Machine Learning."},{"key":"e_1_3_3_2_36_2","doi-asserted-by":"crossref","unstructured":"Wenzhe Shi Jose Caballero Ferenc Husz\u00e1r et\u00a0al. 2016. Real-Time Single Image and Video Super-Resolution Using an Efficient Sub-Pixel Convolutional Neural Network. CVPR (2016) 1874\u20131883.","DOI":"10.1109\/CVPR.2016.207"},{"key":"e_1_3_3_2_37_2","doi-asserted-by":"crossref","unstructured":"Suwon Shon Felix Wu and Kwangyoun Kim. 2022. Context-Aware Fine-Tuning of Self-Supervised Speech Models. ICASSP (2022) 1\u20135.","DOI":"10.1109\/ICASSP49357.2023.10094687"},{"key":"e_1_3_3_2_38_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10447240"},{"key":"e_1_3_3_2_39_2","volume-title":"Interspeech","author":"Sundararaman Mukuntha\u00a0Narayanan","year":"2021","unstructured":"Mukuntha\u00a0Narayanan Sundararaman, Ayush Kumar, and Jithendra Vepa. 2021. Phoneme-BERT: Joint Language Modelling of Phoneme Sequence and ASR Transcript. In Interspeech."},{"key":"e_1_3_3_2_40_2","doi-asserted-by":"crossref","unstructured":"Sei Ueno and Tatsuya Kawahara. 2022. Phone-Informed Refinement of Synthesized Mel Spectrogram for Data Augmentation in Speech Recognition. ICASSP (2022) 8572\u20138576.","DOI":"10.1109\/ICASSP43922.2022.9746582"},{"key":"e_1_3_3_2_41_2","doi-asserted-by":"crossref","unstructured":"Sei Ueno Akinobu Lee and Tatsuya Kawahara. 2024. Refining Synthesized Speech Using Speaker Information and Phone Masking for Data Augmentation of Speech Recognition. TASLP 32 (2024) 3924\u20133933.","DOI":"10.1109\/TASLP.2024.3451982"},{"key":"e_1_3_3_2_42_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-industry.35"},{"key":"e_1_3_3_2_43_2","unstructured":"YongHui Wei Chuluunbandi Naimannaran Tuyatsetseg Badarch et\u00a0al. 2024. Mongolian Text Recognition Based on CRNN Algorithm. Data Intelligence (2024)."},{"key":"e_1_3_3_2_44_2","unstructured":"Xianghu Yue Xiaoxue Gao and Xinyuan Qian. 2024. Adapting Pre-Trained Self-Supervised Learning Model for Speech Recognition with Light-Weight Adapters. Electronics (2024)."},{"key":"e_1_3_3_2_45_2","volume-title":"Interspeech","author":"Zen Heiga","year":"2019","unstructured":"Heiga Zen, Viet Dang, and Robert A.\u00a0J. Clark. 2019. LibriTTS: A Corpus Derived from LibriSpeech for Text-to-Speech. In Interspeech."},{"key":"e_1_3_3_2_46_2","doi-asserted-by":"crossref","unstructured":"Haoyi Zhou Jianxin Li Jieqi Peng et\u00a0al. 2021. Triplet Attention: Rethinking the Similarity in Transformers. Proc. 27th ACM SIGKDD Conf. Knowl. Discov. Data Min. (2021).","DOI":"10.1145\/3447548.3467241"}],"event":{"name":"MMAsia '25: ACM Multimedia Asia","location":"Kuala Lumpur Malaysia","acronym":"MMAsia '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 7th ACM International Conference on Multimedia in Asia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3743093.3771018","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,6]],"date-time":"2025-12-06T08:08:25Z","timestamp":1765008505000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3743093.3771018"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,6]]},"references-count":45,"alternative-id":["10.1145\/3743093.3771018","10.1145\/3743093"],"URL":"https:\/\/doi.org\/10.1145\/3743093.3771018","relation":{},"subject":[],"published":{"date-parts":[[2025,12,6]]},"assertion":[{"value":"2025-12-06","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}