{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,15]],"date-time":"2025-11-15T07:33:46Z","timestamp":1763192026248,"version":"3.45.0"},"reference-count":42,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,6,30]],"date-time":"2025-06-30T00:00:00Z","timestamp":1751241600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,6,30]],"date-time":"2025-06-30T00:00:00Z","timestamp":1751241600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,6,30]]},"DOI":"10.1109\/ijcnn64981.2025.11229410","type":"proceedings-article","created":{"date-parts":[[2025,11,14]],"date-time":"2025-11-14T18:46:15Z","timestamp":1763145975000},"page":"1-8","source":"Crossref","is-referenced-by-count":0,"title":["Denoising GER: A Noise-Robust Generative Error Correction with LLM for Speech Recognition"],"prefix":"10.1109","author":[{"given":"Yanyan","family":"Liu","sequence":"first","affiliation":[{"name":"Xinjiang University,School of Computer Science and Technology,Urumqi,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Minqiang","family":"Xu","sequence":"additional","affiliation":[{"name":"Hefei iFly Digital Technology Co. Ltd.,Hefei,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yihao","family":"Chen","sequence":"additional","affiliation":[{"name":"Hefei iFly Digital Technology Co. Ltd.,Hefei,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Liang","family":"He","sequence":"additional","affiliation":[{"name":"Xinjiang University,School of Computer Science and Technology,Urumqi,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lei","family":"Fang","sequence":"additional","affiliation":[{"name":"Hefei iFly Digital Technology Co. Ltd.,Hefei,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sian","family":"Fang","sequence":"additional","affiliation":[{"name":"Hefei iFly Digital Technology Co. Ltd.,Hefei,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lin","family":"Liu","sequence":"additional","affiliation":[{"name":"Hefei iFly Digital Technology Co. Ltd.,Hefei,China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10447563"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053606"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462682"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682490"},{"key":"ref5","first-page":"21708","article-title":"Fastcorrect: Fast error correction with edit alignment for automatic speech recognition","volume":"34","author":"Leng","year":"2021","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref6","first-page":"21708","article-title":"Fastcorrect: Fast error correction with edit alignment for automatic speech recognition","volume":"34","author":"Leng","year":"2021","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1145\/3557894"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2023-1616"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2024-989"},{"key":"ref10","first-page":"31665","article-title":"Hyporadise: An open baseline for generative speech recognition with large language models","volume":"36","author":"Chen","year":"2023","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747755"},{"article-title":"It\u2019s never too late: Fusing acoustic information into large language models for automatic speech recognition","volume-title":"International Conference on Learning Representations, ICLR","author":"Chen","key":"ref12"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.618"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10445874"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.37"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2024.3432275"},{"key":"ref17","article-title":"Qwen-audio: Advancing universal audio understanding via unified large-scale audio-language models","author":"Chu","year":"2023","journal-title":"CoRR"},{"key":"ref18","article-title":"An embarrassingly simple approach for LLM with strong ASR capacity","author":"Ma","year":"2024","journal-title":"CoRR"},{"article-title":"Salmonn: Towards generic hearing abilities for large language models","volume-title":"International Conference on Learning Representations, ICLR","author":"Yu","key":"ref19"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10447605"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU57964.2023.10389703"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1108\/LHTN-01-2023-0009"},{"article-title":"Gpt-4 technical report","year":"2023","author":"Achiam","key":"ref23"},{"article-title":"Llama: Open and efficient foundation language models","year":"2023","author":"Touvron","key":"ref24"},{"key":"ref25","first-page":"19730","article-title":"Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","volume-title":"Proc.ICML","author":"Li"},{"article-title":"Macaw-llm: Multi-modal language modeling with image, audio, video, and text integration","year":"2023","author":"Lyu","key":"ref26"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU57964.2023.10389705"},{"article-title":"Seamless: Multilingual expressive and streaming speech translation","year":"2023","author":"Barrault","key":"ref28"},{"key":"ref29","article-title":"Audiopalm: A large language model that can speak and listen","author":"Rubenstein","year":"2023","journal-title":"CoRR"},{"article-title":"Can generative large language models perform asr error correction?","year":"2023","author":"Ma","key":"ref30"},{"key":"ref31","first-page":"28492","article-title":"Robust speech recognition via large-scale weak supervision","volume-title":"International conference on machine learning","author":"Radford"},{"article-title":"Beats: Audio pre-training with acoustic tokenizers","volume-title":"Proc. ICML","author":"Chen","key":"ref32"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-emnlp.1055"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i3.25459"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2024.3425536"},{"key":"ref37","article-title":"Seed-asr: Understanding diverse speech and contexts with llm-based speech recognition","author":"Bai","year":"2024","journal-title":"CoRR"},{"key":"ref38","article-title":"MUSAN: A music, speech, and noise corpus","author":"Snyder","year":"2015","journal-title":"CoRR"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.3115\/1075527.1075614"},{"article-title":"The rwth\/upb\/forth system combination for the 4th chime challenge evaluation","year":"2016","author":"Menne","key":"ref40"},{"article-title":"Qwen technical report","year":"2023","author":"Bai","key":"ref41"},{"article-title":"Lora: Low-rank adaptation of large language models","volume-title":"International Conference on Learning Representations, ICLR","author":"Hu","key":"ref42"}],"event":{"name":"2025 International Joint Conference on Neural Networks (IJCNN)","start":{"date-parts":[[2025,6,30]]},"location":"Rome, Italy","end":{"date-parts":[[2025,7,5]]}},"container-title":["2025 International Joint Conference on Neural Networks (IJCNN)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11227166\/11227148\/11229410.pdf?arnumber=11229410","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,15]],"date-time":"2025-11-15T07:31:16Z","timestamp":1763191876000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11229410\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,30]]},"references-count":42,"URL":"https:\/\/doi.org\/10.1109\/ijcnn64981.2025.11229410","relation":{},"subject":[],"published":{"date-parts":[[2025,6,30]]}}}