{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,20]],"date-time":"2026-05-20T04:39:30Z","timestamp":1779251970587,"version":"3.51.4"},"reference-count":34,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100018537","name":"National Science and Technology Major Project","doi-asserted-by":"publisher","award":["2022ZD0116101"],"award-info":[{"award-number":["2022ZD0116101"]}],"id":[{"id":"10.13039\/501100018537","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Key Support Project of NSFC-Liaoning Joint Foundation","award":["U1908216"],"award-info":[{"award-number":["U1908216"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Signal Process. Lett."],"published-print":{"date-parts":[[2025]]},"DOI":"10.1109\/lsp.2025.3562825","type":"journal-article","created":{"date-parts":[[2025,4,21]],"date-time":"2025-04-21T17:41:49Z","timestamp":1745257309000},"page":"1955-1959","source":"Crossref","is-referenced-by-count":4,"title":["Boosting Context-Aware Speech Translation With Large Language Models"],"prefix":"10.1109","volume":"32","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-4941-2099","authenticated-orcid":false,"given":"Yue","family":"Zhou","sequence":"first","affiliation":[{"name":"Department of Artificial Intelligence, School of Informatics, Xiamen University, Xiamen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuxuan","family":"Yuan","sequence":"additional","affiliation":[{"name":"Department of Artificial Intelligence, School of Informatics, Xiamen University, Xiamen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chengwei","family":"Zhang","sequence":"additional","affiliation":[{"name":"Department of Artificial Intelligence, School of Informatics, Xiamen University, Xiamen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8163-7139","authenticated-orcid":false,"given":"Xiaodong","family":"Shi","sequence":"additional","affiliation":[{"name":"Department of Artificial Intelligence, School of Informatics, Xiamen University, Xiamen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2023.3313513"},{"key":"ref2","first-page":"26193","article-title":"Revisiting end-to-end speech-to-text translation from scratch","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Zhang","year":"2022"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2024.3353039"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-acl.447"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-long.200"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10447450"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.81"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.262"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-long.267"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746117"},{"key":"ref11","first-page":"1877","article-title":"Language models are few-shot learners","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"33","author":"Brown","year":"2020"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.26"},{"issue":"240","key":"ref13","first-page":"1","article-title":"PaLM: Scaling language modeling with pathways","volume":"24","author":"Chowdhery","year":"2023","journal-title":"J. Mach. Learn. Res."},{"key":"ref14","doi-asserted-by":"crossref","first-page":"15757","DOI":"10.18653\/v1\/2023.findings-emnlp.1055","article-title":"SpeechGPT: Empowering large language models with intrinsic cross-modal conversational abilities","volume-title":"Proc. Findings Assoc. Comput. Linguistics: EMNLP 2023","author":"Zhang","year":"2023"},{"key":"ref15","article-title":"PolyVoice: Language models for speech to speech translation","volume-title":"Proc. 12th Int. Conf. Learn. Representations","author":"qian Dong","year":"2024"},{"key":"ref16","article-title":"Qwen-audio: Advancing universal audio understanding via unified large-scale audio-language models","author":"Chu","year":"2023","journal-title":"CoRR"},{"key":"ref17","article-title":"Qwen2-Audio technical report","author":"Chu","year":"2024"},{"key":"ref18","first-page":"24824","article-title":"Chain-of-thought prompting elicits reasoning in large language models","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"35","author":"Wei","year":"2022"},{"key":"ref19","first-page":"2012","article-title":"MuST-C: A multilingual speech translation corpus","volume-title":"Proc. 2019 Conf. North Amer. Ch. Assoc. Comput. Linguistics: Hum. Lang. Technol., Vol. 1","author":"Di Gangi","year":"2019"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2021-2027"},{"key":"ref21","first-page":"28492","article-title":"Robust speech recognition via large-scale weak supervision","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Radford","year":"2023"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref23","article-title":"LoRA: Low-rank adaptation of large language models","volume-title":"Proc. 10th Int. Conf. Learn. Representations","author":"Hu","year":"2022"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49660.2025.10890560"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W18-6319"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W17-4770"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.704"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.aacl-demo.6"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU57964.2023.10389705"},{"key":"ref30","article-title":"Seamless: Multilingual expressive and streaming speech translation","author":"Barrault","year":"2023"},{"key":"ref31","article-title":"AudioPaLM: A large language model that can speak and listen","author":"Rubenstein","year":"2023"},{"key":"ref32","first-page":"15246","article-title":"Challenges in context-aware neural machine translation","volume-title":"Proc. 2023 Conf. Empirical Methods Natural Lang. Process.","author":"Jin","year":"2023"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N18-1118"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1117"}],"container-title":["IEEE Signal Processing Letters"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/97\/10802935\/10971235.pdf?arnumber=10971235","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,15]],"date-time":"2025-05-15T17:36:18Z","timestamp":1747330578000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10971235\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"references-count":34,"URL":"https:\/\/doi.org\/10.1109\/lsp.2025.3562825","relation":{},"ISSN":["1070-9908","1558-2361"],"issn-type":[{"value":"1070-9908","type":"print"},{"value":"1558-2361","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]}}}