{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,13]],"date-time":"2026-07-13T17:42:58Z","timestamp":1783964578603,"version":"3.55.0"},"reference-count":14,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,4,6]],"date-time":"2025-04-06T00:00:00Z","timestamp":1743897600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,4,6]],"date-time":"2025-04-06T00:00:00Z","timestamp":1743897600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,4,6]]},"DOI":"10.1109\/icassp49660.2025.10889964","type":"proceedings-article","created":{"date-parts":[[2025,3,12]],"date-time":"2025-03-12T13:52:43Z","timestamp":1741787563000},"page":"1-5","source":"Crossref","is-referenced-by-count":3,"title":["Can RAG-Driven Enhancements Amplify Audio LLMs for Low-Resource Languages?"],"prefix":"10.1109","author":[{"given":"Bikash","family":"Dutta","sequence":"first","affiliation":[{"name":"Indian Institute of Technology,Jodhpur,India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rishabh","family":"Ranjan","sequence":"additional","affiliation":[{"name":"Indian Institute of Technology,Jodhpur,India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Akshat","family":"Jain","sequence":"additional","affiliation":[{"name":"Indian Institute of Technology,Jodhpur,India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Richa","family":"Singh","sequence":"additional","affiliation":[{"name":"Indian Institute of Technology,Jodhpur,India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mayank","family":"Vatsa","sequence":"additional","affiliation":[{"name":"Indian Institute of Technology,Jodhpur,India"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","first-page":"13590","article-title":"The revolution of multimodal large language models: A survey","volume-title":"Findings of the Association for Computational Linguistics, ACL 2024","author":"Caffagni"},{"key":"ref2","article-title":"L3cube-indicsbert: A simple approach for learning cross-lingual sentence representations using multilingual bert","author":"Deode","year":"2023"},{"key":"ref3","article-title":"Pengi: An audio language model for audio tasks","author":"Deshmukh","year":"2024"},{"key":"ref4","doi-asserted-by":"crossref","DOI":"10.18653\/v1\/2023.acl-long.693","article-title":"Towards leaving no indic language behind: Building monolingual corpora, benchmark and models for indic languages","volume-title":"Annual Meeting of the Association for Computational Linguistics","author":"Doddapaneni"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2022-11031"},{"key":"ref6","doi-asserted-by":"crossref","DOI":"10.18653\/v1\/2024.emnlp-main.361","article-title":"Gama: A large audio-language model with advanced audio understanding and complex reasoning abilities","author":"Ghosh","year":"2024"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/asru57964.2023.10389742"},{"key":"ref8","article-title":"Listen, think, and understand","author":"Gong","year":"2023"},{"key":"ref9","article-title":"Large language models are efficient learners of noise-robust speech recognition","volume-title":"The Twelfth International Conference on Learning Representations","author":"Hu"},{"key":"ref10","article-title":"Retrieval-augmented generation for knowledge-intensive NLP tasks","volume-title":"Advances in Neural Information Processing Systems 33: Annual Conference on Neural Information Processing Systems 2020, NeurIPS 2020","author":"Lewis"},{"key":"ref11","article-title":"Training language models to follow instructions with human feedback","volume-title":"Advances in Neural Information Processing Systems 35: Annual Conference on Neural Information Processing Systems 2022, NeurIPS 2022","author":"Ouyang"},{"key":"ref12","article-title":"Robust speech recognition via large-scale weak supervision","author":"Radford","year":"2022"},{"key":"ref13","article-title":"BLSP: bootstrapping language-speech pre-training via behavior alignment of continuation writing","volume-title":"CoRR","author":"Wang","year":"2023"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-emnlp.1055"}],"event":{"name":"ICASSP 2025 - 2025 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","location":"Hyderabad, India","start":{"date-parts":[[2025,4,6]]},"end":{"date-parts":[[2025,4,11]]}},"container-title":["ICASSP 2025 - 2025 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10887540\/10887541\/10889964.pdf?arnumber=10889964","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,25]],"date-time":"2026-03-25T05:22:32Z","timestamp":1774416152000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10889964\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,4,6]]},"references-count":14,"URL":"https:\/\/doi.org\/10.1109\/icassp49660.2025.10889964","relation":{},"subject":[],"published":{"date-parts":[[2025,4,6]]}}}