{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T17:07:47Z","timestamp":1777568867748,"version":"3.51.4"},"publisher-location":"Singapore","reference-count":23,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819743988","type":"print"},{"value":"9789819743995","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024]]},"DOI":"10.1007\/978-981-97-4399-5_13","type":"book-chapter","created":{"date-parts":[[2024,7,6]],"date-time":"2024-07-06T16:01:52Z","timestamp":1720281712000},"page":"133-142","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["Audio-LLM: Activating the\u00a0Capabilities of\u00a0Large Language Models to\u00a0Comprehend Audio Data"],"prefix":"10.1007","author":[{"given":"Dongting","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chenchong","family":"Tang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Han","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,7,7]]},"reference":[{"key":"13_CR1","unstructured":"Brown, T.B., Mann, B., Ryder, N., Subbiah, M., Kaplan, J., et al.: Language models are few-shot learners. In: NeurIPS (2020)"},{"key":"13_CR2","doi-asserted-by":"crossref","unstructured":"Du, Z., et al.: GLM: general language model pretraining with autoregressive blank infilling. ACL (2022)","DOI":"10.18653\/v1\/2022.acl-long.26"},{"key":"13_CR3","unstructured":"Anil, R., Dai, A.M., Firat, O., Johnson, M., Lepikhin, D., et al.: PaLM 2 technical report. arXiv preprint, arXiv:2305.10403 (2023)"},{"key":"13_CR4","unstructured":"Chiang, W.L., et al.: Vicuna: an open-source chatbot impressing GPT-4 with 90% * ChatGPT quality (2023). https:\/\/lmsys.org\/blog\/2023-03-30-vicuna\/"},{"key":"13_CR5","unstructured":"OpenAI: GPT-4 technical report. arXiv preprint, arXiv:2303.08774 (2023)"},{"key":"13_CR6","unstructured":"Touvron, H., Lavril, T., Izacard, G., Martinet, X., et al.: LLaMA: open and efficient foundation language models. arXiv preprint, arXiv:2302.13971 (2023)"},{"key":"13_CR7","unstructured":"Chen, F., et al.: LLaMA: open and efficient foundation language models. arXiv preprint, arXiv:2305.04160 (2023)"},{"key":"13_CR8","unstructured":"Gong, Y., Luo, H., Liu, A.H., Karlinsky, L., Glass, J.: Listen, think, and understand. arXiv preprint, arXiv:2305.10790 (2023)"},{"key":"13_CR9","unstructured":"Rubenstein, P.K., Asawaroengchai, C., Nguyen, D.D., Bapna, A., et al.: AudioPaLM: a large language model that can speak and listen. arXiv preprint, arXiv:2306.12925 (2023)"},{"key":"13_CR10","unstructured":"Tang, C., Yu, W., Sun, G., et al.: SALMONN: towards generic hearing abilities for large language models. arXiv preprint, arXiv:2310.13289 (2023)"},{"key":"13_CR11","unstructured":"Radford, A., Kim, J.W., Xu, T., Brockman, G., McLeavey, C., Sutskever, I.: Robust speech recognition via large-scale weak supervision. In: ICML (2023)"},{"key":"13_CR12","unstructured":"Chen, S., et al.: BEATs: audio pre-training with acoustic tokenizers. In: ICML (2023)"},{"key":"13_CR13","unstructured":"Li, J., Li, D., Savarese, S., Hoi, S.: BLIP-2: bootstrapping language-image pre-training with frozen image encoders and large language models. In: ICML (2023)"},{"key":"13_CR14","unstructured":"Vaswani, A., et al.: Attention is all you need. In: NeurIPS (2017)"},{"key":"13_CR15","unstructured":"Radford, A., Wu, J., Child, R., Luan, D., Amodei, D., Sutskever, I.: Language models are unsupervised multitask learners (2019)"},{"key":"13_CR16","unstructured":"Hu, E.J., et al.: LoRA: low-rank adaptation of large language models. In: ICLR (2022)"},{"key":"13_CR17","doi-asserted-by":"crossref","unstructured":"Fathullah, Y., Wu, C., Lakomkin, E., et al.: Prompting large language models with speech recognition abilities. arXiv preprint, arXiv:2307.11795 (2023)","DOI":"10.1109\/ICASSP48485.2024.10447605"},{"key":"13_CR18","doi-asserted-by":"crossref","unstructured":"Borsos, Z., Marinier, R., Vincent, D., et al.: AudioLM: a language modeling approach to audio generation. arXiv preprint, arXiv:2209.03143 (2023)","DOI":"10.1109\/TASLP.2023.3288409"},{"key":"13_CR19","doi-asserted-by":"crossref","unstructured":"Panayotov, V., Chen, G., Povey, D., Khudanpur, S.: LibriSpeech: an ASR corpus based on public domain audio books. In: ICASSP (2015)","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"13_CR20","doi-asserted-by":"crossref","unstructured":"Wang, C., Wu, A., Gu, J., Pino, J.: CoVoST 2 and massively multilingual speech translation. In: Interspeech (2021)","DOI":"10.21437\/Interspeech.2021-2027"},{"key":"13_CR21","doi-asserted-by":"publisher","first-page":"335","DOI":"10.1007\/s10579-008-9076-6","volume":"42","author":"C Busso","year":"2008","unstructured":"Busso, C., Bulut, M., Lee, C.C., et al.: IEMOCAP: interactive emotional dyadic motion capture database. Lang. Resour. Eval. 42, 335\u2013359 (2008)","journal-title":"Lang. Resour. Eval."},{"key":"13_CR22","unstructured":"Agostinelli, A., Denk, T.I., Borsos, Z., Engel, J., et al.: MusicLM: generating Music From Text. arXiv preprint, arXiv:2301.11325 (2015)"},{"key":"13_CR23","unstructured":"Baevski, A., Zhou, H., Mohamed, A., et al.: wav2vec 2.0: a framework for self-supervised learning of speech representations. In: NeurIPS (2020)"}],"container-title":["Lecture Notes in Computer Science","Advances in Neural Networks \u2013 ISNN 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-97-4399-5_13","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,7,6]],"date-time":"2024-07-06T16:03:45Z","timestamp":1720281825000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-97-4399-5_13"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"ISBN":["9789819743988","9789819743995"],"references-count":23,"URL":"https:\/\/doi.org\/10.1007\/978-981-97-4399-5_13","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024]]},"assertion":[{"value":"7 July 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ISNN","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Symposium on Neural Networks","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Weihai","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"11 July 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"14 July 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"isnn2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/conference.cs.cityu.edu.hk\/isnn\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}