{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,25]],"date-time":"2025-03-25T17:40:31Z","timestamp":1742924431937,"version":"3.40.3"},"publisher-location":"Cham","reference-count":30,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031702389"},{"type":"electronic","value":"9783031702396"}],"license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024]]},"DOI":"10.1007\/978-3-031-70239-6_17","type":"book-chapter","created":{"date-parts":[[2024,9,19]],"date-time":"2024-09-19T06:02:09Z","timestamp":1726725729000},"page":"241-255","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["S$$^3$$: A Simple Strong Sample-Effective Multimodal Dialog System"],"prefix":"10.1007","author":[{"given":"Elisei","family":"Rykov","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Egor","family":"Malkershin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Alexander","family":"Panchenko","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,9,20]]},"reference":[{"key":"17_CR1","doi-asserted-by":"publisher","unstructured":"Ainslie, J., Lee-Thorp, J., de\u00a0Jong, M., Zemlyanskiy, Y., Lebron, F., Sanghai, S.: GQA: training generalized multi-query transformer models from multi-head checkpoints. In: Bouamor, H., Pino, J., Bali, K. (eds.) Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, pp. 4895\u20134901. Association for Computational Linguistics, Singapore (Dec 2023). https:\/\/doi.org\/10.18653\/v1\/2023.emnlp-main.298, https:\/\/aclanthology.org\/2023.emnlp-main.298","DOI":"10.18653\/v1\/2023.emnlp-main.298"},{"key":"17_CR2","doi-asserted-by":"publisher","unstructured":"Awadalla, A., et al.: Openflamingo: an open-source framework for training large autoregressive vision-language models. CoRR abs\/2308.01390 (2023). https:\/\/doi.org\/10.48550\/ARXIV.2308.01390","DOI":"10.48550\/ARXIV.2308.01390"},{"key":"17_CR3","doi-asserted-by":"publisher","unstructured":"Bai, J., et al.: Qwen-vl: a frontier large vision-language model with versatile abilities. CoRR abs\/2308.12966 (2023). https:\/\/doi.org\/10.48550\/ARXIV.2308.12966","DOI":"10.48550\/ARXIV.2308.12966"},{"key":"17_CR4","unstructured":"Banerjee, S., Lavie, A.: METEOR: an automatic metric for MT evaluation with improved correlation with human judgments. In: Goldstein, J., Lavie, A., Lin, C.Y., Voss, C. (eds.) Proceedings of the ACL Workshop on Intrinsic and Extrinsic Evaluation Measures for Machine Translation and\/or Summarization, pp. 65\u201372. Association for Computational Linguistics, Ann Arbor, Michigan (Jun 2005)"},{"key":"17_CR5","doi-asserted-by":"publisher","unstructured":"Cha, J., Kang, W., Mun, J., Roh, B.: Honeybee: locality-enhanced projector for multimodal LLM. CoRR abs\/2312.06742 (2023). https:\/\/doi.org\/10.48550\/ARXIV.2312.06742","DOI":"10.48550\/ARXIV.2312.06742"},{"key":"17_CR6","unstructured":"Dai, W., et al.: Instructblip: towards general-purpose vision-language models with instruction tuning. In: Oh, A., Neumann, T., Globerson, A., Saenko, K., Hardt, M., Levine, S. (eds.) Advances in Neural Information Processing Systems. vol.\u00a036, pp. 49250\u201349267. Curran Associates, Inc. (2023)"},{"key":"17_CR7","doi-asserted-by":"crossref","unstructured":"Das, A., et al.: Visual Dialog. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2017)","DOI":"10.1109\/CVPR.2017.121"},{"key":"17_CR8","doi-asserted-by":"publisher","unstructured":"Drossos, K., Lipping, S., Virtanen, T.: Clotho: an audio captioning dataset. In: ICASSP 2020 - 2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 736\u2013740 (2020). https:\/\/doi.org\/10.1109\/ICASSP40776.2020.9052990","DOI":"10.1109\/ICASSP40776.2020.9052990"},{"key":"17_CR9","doi-asserted-by":"publisher","unstructured":"Gao, P., et al.: Llama-adapter V2: parameter-efficient visual instruction model. CoRR abs\/2304.15010 (2023). https:\/\/doi.org\/10.48550\/ARXIV.2304.15010","DOI":"10.48550\/ARXIV.2304.15010"},{"key":"17_CR10","doi-asserted-by":"publisher","unstructured":"Girdhar, R., et al.: Imagebind one embedding space to bind them all. In: 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 15180\u201315190 (2023). https:\/\/doi.org\/10.1109\/CVPR52729.2023.01457","DOI":"10.1109\/CVPR52729.2023.01457"},{"key":"17_CR11","doi-asserted-by":"crossref","unstructured":"Gurari, D., et al.: Vizwiz grand challenge: answering visual questions from blind people. In: 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3608\u20133617 (2018). https:\/\/api.semanticscholar.org\/CorpusID:3831582","DOI":"10.1109\/CVPR.2018.00380"},{"key":"17_CR12","doi-asserted-by":"publisher","unstructured":"Jiang, A.Q., et al.: Mistral 7b. CoRR abs\/2310.06825 (2023). https:\/\/doi.org\/10.48550\/ARXIV.2310.06825","DOI":"10.48550\/ARXIV.2310.06825"},{"key":"17_CR13","unstructured":"Koh, J.Y., Salakhutdinov, R., Fried, D.: Grounding language models to images for multimodal inputs and outputs. In: Proceedings of the 40th International Conference on Machine Learning. ICML\u201923, JMLR.org (2023)"},{"key":"17_CR14","unstructured":"K\u00f6pf, A., et al.: Openassistant conversations - democratizing large language model alignment. In: Oh, A., Naumann, T., Globerson, A., Saenko, K., Hardt, M., Levine, S. (eds.) Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10\u201316, 2023 (2023), http:\/\/papers.nips.cc\/paper_files\/paper\/2023\/hash\/949f0f8f32267d297c2d4e3ee10a2e7e-Abstract-Datasets_and_Benchmarks.html"},{"key":"17_CR15","doi-asserted-by":"publisher","unstructured":"Li, B., Zhang, Y., Chen, L., Wang, J., Yang, J., Liu, Z.: Otter: a multi-modal model with in-context instruction tuning. CoRR abs\/2305.03726 (2023). https:\/\/doi.org\/10.48550\/ARXIV.2305.03726","DOI":"10.48550\/ARXIV.2305.03726"},{"key":"17_CR16","doi-asserted-by":"publisher","first-page":"740","DOI":"10.1007\/978-3-319-10602-1_48","volume-title":"Computer Vision - ECCV 2014","author":"TY Lin","year":"2014","unstructured":"Lin, T.Y., et al.: Microsoft coco: common objects in context. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) Computer Vision - ECCV 2014, pp. 740\u2013755. Springer International Publishing, Cham (2014)"},{"key":"17_CR17","doi-asserted-by":"publisher","unstructured":"Lin, Z., et al.: SPHINX: the joint mixing of weights, tasks, and visual embeddings for multi-modal large language models. CoRR abs\/2311.07575 (2023). https:\/\/doi.org\/10.48550\/ARXIV.2311.07575","DOI":"10.48550\/ARXIV.2311.07575"},{"key":"17_CR18","doi-asserted-by":"publisher","unstructured":"Liu, H., Li, C., Li, Y., Lee, Y.J.: Improved baselines with visual instruction tuning. CoRR abs\/2310.03744 (2023). https:\/\/doi.org\/10.48550\/ARXIV.2310.03744","DOI":"10.48550\/ARXIV.2310.03744"},{"key":"17_CR19","unstructured":"Liu, H., Li, C., Wu, Q., Lee, Y.J.: Visual instruction tuning. In: Oh, A., Naumann, T., Globerson, A., Saenko, K., Hardt, M., Levine, S. (eds.) Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10\u201316, 2023 (2023), http:\/\/papers.nips.cc\/paper_files\/paper\/2023\/hash\/6dcf277ea32ce3288914faf369fe6de0-Abstract-Conference.html"},{"key":"17_CR20","unstructured":"Lu, P., et al.: Learn to explain: multimodal reasoning via thought chains for science question answering. In: NeurIPS (2022)"},{"key":"17_CR21","doi-asserted-by":"publisher","unstructured":"Peng, Z., et al.: Kosmos-2: grounding multimodal large language models to the world. CoRR abs\/2306.14824 (2023). https:\/\/doi.org\/10.48550\/ARXIV.2306.14824","DOI":"10.48550\/ARXIV.2306.14824"},{"key":"17_CR22","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: Meila, M., Zhang, T. (eds.) Proceedings of the 38th International Conference on Machine Learning, ICML 2021, 18\u201324 July 2021, Virtual Event. Proceedings of Machine Learning Research, vol.\u00a0139, pp. 8748\u20138763. PMLR (2021). http:\/\/proceedings.mlr.press\/v139\/radford21a.html"},{"key":"17_CR23","unstructured":"Radford, A., Kim, J.W., Xu, T., Brockman, G., McLeavey, C., Sutskever, I.: Robust speech recognition via large-scale weak supervision. In: Proceedings of the 40th International Conference on Machine Learning. In: ICML\u201923, JMLR.org (2023)"},{"key":"17_CR24","doi-asserted-by":"publisher","first-page":"742","DOI":"10.1007\/978-3-030-58536-5_44","volume-title":"Computer Vision - ECCV 2020","author":"O Sidorov","year":"2020","unstructured":"Sidorov, O., Hu, R., Rohrbach, M., Singh, A.: Textcaps: a dataset for image captioning with reading comprehension. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.M. (eds.) Computer Vision - ECCV 2020, pp. 742\u2013758. Springer International Publishing, Cham (2020)"},{"key":"17_CR25","doi-asserted-by":"publisher","unstructured":"Touvron, H., et al: Llama 2: open foundation and fine-tuned chat models. CoRR abs\/2307.09288 (2023). https:\/\/doi.org\/10.48550\/ARXIV.2307.09288","DOI":"10.48550\/ARXIV.2307.09288"},{"key":"17_CR26","unstructured":"Wang, W., et al.: CogVLM: visual expert for large language models (2024). https:\/\/openreview.net\/forum?id=c72vop46KY"},{"key":"17_CR27","doi-asserted-by":"publisher","unstructured":"Ye, Q., et al.: mplug-owl2: revolutionizing multi-modal large language model with modality collaboration. CoRR abs\/2311.04257 (2023). https:\/\/doi.org\/10.48550\/ARXIV.2311.04257","DOI":"10.48550\/ARXIV.2311.04257"},{"key":"17_CR28","doi-asserted-by":"publisher","unstructured":"Young, A., et al.: Open foundation models by 01.ai. CoRR abs\/2403.04652 (2024). https:\/\/doi.org\/10.48550\/ARXIV.2403.04652","DOI":"10.48550\/ARXIV.2403.04652"},{"key":"17_CR29","doi-asserted-by":"crossref","unstructured":"Yue, X., et al.: Mmmu: a massive multi-discipline multimodal understanding and reasoning benchmark for expert agi. In: Proceedings of CVPR (2024)","DOI":"10.1109\/CVPR52733.2024.00913"},{"key":"17_CR30","doi-asserted-by":"publisher","unstructured":"Zhu, D., Chen, J., Shen, X., Li, X., Elhoseiny, M.: Minigpt-4: enhancing vision-language understanding with advanced large language models. CoRR abs\/2304.10592 (2023). https:\/\/doi.org\/10.48550\/ARXIV.2304.10592","DOI":"10.48550\/ARXIV.2304.10592"}],"container-title":["Lecture Notes in Computer Science","Natural Language Processing and Information Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-70239-6_17","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,28]],"date-time":"2024-11-28T10:23:58Z","timestamp":1732789438000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-70239-6_17"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"ISBN":["9783031702389","9783031702396"],"references-count":30,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-70239-6_17","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2024]]},"assertion":[{"value":"20 September 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"NLDB","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Applications of Natural Language to Information Systems","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Turin","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"25 June 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 June 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"nldb2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/nldb2024.di.unito.it\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}