{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,25]],"date-time":"2026-07-25T01:04:15Z","timestamp":1784941455753,"version":"3.55.0"},"reference-count":57,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"DOI":"10.13039\/501100003725","name":"National Research Foundation of Korea","doi-asserted-by":"publisher","award":["NRF-2020R1A2C2007139"],"award-info":[{"award-number":["NRF-2020R1A2C2007139"]}],"id":[{"id":"10.13039\/501100003725","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Institute of Information & Communications Technology Planning & Evaluation","award":["RS-2021-II211343"],"award-info":[{"award-number":["RS-2021-II211343"]}]},{"name":"Institute of Information & Communications Technology Planning & Evaluation","award":["RS-2023-00235293"],"award-info":[{"award-number":["RS-2023-00235293"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Access"],"published-print":{"date-parts":[[2024]]},"DOI":"10.1109\/access.2024.3517625","type":"journal-article","created":{"date-parts":[[2024,12,16]],"date-time":"2024-12-16T19:28:05Z","timestamp":1734377285000},"page":"193057-193075","source":"Crossref","is-referenced-by-count":45,"title":["An Image Grid Can Be Worth a Video: Zero-Shot Video Question Answering Using a VLM"],"prefix":"10.1109","volume":"12","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-2318-9554","authenticated-orcid":false,"given":"Wonkyun","family":"Kim","sequence":"first","affiliation":[{"name":"Department of Intelligence and Information, Seoul National University, Gwanak-gu, Seoul, South Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Changin","family":"Choi","sequence":"additional","affiliation":[{"name":"Interdisciplinary Program in Artificial Intelligence, Seoul National University, Gwanak-gu, Seoul, South Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-8995-339X","authenticated-orcid":false,"given":"Wonseok","family":"Lee","sequence":"additional","affiliation":[{"name":"Interdisciplinary Program in Artificial Intelligence, Seoul National University, Gwanak-gu, Seoul, South Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2590-8774","authenticated-orcid":false,"given":"Wonjong","family":"Rhee","sequence":"additional","affiliation":[{"name":"Department of Intelligence and Information, Seoul National University, Gwanak-gu, Seoul, South Korea"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","first-page":"1877","article-title":"Language models are few-shot learners","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"33","author":"Brown"},{"issue":"240","key":"ref2","first-page":"1","article-title":"PaLM: Scaling language modeling with pathways","volume":"24","author":"Chowdhery","year":"2022","journal-title":"J. Mach. Learn. Res."},{"key":"ref3","article-title":"PaLM 2 technical report","volume-title":"arXiv:2305.10403","author":"Anil","year":"2023"},{"key":"ref4","article-title":"GPT-4 technical report","volume-title":"arXiv:2303.08774","author":"Achiam","year":"2023"},{"key":"ref5","article-title":"LLaMA: Open and efficient foundation language models","author":"Touvron","year":"2023","journal-title":"arXiv:2302.13971"},{"key":"ref6","first-page":"23716","article-title":"Flamingo: A visual language model for few-shot learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"35","author":"Alayrac"},{"key":"ref7","first-page":"1","article-title":"SimVLM: Simple visual language model pretraining with weak supervision","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Wang"},{"key":"ref8","first-page":"1","article-title":"CoCa: Contrastive captioners are image-text foundation models","volume":"2022","author":"Yu","year":"2022","journal-title":"Trans. Mach. Learn. Res."},{"key":"ref9","first-page":"1","article-title":"BLIP: Bootstrapping language-image pre-training for unified vision-language understanding and generation","volume-title":"Proc. Porc. Int. Conf. Mach. Learn.","author":"Li"},{"key":"ref10","first-page":"1","article-title":"BLIP-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Li"},{"key":"ref11","article-title":"MPLUG-owl: Modularization empowers large language models with multimodality","author":"Ye","year":"2023","journal-title":"arXiv:2304.14178"},{"key":"ref12","article-title":"InstructBLIP: Towards general-purpose vision-language models with instruction tuning","author":"Dai","year":"2023","journal-title":"arXiv:2305.06500"},{"key":"ref13","first-page":"1","article-title":"Visual instruction tuning","volume-title":"Proc. NeurIPS","author":"Liu"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i21.30570"},{"key":"ref15","first-page":"1","article-title":"Pengi: An audio language model for audio tasks","volume-title":"Proc. Adv. Neural Inform. Process. Syst.","author":"Deshmukh"},{"key":"ref16","article-title":"SALMONN: Towards generic hearing abilities for large language models","author":"Tang","year":"2023","journal-title":"arXiv:2310.13289"},{"key":"ref17","article-title":"Qwen-audio: Advancing universal audio understanding via unified large-scale audio-language models","author":"Chu","year":"2023","journal-title":"arXiv:2311.07919"},{"key":"ref18","article-title":"VideoChat: Chat-centric video understanding","author":"Li","year":"2023","journal-title":"arXiv:2305.06355"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-demo.49"},{"key":"ref20","article-title":"Video-ChatGPT: Towards detailed video understanding via large vision and language models","author":"Maaz","year":"2023","journal-title":"arXiv:2306.05424"},{"key":"ref21","article-title":"Video-LLaVA: Learning united visual representation by alignment before projection","author":"Lin","year":"2023","journal-title":"arXiv:2311.10122"},{"key":"ref22","first-page":"124","article-title":"Zero-shot video question answering via frozen bidirectional language models","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Yang"},{"key":"ref23","article-title":"LLaMA-VID: An image is worth 2 tokens in large language models","author":"Li","year":"2023","journal-title":"arXiv:2311.17043"},{"key":"ref24","article-title":"Valley: Video assistant with large language model enhanced ability","author":"Luo","year":"2023","journal-title":"arXiv:2306.07207"},{"key":"ref25","article-title":"Vista-LLaMA: Reliable video narrator via equal distance to visual tokens","author":"Ma","year":"2023","journal-title":"arXiv:2312.08870"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02095"},{"key":"ref27","article-title":"MovieChat: From dense token to sparse memory for long video understanding","author":"Song","year":"2023","journal-title":"arXiv:2307.16449"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01300"},{"key":"ref29","article-title":"ViLA: Efficient video-language alignment via frame prompting and distilling for video question answering","author":"Wang","year":"2023","journal-title":"arXiv:2312.08367"},{"key":"ref30","first-page":"1","article-title":"Self-chained image-language model for video localization and question answering","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"36","author":"Yu"},{"key":"ref31","article-title":"Zero-shot video question answering with procedural programs","author":"Choudhury","year":"2023","journal-title":"arXiv:2312.00937"},{"key":"ref32","article-title":"A simple LLM framework for long-range video question-answering","author":"Zhang","year":"2023","journal-title":"arXiv:2312.17235"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01032"},{"key":"ref34","first-page":"1","article-title":"VidChapters-7m: Video chapters at scale","volume-title":"Proc. NeurIPS","volume":"36","author":"Yang"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01046"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01438"},{"key":"ref37","first-page":"17612","article-title":"Mind the gap: Understanding the modality gap in multi-modal contrastive representation learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Liang"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-emnlp.189"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01092"},{"key":"ref40","article-title":"Evaluating large language models trained on code","author":"Chen","year":"2021","journal-title":"arXiv:2107.03374"},{"key":"ref41","article-title":"LLaMA-adapter: Efficient fine-tuning of language models with zero-init attention","author":"Zhang","year":"2023","journal-title":"arXiv:2303.16199"},{"key":"ref42","article-title":"CogAgent: A visual language model for GUI agents","author":"Hong","year":"2023","journal-title":"arXiv:2312.08914"},{"key":"ref43","article-title":"LLaVA-NeXT: Improved reasoning, OCR, and world knowledge","author":"Liu","year":"2024"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123427"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33019127"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.502"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00965"},{"key":"ref48","first-page":"1","article-title":"Star: A benchmark for situated reasoning in real-world videos","volume-title":"Proc. 35th Conf. Neural Inf. Process. Syst.","author":"Wu"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/d18-1167"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01099"},{"key":"ref51","first-page":"1","article-title":"EgoSchema: A diagnostic benchmark for very long-form video language understanding","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Mangalam"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01842"},{"key":"ref53","article-title":"InternVideo: General video foundation models via generative and discriminative learning","author":"Wang","year":"2022","journal-title":"arXiv:2212.03191"},{"key":"ref54","article-title":"Can I trust your answer? Visually grounded video question answering","author":"Xiao","year":"2023","journal-title":"arXiv:2309.01327"},{"key":"ref55","first-page":"22199","article-title":"Large language models are zero-shot reasoners","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"35","author":"Kojima"},{"key":"ref56","first-page":"1","article-title":"Self-consistency improves chain of thought reasoning in language models","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Wang"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.147"}],"container-title":["IEEE Access"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/6287639\/10380310\/10802898.pdf?arnumber=10802898","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,25]],"date-time":"2024-12-25T06:19:11Z","timestamp":1735107551000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10802898\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"references-count":57,"URL":"https:\/\/doi.org\/10.1109\/access.2024.3517625","relation":{},"ISSN":["2169-3536"],"issn-type":[{"value":"2169-3536","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024]]}}}