{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T08:57:59Z","timestamp":1785488279596,"version":"3.56.0"},"publisher-location":"New York, NY, USA","reference-count":34,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,12,17]]},"DOI":"10.1145\/3774521.3774607","type":"proceedings-article","created":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T07:34:24Z","timestamp":1785483264000},"page":"1-8","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Improving Video Question Answering through query-based frame selection"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-0269-4451","authenticated-orcid":false,"given":"Himanshu","family":"Patil","sequence":"first","affiliation":[{"name":"Indian Institute of Technology Bombay, Mumbai, Maharashtra, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-8886-770X","authenticated-orcid":false,"given":"Geo","family":"Jolly","sequence":"additional","affiliation":[{"name":"Indian Institute of Technology Bombay, Mumbai, Maharashtra, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-7266-0657","authenticated-orcid":false,"given":"Ramana Raja","family":"Buddala","sequence":"additional","affiliation":[{"name":"Indian Institute of Technology Bombay, Mumbai, Maharashtra, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4533-2490","authenticated-orcid":false,"given":"Ganesh","family":"Ramakrishnan","sequence":"additional","affiliation":[{"name":"Indian Institute of Technology Bombay, Mumbai, Maharashtra, India"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,31]]},"reference":[{"key":"e_1_3_3_2_2_2","unstructured":"Jean-Baptiste Alayrac Jeff Donahue Paul Luc et\u00a0al. 2022. Flamingo: a Visual Language Model for Few-Shot Learning. (12 2022). https:\/\/arxiv.org\/pdf\/2204.14198.pdf"},{"key":"e_1_3_3_2_3_2","doi-asserted-by":"crossref","unstructured":"Max Bain Arsha Nagrani G\u00fcl Varol and Andrew Zisserman. 2021. Frozen in Time: A Joint Video and Image Encoder for End-to-End Video Understanding. (10 2021). https:\/\/arxiv.org\/pdf\/2104.00650.pdf","DOI":"10.1109\/ICCV48922.2021.00175"},{"key":"e_1_3_3_2_4_2","doi-asserted-by":"crossref","unstructured":"Lin Chen Xilin Wei Jinsong Li Xiaoyi Dong Pan Zhang Yuhang Zang Zehui Chen Haodong Duan Bin Lin Zhenyu Tang Li Yuan Yu Qiao Dahua Lin Feng Zhao and Jiaqi Wang. 2024. ShareGPT4Video: Improving Video Understanding and Generation with Better Captions. (6 2024). http:\/\/arxiv.org\/abs\/2406.04325","DOI":"10.52202\/079017-0614"},{"key":"e_1_3_3_2_5_2","unstructured":"Xi Chen Yinpeng Lu Liunian\u00a0Harold Wang et\u00a0al. 2023. PaLI: A Jointly-Scaled Multilingual Language-Image Model. (6 2023). https:\/\/arxiv.org\/pdf\/2209.06794.pdf"},{"key":"e_1_3_3_2_6_2","unstructured":"Moran Feldman Amin Karbasi and Ehsan Kazemi. 2018. Do Less Get More: Streaming Submodular Maximization with Subsampling. (2 2018). http:\/\/arxiv.org\/abs\/1802.07098"},{"key":"e_1_3_3_2_7_2","volume-title":"Advances in Neural Information Processing Systems","author":"Gong Boqing","year":"2014","unstructured":"Boqing Gong, Wei-Lun Chao, Kristen Grauman, and Fei Sha. 2014. Diverse Sequential Subset Selection for Supervised Video Summarization. In Advances in Neural Information Processing Systems , Z.\u00a0Ghahramani, M.\u00a0Welling, C.\u00a0Cortes, N.\u00a0Lawrence, and K.Q. Weinberger (Eds.), Vol.\u00a027. Curran Associates, Inc.https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2014\/file\/5d3b9e06117de70a7e5076cc3ed89e18-Paper.pdf"},{"key":"e_1_3_3_2_8_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298928"},{"key":"e_1_3_3_2_9_2","doi-asserted-by":"crossref","unstructured":"Wei Han Hui Chen Min-Yen Kan and Soujanya Poria. 2024. Self-Adaptive Sampling for Efficient Video Question-Answering on Image\u2013Text Models. (3 2024). http:\/\/arxiv.org\/abs\/2307.04192","DOI":"10.18653\/v1\/2024.findings-naacl.162"},{"key":"e_1_3_3_2_10_2","unstructured":"Christopher Harshaw Ehsan Kazemi Moran Feldman and Amin Karbasi. 2021. The Power of Subsampling in Submodular Maximization. (4 2021). http:\/\/arxiv.org\/abs\/2104.02772"},{"key":"e_1_3_3_2_11_2","doi-asserted-by":"publisher","unstructured":"Vishal Kaushal Rishabh Iyer Khoshrav Doctor Anurag Sahoo Pratik Dubal Suraj Kothawade Rohan Mahadev Kunal Dargan and Ganesh Ramakrishnan. 2019. Demystifying multi-faceted video summarization: Tradeoff between diversity representation coverage and importance. Proceedings - 2019 IEEE Winter Conference on Applications of Computer Vision WACV 2019 452\u2013461. 10.1109\/WACV.2019.00054","DOI":"10.1109\/WACV.2019.00054"},{"key":"e_1_3_3_2_12_2","doi-asserted-by":"publisher","unstructured":"Vishal Kaushal Suraj Kothawade Rishabh Iyer and Ganesh Ramakrishnan. 2020. Realistic Video Summarization through VISIOCITY: A New Benchmark and Evaluation Framework. AI4TV 2020 - Proceedings of the 2nd International Workshop on AI for Smart TV Content Production Access and Delivery 37\u201344. 10.1145\/3422839.3423064","DOI":"10.1145\/3422839.3423064"},{"key":"e_1_3_3_2_13_2","unstructured":"Vishal Kaushal Ganesh Ramakrishnan and Rishabh Iyer. 2022. Submodlib: A Submodular Optimization Library. arxiv:https:\/\/arXiv.org\/abs\/2202.10680\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2202.10680"},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"crossref","unstructured":"Jie Lei Tamara\u00a0L. Berg and Mohit Bansal. 2021. Less Is More: ClipBERT for Video-and-Language Learning via Sparse Sampling. (6 2021). https:\/\/openaccess.thecvf.com\/content\/CVPR2021\/papers\/Lei_Less_Is_More_ClipBERT_for_Video-and-Language_Learning_via_Sparse_Sampling_CVPR_2021_paper.pdf","DOI":"10.1109\/CVPR46437.2021.00725"},{"key":"e_1_3_3_2_15_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00725"},{"key":"e_1_3_3_2_16_2","unstructured":"Junnan Li Dongxu Li Silvio Savarese and Steven Hoi. 2023. BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models. arxiv:https:\/\/arXiv.org\/abs\/2301.12597\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2301.12597"},{"key":"e_1_3_3_2_17_2","unstructured":"Junnan Li Dongxu Li Silvio Savarese and Steven\u00a0C.H. Hoi. 2023. BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models. (1 2023). https:\/\/arxiv.org\/pdf\/2301.12597.pdf"},{"key":"e_1_3_3_2_18_2","unstructured":"Junnan Li Ramprasaath\u00a0R. Selvaraju Akhilesh Gotmare et\u00a0al. 2021. Align Before Fuse: Vision and Language Representation Learning with Momentum Distillation. (12 2021). https:\/\/arxiv.org\/pdf\/2107.07651.pdf"},{"key":"e_1_3_3_2_19_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02095"},{"key":"e_1_3_3_2_20_2","doi-asserted-by":"crossref","unstructured":"Linjie Li Mark Yatskar Da Yin et\u00a0al. 2020. HERO: Hierarchical Encoder for Video+Language Omni-representation Pre-training. (11 2020). https:\/\/arxiv.org\/pdf\/2005.00200.pdf","DOI":"10.18653\/v1\/2020.emnlp-main.161"},{"key":"e_1_3_3_2_21_2","unstructured":"Hao Liang Jiapeng Li Tianyi Bai Xijie Huang Linzhuang Sun Zhengren Wang Conghui He Bin Cui Chong Chen and Wentao Zhang. 2024. KeyVideoLLM: Towards Large-scale Video Keyframe Selection. (8 2024). http:\/\/arxiv.org\/abs\/2407.03104"},{"key":"e_1_3_3_2_22_2","unstructured":"Bin Lin Yang Ye Bin Zhu Jiaxi Cui Munan Ning Peng Jin and Li Yuan. 2024. Video-LLaVA: Learning United Visual Representation by Alignment Before Projection. arxiv:https:\/\/arXiv.org\/abs\/2311.10122\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2311.10122"},{"key":"e_1_3_3_2_23_2","unstructured":"Kevin Lin Dongxu Li Yucheng Wang Silvio Savarese and Steven\u00a0C.H. Hoi. 2023. Video-LLaVA: Learning Unified Video-Language Representation with LLaMA. (11 2023). https:\/\/arxiv.org\/pdf\/2311.10122.pdf"},{"key":"e_1_3_3_2_24_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.679"},{"key":"e_1_3_3_2_25_2","unstructured":"Baharan Mirzasoleiman Stefanie Jegelka and Andreas Krause. 2017. Streaming Non-monotone Submodular Maximization: Personalized Video Summarization on the Fly. (12 2017). http:\/\/arxiv.org\/abs\/1706.03583"},{"key":"e_1_3_3_2_26_2","unstructured":"Alec Radford Jong\u00a0Wook Kim Chris Hallacy Aditya Ramesh Gabriel Goh Sandhini Agarwal Girish Sastry Amanda Askell Pamela Mishkin Jack Clark Gretchen Krueger and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. (2 2021). http:\/\/arxiv.org\/abs\/2103.00020"},{"key":"e_1_3_3_2_27_2","unstructured":"Anurag Sahoo Vishal Kaushal Khoshrav Doctor Suyash Shetty Rishabh Iyer and Ganesh Ramakrishnan. 2017. A Unified Multi-Faceted Video Summarization System. (4 2017). http:\/\/arxiv.org\/abs\/1704.01466"},{"key":"e_1_3_3_2_28_2","unstructured":"Xi Tang Jihao Qiu Lingxi Xie Yunjie Tian Jianbin Jiao and Qixiang Ye. 2025. Adaptive Keyframe Sampling for Long Video Understanding. arxiv:https:\/\/arXiv.org\/abs\/2502.21271\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2502.21271"},{"key":"e_1_3_3_2_29_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73033-711"},{"key":"e_1_3_3_2_30_2","unstructured":"Yizhou Wang Meng Li Yizhou Wang et\u00a0al. 2023. ViLA: Learning to Select Frames via Question-aware Vision-Language Alignment. (12 2023). https:\/\/arxiv.org\/pdf\/2312.08367.pdf"},{"key":"e_1_3_3_2_31_2","unstructured":"Rowan Zellers Ari Holtzman Matthew Peters et\u00a0al. 2021. MERLOT: Multimodal Neural Script Knowledge Models. (12 2021). https:\/\/arxiv.org\/pdf\/2106.02636.pdf"},{"key":"e_1_3_3_2_32_2","unstructured":"Hang Zhang Xin Li and Lidong Bing. 2023. Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding. arxiv:https:\/\/arXiv.org\/abs\/2306.02858\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2306.02858"},{"key":"e_1_3_3_2_33_2","doi-asserted-by":"crossref","unstructured":"Haotian Zhang Qianyu Peng Jiahui Wu et\u00a0al. 2023. Video-LLaMA: An Audio-Visual Language Model for Video Understanding. (6 2023). https:\/\/arxiv.org\/pdf\/2306.02858.pdf","DOI":"10.18653\/v1\/2023.emnlp-demo.49"},{"key":"e_1_3_3_2_34_2","unstructured":"Yuanhan Zhang Bo Li haotian Liu Yong\u00a0jae Lee Liangke Gui Di Fu Jiashi Feng Ziwei Liu and Chunyuan Li. 2024. LLaVA-NeXT: A Strong Zero-shot Video Understanding Model. https:\/\/llava-vl.github.io\/blog\/2024-04-30-llava-next-video\/"},{"key":"e_1_3_3_2_35_2","unstructured":"Deyao Zhu Jun Chen Zhe Shen et\u00a0al. 2023. MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models. (4 2023). https:\/\/arxiv.org\/pdf\/2304.10592.pdf"}],"event":{"name":"ICVGIP 2025: Indian Conference on Computer Vision, Graphics, and Image Processing","location":"Mandi Himachal Pradesh India","acronym":"ICVGIP 2025"},"container-title":["Proceedings of the Sixteen Indian Conference on Computer Vision, Graphics and Image Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3774521.3774607","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T08:05:48Z","timestamp":1785485148000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3774521.3774607"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,17]]},"references-count":34,"alternative-id":["10.1145\/3774521.3774607","10.1145\/3774521"],"URL":"https:\/\/doi.org\/10.1145\/3774521.3774607","relation":{},"subject":[],"published":{"date-parts":[[2025,12,17]]},"assertion":[{"value":"2026-07-31","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}