{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,4]],"date-time":"2026-08-04T15:36:09Z","timestamp":1785857769672,"version":"3.56.0"},"publisher-location":"New York, NY, USA","reference-count":46,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,12,17]],"date-time":"2025-12-17T00:00:00Z","timestamp":1765929600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,12,17]]},"DOI":"10.1145\/3774521.3774575","type":"proceedings-article","created":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T07:34:24Z","timestamp":1785483264000},"page":"1-9","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Are We There Yet? Exploring the Capabilities of MLLMs in Assistive AI Applications"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-5622-0209","authenticated-orcid":false,"given":"Shayon","family":"Dasgupta","sequence":"first","affiliation":[{"name":"IIT (BHU) Varanasi, Varanasi, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5633-1843","authenticated-orcid":false,"given":"Avijit","family":"Dasgupta","sequence":"additional","affiliation":[{"name":"IIIT Hyderabad, Hyderabad, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6767-7057","authenticated-orcid":false,"given":"Jawahar","family":"C. V.","sequence":"additional","affiliation":[{"name":"IIIT Hyderabad, Hyderabad, India"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,31]]},"reference":[{"key":"e_1_3_3_3_2_2","unstructured":"Marah Abdin Jyoti Aneja Harkirat Behl S\u00e9bastien Bubeck Ronen Eldan Suriya Gunasekar Michael Harrison Russell\u00a0J Hewett Mojan Javaheripi Piero Kauffmann et\u00a0al. 2024. Phi-4 technical report. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2412.08905 (2024)."},{"key":"e_1_3_3_3_3_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00103"},{"key":"e_1_3_3_3_4_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00959"},{"key":"e_1_3_3_3_5_2","unstructured":"Shuai Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge Sibo Song Kai Dang Peng Wang Shijie Wang Jun Tang et\u00a0al. 2025. Qwen2. 5-vl technical report. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2502.13923 (2025)."},{"key":"e_1_3_3_3_6_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00439"},{"key":"e_1_3_3_3_7_2","unstructured":"Matthew\u00a0J Burton Jacqueline Ramke Ana\u00a0Patricia Marques Rupert\u00a0RA Bourne Nathan Congdon Iain Jones Brandon AM\u00a0Ah Tong Simon Arunga Damodar Bachani Covadonga Bascaran et\u00a0al. 2021. The lancet global health commission on global eye health: vision beyond 2020. The Lancet Global Health (2021)."},{"key":"e_1_3_3_3_8_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46454-1_49"},{"key":"e_1_3_3_3_9_2","doi-asserted-by":"crossref","unstructured":"Moran Cerf E\u00a0Paxon Frady and Christof Koch. 2009. Faces and text attract gaze independent of the task: Experimental data and computer model. Journal of vision (2009).","DOI":"10.1167\/9.12.10"},{"key":"e_1_3_3_3_10_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02506"},{"key":"e_1_3_3_3_11_2","volume-title":"ECCV","author":"Damen Dima","year":"2018","unstructured":"Dima Damen, Hazel Doughty, Giovanni\u00a0Maria Farinella, Sanja Fidler, Antonino Furnari, Evangelos Kazakos, Davide Moltisanti, Jonathan Munro, Toby Perrett, Will Price, et\u00a0al. 2018. Scaling egocentric vision: The epic-kitchens dataset. In ECCV."},{"key":"e_1_3_3_3_12_2","unstructured":"Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Amy Yang Angela Fan et\u00a0al. 2024. The llama 3 herd of models. arXiv e-prints (2024) arXiv\u20132407."},{"key":"e_1_3_3_3_13_2","doi-asserted-by":"crossref","unstructured":"Timothy\u00a0R Fricke Nina Tahhan Serge Resnikoff Eric Papas Anthea Burnett Suit\u00a0May Ho Thomas Naduvilath and Kovin\u00a0S Naidoo. 2018. Global prevalence of presbyopia and vision impairment from uncorrected presbyopia: systematic review meta-analysis and modelling. Ophthalmology (2018).","DOI":"10.1016\/j.ophtha.2018.04.013"},{"key":"e_1_3_3_3_14_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01842"},{"key":"e_1_3_3_3_15_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01834"},{"key":"e_1_3_3_3_16_2","unstructured":"Daya Guo Dejian Yang Haowei Zhang Junxiao Song Ruoyu Zhang Runxin Xu Qihao Zhu Shirong Ma Peiyi Wang Xiao Bi et\u00a0al. 2025. Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2501.12948 (2025)."},{"key":"e_1_3_3_3_17_2","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548112"},{"key":"e_1_3_3_3_18_2","doi-asserted-by":"crossref","unstructured":"Antonia Karamolegkou Malvina Nikandrou Georgios Pantazopoulos Danae\u00a0Sanchez Villegas Phillip Rust Ruchira Dhar Daniel Hershcovich and Anders S\u00f8gaard. 2025. Evaluating Multimodal Language Models as Visual Assistants for Visually Impaired Users. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.22610 (2025).","DOI":"10.18653\/v1\/2025.acl-long.1260"},{"key":"e_1_3_3_3_19_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19815-1_29"},{"key":"e_1_3_3_3_20_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00915"},{"key":"e_1_3_3_3_21_2","unstructured":"Haitao Li Qian Dong Junjie Chen Huixue Su Yujia Zhou Qingyao Ai Ziyi Ye and Yiqun Liu. 2024. Llms-as-judges: a comprehensive survey on llm-based evaluation methods. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2412.05579 (2024)."},{"key":"e_1_3_3_3_22_2","volume-title":"ICML","author":"Li Junnan","year":"2023","unstructured":"Junnan Li, Dongxu Li, Silvio Savarese, and Steven Hoi. 2023. Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models. In ICML."},{"key":"e_1_3_3_3_23_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6812"},{"key":"e_1_3_3_3_24_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"e_1_3_3_3_25_2","unstructured":"Haotian Liu Chunyuan Li Yuheng Li Bo Li Yuanhan Zhang Sheng Shen and Yong\u00a0Jae Lee. 2024. LLaVA-NeXT: Improved reasoning OCR and world knowledge. https:\/\/llava-vl.github.io\/blog\/2024-01-30-llava-next\/"},{"key":"e_1_3_3_3_26_2","unstructured":"Haotian Liu Chunyuan Li Qingyang Wu and Yong\u00a0Jae Lee. 2023. Visual Instruction Tuning."},{"key":"e_1_3_3_3_27_2","unstructured":"Haotian Liu Chunyuan Li Qingyang Wu and Yong\u00a0Jae Lee. 2023. Visual instruction tuning. NeurIPS (2023)."},{"key":"e_1_3_3_3_28_2","doi-asserted-by":"crossref","unstructured":"Sunny Mannava Rishi\u00a0Raj Borah and BR Shamanna. 2022. Current estimates of the economic burden of blindness and visual impairment in India: A cost of illness study. Indian Journal of Ophthalmology (2022).","DOI":"10.4103\/ijo.IJO_2804_21"},{"key":"e_1_3_3_3_29_2","doi-asserted-by":"publisher","DOI":"10.1109\/WACV48630.2021.00225"},{"key":"e_1_3_3_3_30_2","unstructured":"Vision Loss Expert\u00a0Group of\u00a0the Global Burden\u00a0of Disease\u00a0Study et\u00a0al. 2024. Global estimates on the number of people blind or visually impaired by cataract: a meta-analysis from 2000 to 2020. Eye (2024)."},{"key":"e_1_3_3_3_31_2","volume-title":"CVPR","author":"Peng Ruotian","year":"2025","unstructured":"Ruotian Peng, Haiying He, Yake Wei, Yandong Wen, and Di Hu. 2025. Patch Matters: Training-free Fine-grained Image Caption Enhancement via Local Perception. In CVPR."},{"key":"e_1_3_3_3_32_2","unstructured":"Zhiliang Peng Wenhui Wang Li Dong Yaru Hao Shaohan Huang Shuming Ma and Furu Wei. 2023. Kosmos-2: Grounding multimodal large language models to the world. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2306.14824 (2023)."},{"key":"e_1_3_3_3_33_2","volume-title":"ICML","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et\u00a0al. 2021. Learning transferable visual models from natural language supervision. In ICML."},{"key":"e_1_3_3_3_34_2","unstructured":"Hanoona Rasheed Muhammad Maaz Salman Khan and Fahad\u00a0S. Khan. 2024. LLaVA++: Extending Visual Capabilities with LLaMA-3 and Phi-3. https:\/\/github.com\/mbzuai-oryx\/LLaVA-pp"},{"key":"e_1_3_3_3_35_2","unstructured":"Baoguang Shi Xiang Bai and Cong Yao. 2016. An end-to-end trainable neural network for image-based sequence recognition and its application to scene text recognition. PAMI (2016)."},{"key":"e_1_3_3_3_36_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.452"},{"key":"e_1_3_3_3_37_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00851"},{"key":"e_1_3_3_3_38_2","doi-asserted-by":"publisher","DOI":"10.5555\/1304596.1304846"},{"key":"e_1_3_3_3_39_2","unstructured":"Gemma Team Aishwarya Kamath Johan Ferret Shreya Pathak Nino Vieillard Ramona Merhej Sarah Perrin Tatiana Matejovicova Alexandre Ram\u00e9 Morgane Rivi\u00e8re et\u00a0al. 2025. Gemma 3 technical report. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.19786 (2025)."},{"key":"e_1_3_3_3_40_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01841"},{"key":"e_1_3_3_3_41_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01841"},{"key":"e_1_3_3_3_42_2","unstructured":"Andreas Veit Tomas Matera Lukas Neumann Jiri Matas and Serge Belongie. 2016. Coco-text: Dataset and benchmark for text detection and recognition in natural images. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1601.07140 (2016)."},{"key":"e_1_3_3_3_43_2","unstructured":"Hsueh-Cheng Wang and Marc Pomplun. 2012. The attraction of visual attention to texts in real-world scenes. Journal of vision (2012)."},{"key":"e_1_3_3_3_44_2","unstructured":"An Yang Anfeng Li Baosong Yang Beichen Zhang Binyuan Hui Bo Zheng Bowen Yu Chang Gao Chengen Huang Chenxu Lv et\u00a0al. 2025. Qwen3 technical report. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2505.09388 (2025)."},{"key":"e_1_3_3_3_45_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01339"},{"key":"e_1_3_3_3_46_2","unstructured":"Xinlu Zhang Yujie Lu Weizhi Wang An Yan Jun Yan Lianke Qin Heng Wang Xifeng Yan William\u00a0Yang Wang and Linda\u00a0Ruth Petzold. 2023. Gpt-4v (ision) as a generalist evaluator for vision-language tasks. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2311.01361 (2023)."},{"key":"e_1_3_3_3_47_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.283"}],"event":{"name":"ICVGIP 2025: Indian Conference on Computer Vision, Graphics, and Image Processing","location":"Mandi Himachal Pradesh India","acronym":"ICVGIP 2025"},"container-title":["Proceedings of the Sixteen Indian Conference on Computer Vision, Graphics and Image Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3774521.3774575","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T08:06:09Z","timestamp":1785485169000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3774521.3774575"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,17]]},"references-count":46,"alternative-id":["10.1145\/3774521.3774575","10.1145\/3774521"],"URL":"https:\/\/doi.org\/10.1145\/3774521.3774575","relation":{},"subject":[],"published":{"date-parts":[[2025,12,17]]},"assertion":[{"value":"2026-07-31","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}