{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T08:57:50Z","timestamp":1785488270690,"version":"3.56.0"},"publisher-location":"New York, NY, USA","reference-count":48,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,12,17]],"date-time":"2025-12-17T00:00:00Z","timestamp":1765929600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,12,17]]},"DOI":"10.1145\/3774521.3774596","type":"proceedings-article","created":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T07:34:24Z","timestamp":1785483264000},"page":"1-10","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["How Does India Cook Biryani?"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-6721-008X","authenticated-orcid":false,"given":"Farzana","family":"S","sequence":"first","affiliation":[{"name":"International Institute of Information Technology Hyderabad, Hyderabad, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-9727-3731","authenticated-orcid":false,"given":"C V","family":"Rishi","sequence":"additional","affiliation":[{"name":"International Institute of Information Technology Hyderabad, Hyderabad, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-3690-4490","authenticated-orcid":false,"given":"Shubham","family":"Goel","sequence":"additional","affiliation":[{"name":"International Institute of Information Technology Hyderabad, Hyderabad, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1187-9108","authenticated-orcid":false,"given":"Aditya","family":"Arun","sequence":"additional","affiliation":[{"name":"International Institute of Information Technology Hyderabad, Hyderabad, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6767-7057","authenticated-orcid":false,"given":"C V","family":"Jawahar","sequence":"additional","affiliation":[{"name":"International Institute of Information Technology Hyderabad, Hyderabad, India"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,31]]},"reference":[{"key":"e_1_3_3_3_2_2","doi-asserted-by":"publisher","DOI":"10.5281\/zenodo.4067897"},{"key":"e_1_3_3_3_3_2","unstructured":"Josh Achiam Steven Adler Sandhini Agarwal Lama Ahmad Ilge Akkaya Florencia\u00a0Leoni Aleman Diogo Almeida Janko Altenschmidt Sam Altman Shyamal Anadkat et\u00a0al. 2023. Gpt-4 technical report. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2303.08774 (2023)."},{"key":"e_1_3_3_3_4_2","doi-asserted-by":"crossref","unstructured":"Mahmoud Al-Faris John Chiverton David Ndzi and Ahmed\u00a0Isam Ahmed. 2020. A review on computer vision-based methods for human action recognition. Journal of imaging 6 6 (2020) 46.","DOI":"10.3390\/jimaging6060046"},{"key":"e_1_3_3_3_5_2","doi-asserted-by":"crossref","unstructured":"Vishu Antani and Santosh Mahapatra. 2022. Evolution of Indian cuisine: a socio-historical review. Journal of Ethnic Foods 9 1 (2022) 15.","DOI":"10.1186\/s42779-022-00129-4"},{"key":"e_1_3_3_3_6_2","doi-asserted-by":"crossref","unstructured":"Max Bain Jaesung Huh Tengda Han and Andrew Zisserman. 2023. Whisperx: Time-accurate speech transcription of long-form audio. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2303.00747 (2023).","DOI":"10.21437\/Interspeech.2023-78"},{"key":"e_1_3_3_3_7_2","doi-asserted-by":"publisher","DOI":"10.1109\/WCICA.2014.7053655"},{"key":"e_1_3_3_3_8_2","volume-title":"The Thirteenth International Conference on Learning Representations","author":"Burgess James","year":"2025","unstructured":"James Burgess, Xiaohan Wang, Yuhui Zhang, Anita Rau, Alejandro Lozano, Lisa Dunlap, Trevor Darrell, and Serena Yeung-Levy. 2025. Video Action Differencing. In The Thirteenth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=3bcN6xlO6f"},{"key":"e_1_3_3_3_9_2","first-page":"720","volume-title":"Proceedings of the European conference on computer vision (ECCV)","author":"Damen Dima","year":"2018","unstructured":"Dima Damen, Hazel Doughty, Giovanni\u00a0Maria Farinella, Sanja Fidler, Antonino Furnari, Evangelos Kazakos, Davide Moltisanti, Jonathan Munro, Toby Perrett, Will Price, et\u00a0al. 2018. Scaling egocentric vision: The epic-kitchens dataset. In Proceedings of the European conference on computer vision (ECCV). 720\u2013736."},{"key":"e_1_3_3_3_10_2","doi-asserted-by":"crossref","unstructured":"Mohamed\u00a0M Elgamml Fazly\u00a0S Abas and H\u00a0Ann Goh. 2020. Semantic analysis in soccer videos using support vector machine. International Journal of Pattern Recognition and Artificial Intelligence 34 09 (2020) 2055018.","DOI":"10.1142\/S0218001420550186"},{"key":"e_1_3_3_3_11_2","doi-asserted-by":"crossref","unstructured":"Junyu Gao and Changsheng Xu. 2021. Learning video moment retrieval without a single annotated video. IEEE Transactions on Circuits and Systems for Video Technology 32 3 (2021) 1646\u20131657.","DOI":"10.1109\/TCSVT.2021.3075470"},{"key":"e_1_3_3_3_12_2","unstructured":"Aaron Grattafiori Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Alex Vaughan et\u00a0al. 2024. The llama 3 herd of models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2407.21783 (2024)."},{"key":"e_1_3_3_3_13_2","unstructured":"Matthew Honnibal Ines Montani Sofie Van\u00a0Landeghem Adriane Boyd et\u00a0al. 2020. spaCy: Industrial-strength natural language processing in python. (2020)."},{"key":"e_1_3_3_3_14_2","doi-asserted-by":"publisher","DOI":"10.5281\/zenodo.5143773"},{"key":"e_1_3_3_3_15_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.149"},{"key":"e_1_3_3_3_16_2","doi-asserted-by":"crossref","unstructured":"Licheng Jiao Ruohan Zhang Fang Liu Shuyuan Yang Biao Hou Lingling Li and Xu Tang. 2021. New generation deep learning for video object detection: A survey. IEEE Transactions on Neural Networks and Learning Systems 33 8 (2021) 3195\u20133215.","DOI":"10.1109\/TNNLS.2021.3053249"},{"key":"e_1_3_3_3_17_2","unstructured":"Bo Li Yuanhan Zhang Dong Guo Renrui Zhang Feng Li Hao Zhang Kaichen Zhang Peiyuan Zhang Yanwei Li Ziwei Liu et\u00a0al. 2024. Llava-onevision: Easy visual task transfer. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2408.03326 (2024)."},{"key":"e_1_3_3_3_18_2","doi-asserted-by":"publisher","DOI":"10.1145\/3706599.3720172"},{"key":"e_1_3_3_3_19_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"e_1_3_3_3_20_2","doi-asserted-by":"publisher","DOI":"10.1145\/500141.500217"},{"key":"e_1_3_3_3_21_2","unstructured":"Muhammad Maaz Hanoona Rasheed Salman Khan and Fahad\u00a0Shahbaz Khan. 2023. Video-chatgpt: Towards detailed video understanding via large vision and language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2306.05424 (2023)."},{"key":"e_1_3_3_3_22_2","doi-asserted-by":"crossref","unstructured":"Jonathan Malmaud Jonathan Huang Vivek Rathod Nick Johnston Andrew Rabinovich and Kevin Murphy. 2015. What\u2019s cookin\u2019? interpreting cooking videos using text speech and vision. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1503.01558 (2015).","DOI":"10.3115\/v1\/N15-1015"},{"key":"e_1_3_3_3_23_2","doi-asserted-by":"crossref","unstructured":"Zhang Min-qing and Li Wen-ping. 2021. An automatic classification method of sports teaching video using support vector machine. Scientific programming 2021 1 (2021) 4728584.","DOI":"10.1155\/2021\/4728584"},{"key":"e_1_3_3_3_24_2","doi-asserted-by":"crossref","unstructured":"Loris Nanni Stefano Ghidoni and Sheryl Brahnam. 2017. Handcrafted vs. non-handcrafted features for computer vision classification. Pattern recognition 71 (2017) 158\u2013172.","DOI":"10.1016\/j.patcog.2017.05.025"},{"key":"e_1_3_3_3_25_2","doi-asserted-by":"crossref","unstructured":"Pradyumna Narayana J\u00a0Ross Beveridge and Bruce\u00a0A Draper. 2018. Interacting Hidden Markov Models for Video Understanding. International Journal of Pattern Recognition and Artificial Intelligence 32 11 (2018) 1855020.","DOI":"10.1142\/S0218001418550200"},{"key":"e_1_3_3_3_26_2","doi-asserted-by":"publisher","DOI":"10.1109\/MECO.2017.7977207"},{"key":"e_1_3_3_3_27_2","doi-asserted-by":"crossref","unstructured":"Taichi Nishimura Atsushi Hashimoto Yoshitaka Ushiku Hirotaka Kameko and Shinsuke Mori. 2024. Recipe generation from unsegmented cooking videos. ACM Transactions on Multimedia Computing Communications and Applications (2024).","DOI":"10.1145\/3649137"},{"key":"e_1_3_3_3_28_2","doi-asserted-by":"publisher","DOI":"10.1109\/IEEECONF59524.2023.10477002"},{"key":"e_1_3_3_3_29_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.02226"},{"key":"e_1_3_3_3_30_2","first-page":"28492","volume-title":"International conference on machine learning","author":"Radford Alec","year":"2023","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Tao Xu, Greg Brockman, Christine McLeavey, and Ilya Sutskever. 2023. Robust speech recognition via large-scale weak supervision. In International conference on machine learning. PMLR, 28492\u201328518."},{"key":"e_1_3_3_3_31_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1410"},{"key":"e_1_3_3_3_32_2","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475263"},{"key":"e_1_3_3_3_33_2","doi-asserted-by":"crossref","unstructured":"Vijeta Sharma Manjari Gupta Ajai Kumar and Deepti Mishra. 2021. Video processing using deep learning techniques: A systematic literature review. IEEE Access 9 (2021) 139489\u2013139507.","DOI":"10.1109\/ACCESS.2021.3118541"},{"key":"e_1_3_3_3_34_2","doi-asserted-by":"crossref","unstructured":"Tulasi Srinivas. 2011. Exploring Indian culture through food. Education about Asia 16 3 (2011) 38\u201341.","DOI":"10.65959\/eaa.1055"},{"key":"e_1_3_3_3_35_2","unstructured":"Yunlong Tang Jing Bi Siting Xu Luchuan Song Susan Liang Teng Wang Daoan Zhang Jie An Jingyang Lin Rongyi Zhu et\u00a0al. 2025. Video understanding with large language models: A survey. IEEE Transactions on Circuits and Systems for Video Technology (2025)."},{"key":"e_1_3_3_3_36_2","unstructured":"Gemini Team. 2025. Gemini 2.5: Pushing the Frontier with Advanced Reasoning Multimodality Long Context and Next Generation Agentic Capabilities. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2507.06261 (July 2025)."},{"key":"e_1_3_3_3_37_2","unstructured":"Gemini\u00a0Robotics Team Saminda Abeyruwan Joshua Ainslie Jean-Baptiste Alayrac Montserrat\u00a0Gonzalez Arenas Travis Armstrong Ashwin Balakrishna Robert Baruch Maria Bauza Michiel Blokzijl et\u00a0al. 2025. Gemini robotics: Bringing ai into the physical world. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.20020 (2025)."},{"key":"e_1_3_3_3_38_2","unstructured":"Qwen Team. 2024. Qwen2.5: A Party of Foundation Models. https:\/\/qwenlm.github.io\/blog\/qwen2.5\/"},{"key":"e_1_3_3_3_39_2","unstructured":"Peng Wang Shuai Bai Sinan Tan Shijie Wang Zhihao Fan Jinze Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge Yang Fan Kai Dang Mengfei Du Xuancheng Ren Rui Men Dayiheng Liu Chang Zhou Jingren Zhou and Junyang Lin. 2024. Qwen2-VL: Enhancing Vision-Language Model\u2019s Perception of the World at Any Resolution. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2409.12191 (2024)."},{"key":"e_1_3_3_3_40_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00965"},{"key":"e_1_3_3_3_41_2","unstructured":"Shitao Xiao Zheng Liu Peitian Zhang and Niklas Muennighoff. 2023. C-Pack: Packaged Resources To Advance General Chinese Embedding. arxiv:https:\/\/arXiv.org\/abs\/2309.07597\u00a0[cs.CL]"},{"key":"e_1_3_3_3_42_2","unstructured":"Lexing Xie Shih-Fu Chang Ajay Divakaran and Huifang Sun. 2002. Learning hierarchical hidden Markov models for video structure discovery. ADVENT Group Columbia Univ. New York Tech. Rep 6 (2002)."},{"key":"e_1_3_3_3_43_2","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123427"},{"key":"e_1_3_3_3_44_2","unstructured":"Frank\u00a0F Xu Lei Ji Botian Shi Junyi Du Graham Neubig Yonatan Bisk and Nan Duan. 2020. A benchmark for structured procedural knowledge extraction from cooking videos. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2005.00706 (2020)."},{"key":"e_1_3_3_3_45_2","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548291"},{"key":"e_1_3_3_3_46_2","unstructured":"Boqiang Zhang Kehan Li Zesen Cheng Zhiqiang Hu Yuqian Yuan Guanzheng Chen Sicong Leng Yuming Jiang Hang Zhang Xin Li et\u00a0al. 2025. Videollama 3: Frontier multimodal foundation models for image and video understanding. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2501.13106 (2025)."},{"key":"e_1_3_3_3_47_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.12342"},{"key":"e_1_3_3_3_48_2","unstructured":"Deyao Zhu Jun Chen Xiaoqian Shen Xiang Li and Mohamed Elhoseiny. 2023. Minigpt-4: Enhancing vision-language understanding with advanced large language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2304.10592 (2023)."},{"key":"e_1_3_3_3_49_2","unstructured":"Jinguo Zhu Weiyun Wang Zhe Chen Zhaoyang Liu Shenglong Ye Lixin Gu Hao Tian Yuchen Duan Weijie Su Jie Shao et\u00a0al. 2025. Internvl3: Exploring advanced training and test-time recipes for open-source multimodal models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2504.10479 (2025)."}],"event":{"name":"ICVGIP 2025: Indian Conference on Computer Vision, Graphics, and Image Processing","location":"Mandi Himachal Pradesh India","acronym":"ICVGIP 2025"},"container-title":["Proceedings of the Sixteen Indian Conference on Computer Vision, Graphics and Image Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3774521.3774596","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T08:04:13Z","timestamp":1785485053000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3774521.3774596"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,17]]},"references-count":48,"alternative-id":["10.1145\/3774521.3774596","10.1145\/3774521"],"URL":"https:\/\/doi.org\/10.1145\/3774521.3774596","relation":{},"subject":[],"published":{"date-parts":[[2025,12,17]]},"assertion":[{"value":"2026-07-31","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}