{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T08:58:03Z","timestamp":1785488283784,"version":"3.56.0"},"publisher-location":"New York, NY, USA","reference-count":24,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,12,17]],"date-time":"2025-12-17T00:00:00Z","timestamp":1765929600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,12,17]]},"DOI":"10.1145\/3774521.3774572","type":"proceedings-article","created":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T07:34:24Z","timestamp":1785483264000},"page":"1-9","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Zero-Shot Autism Behaviour Recognition Leveraging Multimodal Representations and Cross-Dataset Evaluation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9539-4385","authenticated-orcid":false,"given":"Darshan","family":"Gera","sequence":"first","affiliation":[{"name":"Sri Sathya Sai Institute of Higher Learning, Bengaluru, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-8565-8558","authenticated-orcid":false,"given":"Vignesh Sai Sankalp","family":"Sham","sequence":"additional","affiliation":[{"name":"Sri Sathya Sai Institute of Higher Learning, Puttaparthi, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-3950-6390","authenticated-orcid":false,"given":"Ritu","family":"Raj Pradhan","sequence":"additional","affiliation":[{"name":"Sri Sathya Sai Institute of Higher Learning, Puttaparthi, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8947-4840","authenticated-orcid":false,"given":"S","family":"Balasubramanian","sequence":"additional","affiliation":[{"name":"Sri Sathya Sai Institute of Higher Learning, Puttaparthi, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6323-9789","authenticated-orcid":false,"given":"Ravi","family":"Mukkamala","sequence":"additional","affiliation":[{"name":"https:\/\/www.cs.odu.edu\/, Norfolk, Virginia, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-0189-3538","authenticated-orcid":false,"given":"Sreepada V Satya N","family":"Sarma","sequence":"additional","affiliation":[{"name":"https:\/\/www.sssihl.edu.in\/, Puttaparthi, India"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,31]]},"reference":[{"key":"e_1_3_3_2_2_2","doi-asserted-by":"publisher","DOI":"10.5220\/0010839200003124"},{"key":"e_1_3_3_2_3_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"key":"e_1_3_3_2_4_2","unstructured":"Wenliang Dai Junnan Li Dongxu Li Anthony Meng\u00a0Huat Tiong Junqi Zhao Weisheng Wang Boyang Li Pascale Fung and Steven Hoi. 2023. InstructBLIP: Towards General-purpose Vision-Language Models with Instruction Tuning. arxiv:https:\/\/arXiv.org\/abs\/2305.06500\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2305.06500"},{"key":"e_1_3_3_2_5_2","doi-asserted-by":"crossref","unstructured":"Andong Deng Taojiannan Yang Chen Chen Qian Chen Leslie Neely and Sakiko Oyama. 2024. Language-assisted deep learning for autistic behaviors recognition. Smart Health 32 (2024) 100444.","DOI":"10.1016\/j.smhl.2023.100444"},{"key":"e_1_3_3_2_6_2","unstructured":"Shijian Deng Erin\u00a0E Kosloski Siddhi Patel Zeke\u00a0A Barnett Yiyang Nan Alexander Kaplan Sisira Aarukapalli William\u00a0T Doan Matthew Wang Harsh Singh et\u00a0al. 2024. Hear Me See Me Understand Me: Audio-Visual Autism Behavior Recognition. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.02554 (2024)."},{"key":"e_1_3_3_2_7_2","unstructured":"Edward\u00a0J Hu Yelong Shen Phillip Wallis Zeyuan Allen-Zhu Yuanzhi Li Shean Wang Lu Wang Weizhu Chen et\u00a0al. 2022. Lora: Low-rank adaptation of large language models. ICLR 1 2 (2022) 3."},{"key":"e_1_3_3_2_8_2","unstructured":"Will Kay Joao Carreira Karen Simonyan Brian Zhang Chloe Hillier Sudheendra Vijayanarasimhan Fabio Viola Tim Green Trevor Back Paul Natsev et\u00a0al. 2017. The kinetics human action video dataset. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1705.06950 (2017)."},{"key":"e_1_3_3_2_9_2","unstructured":"Bin Lin Yang Ye Bin Zhu Jiaxi Cui Munan Ning Peng Jin and Li Yuan. 2023. Video-llava: Learning united visual representation by alignment before projection. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2311.10122 (2023)."},{"key":"e_1_3_3_2_10_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"e_1_3_3_2_11_2","volume-title":"Forty-first International Conference on Machine Learning","author":"Liu Shih-Yang","year":"2024","unstructured":"Shih-Yang Liu, Chien-Yi Wang, Hongxu Yin, Pavlo Molchanov, Yu-Chiang\u00a0Frank Wang, Kwang-Ting Cheng, and Min-Hung Chen. 2024. Dora: Weight-decomposed low-rank adaptation. In Forty-first International Conference on Machine Learning."},{"key":"e_1_3_3_2_12_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00320"},{"key":"e_1_3_3_2_13_2","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547910"},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"crossref","unstructured":"Farhood Negin Baris Ozyer Saeid Agahian Sibel Kacdioglu and Gulsah\u00a0Tumuklu Ozyer. 2021. Vision-assisted recognition of stereotype behaviors for early diagnosis of autism spectrum disorders. Neurocomputing 446 (2021) 145\u2013155.","DOI":"10.1016\/j.neucom.2021.03.004"},{"key":"e_1_3_3_2_15_2","doi-asserted-by":"crossref","unstructured":"Junting Pan Ziyi Lin Xiatian Zhu Jing Shao and Hongsheng Li. 2022. St-adapter: Parameter-efficient image-to-video transfer learning. Advances in Neural Information Processing Systems 35 (2022) 26462\u201326477.","DOI":"10.52202\/068431-1919"},{"key":"e_1_3_3_2_16_2","first-page":"8748","volume-title":"International conference on machine learning","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et\u00a0al. 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PMLR, 8748\u20138763."},{"key":"e_1_3_3_2_17_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2013.103"},{"key":"e_1_3_3_2_18_2","unstructured":"Xingjian Shi Zhourong Chen Hao Wang Dit-Yan Yeung Wai-Kin Wong and Wang-chun Woo. 2015. Convolutional LSTM network: A machine learning approach for precipitation nowcasting. Advances in neural information processing systems 28 (2015)."},{"key":"e_1_3_3_2_19_2","unstructured":"Mengmeng Wang Jiazheng Xing Boyuan Jiang Jun Chen Jianbiao Mei Xingxing Zuo Guang Dai Jingdong Wang and Yong Liu. 2024. M2-clip: A multimodal multi-task adapting framework for video action recognition. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2401.11649 (2024)."},{"key":"e_1_3_3_2_20_2","unstructured":"Mengmeng Wang Jiazheng Xing and Yong Liu. 2021. Actionclip: A new paradigm for video action recognition. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2109.08472 (2021)."},{"key":"e_1_3_3_2_21_2","doi-asserted-by":"crossref","unstructured":"Pengbo Wei David Ahmedt-Aristizabal Harshala Gammulle Simon Denman and Mohammad\u00a0Ali Armin. 2023. Vision-based activity recognition in children with autism-related behaviors. Heliyon 9 6 (2023).","DOI":"10.1016\/j.heliyon.2023.e16763"},{"key":"e_1_3_3_2_22_2","unstructured":"Muhammad Yaseen. 2024. What is YOLOv8: An In-Depth Exploration of the Internal Features of the Next-Generation Object Detector. arxiv:https:\/\/arXiv.org\/abs\/2408.15857\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2408.15857"},{"key":"e_1_3_3_2_23_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01100"},{"key":"e_1_3_3_2_24_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2017.369"},{"key":"e_1_3_3_2_25_2","doi-asserted-by":"crossref","unstructured":"Lonnie Zwaigenbaum Margaret\u00a0L Bauman Roula Choueiri Connie Kasari Alice Carter Doreen Granpeesheh Zoe Mailloux Susanne Smith\u00a0Roley Sheldon Wagner Deborah Fein et\u00a0al. 2015. Early intervention for children with autism spectrum disorder under 3 years of age: recommendations for practice and research. Pediatrics 136 Supplement_1 (2015) S60\u2013S81.","DOI":"10.1542\/peds.2014-3667E"}],"event":{"name":"ICVGIP 2025: Indian Conference on Computer Vision, Graphics, and Image Processing","location":"Mandi Himachal Pradesh India","acronym":"ICVGIP 2025"},"container-title":["Proceedings of the Sixteen Indian Conference on Computer Vision, Graphics and Image Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3774521.3774572","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T08:06:41Z","timestamp":1785485201000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3774521.3774572"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,17]]},"references-count":24,"alternative-id":["10.1145\/3774521.3774572","10.1145\/3774521"],"URL":"https:\/\/doi.org\/10.1145\/3774521.3774572","relation":{},"subject":[],"published":{"date-parts":[[2025,12,17]]},"assertion":[{"value":"2026-07-31","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}