{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,1]],"date-time":"2025-12-01T11:29:35Z","timestamp":1764588575316,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":61,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"HKUST Special Support for Young Faculty","award":["F0927"],"award-info":[{"award-number":["F0927"]}]},{"name":"HKUST Sports Science and Technology Research Grant","award":["SSTRG24EG04"],"award-info":[{"award-number":["SSTRG24EG04"]}]},{"name":"National Natural Science Foundation of China","award":["62337001"],"award-info":[{"award-number":["62337001"]}]},{"name":"National Key Research & Development Project of China","award":["2021ZD0110700"],"award-info":[{"award-number":["2021ZD0110700"]}]},{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["226-2024-00058"],"award-info":[{"award-number":["226-2024-00058"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3681036","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:33Z","timestamp":1729925973000},"page":"1602-1611","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Seeing Beyond Classes: Zero-Shot Grounded Situation Recognition via Language Explainer"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-6391-071X","authenticated-orcid":false,"given":"Jiaming","family":"Lei","sequence":"first","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5678-4487","authenticated-orcid":false,"given":"Lin","family":"Li","sequence":"additional","affiliation":[{"name":"Hong Kong University of Science and Technology &amp; AI Chip Center for Emerging Smart Systems, Hong Kong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1854-8667","authenticated-orcid":false,"given":"Chunping","family":"Wang","sequence":"additional","affiliation":[{"name":"Finvolution Group, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6142-9914","authenticated-orcid":false,"given":"Jun","family":"Xiao","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6148-9709","authenticated-orcid":false,"given":"Long","family":"Chen","sequence":"additional","affiliation":[{"name":"Hong Kong University of Science and Technology, Hong Kong, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_2_1_1","unstructured":"2022. Exploring structure-aware transformer over interaction proposals for human-object interaction detection. In CVPR."},{"key":"e_1_3_2_2_2_1","unstructured":"Rohan Anil Andrew M Dai Orhan Firat Melvin Johnson Dmitry Lepikhin Alexandre Passos Siamak Shakeri Emanuel Taropa Paige Bailey Zhifeng Chen et al. 2023. Palm 2 technical report. arXiv preprint arXiv:2305.10403 (2023)."},{"key":"e_1_3_2_2_3_1","unstructured":"Rishi Bommasani Drew A Hudson Ehsan Adeli Russ Altman Simran Arora Sydney von Arx Michael S Bernstein Jeannette Bohg Antoine Bosselut Emma Brunskill et al. 2021. On the opportunities and risks of foundation models. arXiv preprint arXiv:2108.07258 (2021)."},{"key":"e_1_3_2_2_4_1","first-page":"1877","article-title":"Language models are few-shot learners","volume":"33","author":"Brown Tom","year":"2020","unstructured":"Tom Brown, Benjamin Mann, Nick Ryder, Melanie Subbiah, Jared D Kaplan, Prafulla Dhariwal, Arvind Neelakantan, Pranav Shyam, Girish Sastry, Amanda Askell, et al. 2020. Language models are few-shot learners. NeurIPS 33 (2020), 1877--1901.","journal-title":"NeurIPS"},{"key":"e_1_3_2_2_5_1","volume-title":"Hico: A benchmark for recognizing human-object interactions in images. In ICCV. 1017--1025.","author":"Chao Yu-Wei","year":"2015","unstructured":"Yu-Wei Chao, Zhan Wang, Yugeng He, Jiaxuan Wang, and Jia Deng. 2015. Hico: A benchmark for recognizing human-object interactions in images. In ICCV. 1017--1025."},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"crossref","unstructured":"Guikun Chen Lin Li Yawei Luo and Jun Xiao. 2023. Addressing Predicate Overlap in Scene Graph Generation with Semantic Granularity Controller. In ICME. 78--83.","DOI":"10.1109\/ICME55011.2023.00022"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"crossref","unstructured":"Guikun Chen Xia Li Yi Yang and Wenguan Wang. 2024. Neural clustering based visual representation learning. In CVPR.","DOI":"10.1109\/CVPR52733.2024.00546"},{"key":"e_1_3_2_2_8_1","volume-title":"A Survey on 3D Gaussian Splatting. CoRR abs\/2401.03890","author":"Chen Guikun","year":"2024","unstructured":"Guikun Chen and Wenguan Wang. 2024. A Survey on 3D Gaussian Splatting. CoRR abs\/2401.03890 (2024)."},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"crossref","unstructured":"Long Chen Hanwang Zhang Jun Xiao Wei Liu and Shih-Fu Chang. 2018. Zero-Shot Visual Recognition Using Semantics-Preserving Adversarial Embedding Networks. In CVPR.","DOI":"10.1109\/CVPR.2018.00115"},{"key":"e_1_3_2_2_10_1","volume-title":"Gsrformer: Grounded situation recognition transformer with alternate semantic attention refinement. In ACM MM. 3272--3281.","author":"Cheng Zhi-Qi","year":"2022","unstructured":"Zhi-Qi Cheng, Qi Dai, Siyao Li, Teruko Mitamura, and Alexander Hauptmann. 2022. Gsrformer: Grounded situation recognition transformer with alternate semantic attention refinement. In ACM MM. 3272--3281."},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475297"},{"key":"e_1_3_2_2_12_1","unstructured":"Junhyeong Cho Youngseok Yoon and Suha Kwak. 2022. Collaborative transformers for grounded situation recognition. In CVPR. 19659--19668."},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"crossref","unstructured":"KR1442 Chowdhary and KR Chowdhary. 2020. Natural language processing. (2020) 603--649.","DOI":"10.1007\/978-81-322-3972-7_19"},{"key":"e_1_3_2_2_14_1","volume-title":"Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805","author":"Devlin Jacob","year":"2018","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2018. Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)."},{"key":"e_1_3_2_2_15_1","volume-title":"A survey on in-context learning. arXiv preprint arXiv:2301.00234","author":"Dong Qingxiu","year":"2022","unstructured":"Qingxiu Dong, Lei Li, Damai Dai, Ce Zheng, Zhiyong Wu, Baobao Chang, Xu Sun, Jingjing Xu, and Zhifang Sui. 2022. A survey on in-context learning. arXiv preprint arXiv:2301.00234 (2022)."},{"key":"e_1_3_2_2_16_1","volume-title":"GPT-3: Its nature, scope, limits, and consequences. Minds and Machines","author":"Floridi Luciano","year":"2020","unstructured":"Luciano Floridi and Massimo Chiriatti. 2020. GPT-3: Its nature, scope, limits, and consequences. Minds and Machines (2020), 681--694."},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"crossref","unstructured":"Zixian Guo Bowen Dong Zhilong Ji Jinfeng Bai Yiwen Guo and Wangmeng Zuo. 2023. Texts as images in prompt tuning for multi-label image recognition. In CVPR. 2808--2817.","DOI":"10.1109\/CVPR52729.2023.00275"},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"crossref","unstructured":"Dong-Jin Kim Xiao Sun Jinsoo Choi Stephen Lin and In So Kweon. 2020. Detecting human-object interactions with action co-occurrence priors. In ECCV. 718--736.","DOI":"10.1007\/978-3-030-58589-1_43"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"crossref","unstructured":"Fanjie Kong Yanbei Chen Jiarui Cai and Davide Modolo. 2024. Hyperbolic learning with synthetic captions for open-world detection. In CVPR.","DOI":"10.1109\/CVPR52733.2024.01586"},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"crossref","unstructured":"Klemens Lagler Michael Schindelegger Johannes B\u00f6hm Hana Kr\u00e1sn\u00e1 and Tobias Nilsson. 2013. GPT2: Empirical slant delay model for radio space geodetic techniques. (2013) 1069--1073.","DOI":"10.1002\/grl.50288"},{"key":"e_1_3_2_2_21_1","volume-title":"International conference on machine learning. PMLR","author":"Li Junnan","year":"2023","unstructured":"Junnan Li, Dongxu Li, Silvio Savarese, and Steven Hoi. 2023. Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models. In International conference on machine learning. PMLR, 19730--19742."},{"key":"e_1_3_2_2_22_1","volume-title":"Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation. In ICML. PMLR, 12888--12900.","author":"Li Junnan","year":"2022","unstructured":"Junnan Li, Dongxu Li, Caiming Xiong, and Steven Hoi. 2022. Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation. In ICML. PMLR, 12888--12900."},{"key":"e_1_3_2_2_23_1","volume-title":"Align before fuse: Vision and language representation learning with momentum distillation. Advances in neural information processing systems","author":"Li Junnan","year":"2021","unstructured":"Junnan Li, Ramprasaath Selvaraju, Akhilesh Gotmare, Shafiq Joty, Caiming Xiong, and Steven Chu Hong Hoi. 2021. Align before fuse: Vision and language representation learning with momentum distillation. Advances in neural information processing systems (2021), 9694--9705."},{"key":"e_1_3_2_2_24_1","volume-title":"Catr: Combinatorial-dependence audio-queried transformer for audio-visual video segmentation. In ACM MM. 1485--1494.","author":"Li Kexin","year":"2023","unstructured":"Kexin Li, Zongxin Yang, Lei Chen, Yi Yang, and Jun Xiao. 2023. Catr: Combinatorial-dependence audio-queried transformer for audio-visual video segmentation. In ACM MM. 1485--1494."},{"key":"e_1_3_2_2_25_1","volume-title":"Compositional zeroshot learning via progressive language-based observations. arXiv preprint arXiv:2311.14749","author":"Li Lin","year":"2024","unstructured":"Lin Li, Guikun Chen, Jun Xiao, and Long Chen. 2024. Compositional zeroshot learning via progressive language-based observations. arXiv preprint arXiv:2311.14749 (2024)."},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"crossref","unstructured":"Lin Li Guikun Chen Jun Xiao Yi Yang Chunping Wang and Long Chen. 2023. Compositional Feature Augmentation for Unbiased Scene Graph Generation. In ICCV. 21628--21638.","DOI":"10.1109\/ICCV51070.2023.01982"},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"crossref","unstructured":"Lin Li Long Chen Yifeng Huang Zhimeng Zhang Songyang Zhang and Jun Xiao. 2022. The devil is in the labels: Noisy label correction for robust scene graph generation. In CVPR. 18869--18878.","DOI":"10.1109\/CVPR52688.2022.01830"},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i4.28098"},{"key":"e_1_3_2_2_29_1","volume-title":"Zero-shot visual relation detection via composite visual cues from large language models. NeurIPS 36","author":"Li Lin","year":"2024","unstructured":"Lin Li, Jun Xiao, Guikun Chen, Jian Shao, Yueting Zhuang, and Long Chen. 2024. Zero-shot visual relation detection via composite visual cues from large language models. NeurIPS 36 (2024)."},{"key":"e_1_3_2_2_30_1","volume-title":"Label Semantic Knowledge Distillation for Unbiased Scene Graph Generation. TCSVT","author":"Li Lin","year":"2024","unstructured":"Lin Li, Jun Xiao, Hanrong Shi, Wenxiao Wang, Jian Shao, An-An Liu, Yi Yang, and Long Chen. 2024. Label Semantic Knowledge Distillation for Unbiased Scene Graph Generation. TCSVT (2024), 195--206."},{"key":"e_1_3_2_2_31_1","volume-title":"NICEST: Noisy Label Correction and Training for Robust Scene Graph Generation. TPAMI","author":"Li Lin","year":"2024","unstructured":"Lin Li, Jun Xiao, Hanrong Shi, Hanwang Zhang, Yi Yang,Wei Liu, and Long Chen. 2024. NICEST: Noisy Label Correction and Training for Robust Scene Graph Generation. TPAMI (2024)."},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"crossref","unstructured":"Liunian Harold Li Pengchuan Zhang Haotian Zhang Jianwei Yang Chunyuan Li Yiwu Zhong LijuanWang Lu Yuan Lei Zhang Jenq-Neng Hwang et al. 2022. Grounded language-image pre-training. In CVPR. 10965--10975.","DOI":"10.1109\/CVPR52688.2022.01069"},{"key":"e_1_3_2_2_33_1","volume-title":"Clip-event: Connecting text and images with event structures. In CVPR. 16420--16429.","author":"Li Manling","year":"2022","unstructured":"Manling Li, Ruochen Xu, ShuohangWang, Luowei Zhou, Xudong Lin, Chenguang Zhu, Michael Zeng, Heng Ji, and Shih-Fu Chang. 2022. Clip-event: Connecting text and images with event structures. In CVPR. 16420--16429."},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"crossref","unstructured":"Wei Lin Leonid Karlinsky Nina Shvetsova Horst Possegger Mateusz Kozinski Rameswar Panda Rogerio Feris Hilde Kuehne and Horst Bischof. 2023. Match expand and improve: Unsupervised finetuning for zero-shot action recognition with language knowledge. In ICCV. 2851--2862.","DOI":"10.1109\/ICCV51070.2023.00267"},{"key":"e_1_3_2_2_35_1","unstructured":"Shilong Liu Zhaoyang Zeng Tianhe Ren Feng Li Hao Zhang Jie Yang Chunyuan Li Jianwei Yang Hang Su Jun Zhu et al. 2023. Grounding dino: Marrying dino with grounded pre-training for open-set object detection. arXiv preprint arXiv:2303.05499 (2023)."},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1080\/01431160600746456"},{"key":"e_1_3_2_2_37_1","unstructured":"Sachit Menon and Carl Vondrick. 2022. Visual Classification via Description from Large Language Models. In ICLR."},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"crossref","unstructured":"Sewon Min Xinxi Lyu Ari Holtzman Mikel Artetxe Mike Lewis Hannaneh Hajishirzi and Luke Zettlemoyer. 2022. Rethinking the Role of Demonstrations: What Makes In-Context Learning Work?. In EMNLP. 11048--11064.","DOI":"10.18653\/v1\/2022.emnlp-main.759"},{"key":"e_1_3_2_2_39_1","volume-title":"Chils: Zero-shot image classification with hierarchical label sets. In ICML.","author":"Novack Zachary","year":"2023","unstructured":"Zachary Novack, Saurabh Garg, Julian McAuley, and Zachary C Lipton. 2023. Chils: Zero-shot image classification with hierarchical label sets. In ICML."},{"key":"e_1_3_2_2_40_1","first-page":"4051","article-title":"A review of generalized zero-shot learning methods","volume":"45","author":"Pourpanah Farhad","year":"2022","unstructured":"Farhad Pourpanah, Moloud Abdar, Yuxuan Luo, Xinlei Zhou, Ran Wang, Chee Peng Lim, Xi-Zhao Wang, and QM Jonathan Wu. 2022. A review of generalized zero-shot learning methods. IEEE TPAMI 45, 4 (2022), 4051--4070.","journal-title":"IEEE TPAMI"},{"volume-title":"Grounded situation recognition","author":"Pratt Sarah","key":"e_1_3_2_2_41_1","unstructured":"Sarah Pratt, Mark Yatskar, Luca Weihs, Ali Farhadi, and Aniruddha Kembhavi. 2020. Grounded situation recognition. In ECCV. Springer, 314--332."},{"key":"e_1_3_2_2_42_1","volume-title":"Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al.","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al. 2021. Learning transferable visual models from natural language supervision. In ICML. 8748--8763."},{"key":"e_1_3_2_2_43_1","unstructured":"Bernardino Romera-Paredes and Philip Torr. 2015. An embarrassingly simple approach to zero-shot learning. In ICML. PMLR 2152--2161."},{"key":"e_1_3_2_2_44_1","volume-title":"From Easy to Hard: Learning Curricular Shape-aware Features for Robust Panoptic Scene Graph Generation. IJCV","author":"Shi Hanrong","year":"2024","unstructured":"Hanrong Shi, Lin Li, Jun Xiao, Yueting Zhuang, and Long Chen. 2024. From Easy to Hard: Learning Curricular Shape-aware Features for Robust Panoptic Scene Graph Generation. IJCV (2024)."},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"crossref","unstructured":"Yibing Song Ruifei Zhang Zhihong Chen Xiang Wan and Guanbin Li. 2023. Advancing visual grounding with scene knowledge: Benchmark and method. In CVPR. 15039--15049.","DOI":"10.1109\/CVPR52729.2023.01444"},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"crossref","unstructured":"Kaihua Tang Yulei Niu Jianqiang Huang Jiaxin Shi and Hanwang Zhang. 2020. Unbiased scene graph generation from biased training. In CVPR. 3716--3725.","DOI":"10.1109\/CVPR42600.2020.00377"},{"key":"e_1_3_2_2_47_1","unstructured":"Hugo Touvron Louis Martin Kevin Stone Peter Albert Amjad Almahairi Yasmine Babaei Nikolay Bashlykov Soumya Batra Prajjwal Bhargava Shruti Bhosale et al. 2023. Llama 2: Open foundation and fine-tuned chat models. arXiv preprint arXiv:2307.09288 (2023)."},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i3.20167"},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2018.2857768"},{"key":"e_1_3_2_2_50_1","doi-asserted-by":"crossref","unstructured":"Yongqin Xian Tobias Lorenz Bernt Schiele and Zeynep Akata. 2018. Feature generating networks for zero-shot learning. In CVPR. 5542--5551.","DOI":"10.1109\/CVPR.2018.00581"},{"key":"e_1_3_2_2_51_1","unstructured":"Danfei Xu Yuke Zhu Christopher B Choy and Li Fei-Fei. 2017. Scene graph generation by iterative message passing. In CVPR. 5410--5419."},{"key":"e_1_3_2_2_52_1","doi-asserted-by":"crossref","unstructured":"Jie Xu Hanbo Zhang Qingyi Si Yifeng Li Xuguang Lan and Tao Kong. 2024. Towards Unified Interactive Visual Grounding in The Wild. In ICRA.","DOI":"10.1109\/ICRA57147.2024.10611354"},{"key":"e_1_3_2_2_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00287"},{"key":"e_1_3_2_2_54_1","volume-title":"Pcpl: Predicate-correlation perception learning for unbiased scene graph generation. In ACM MM. 265--273.","author":"Yan Shaotian","year":"2020","unstructured":"Shaotian Yan, Chen Shen, Zhongming Jin, Jianqiang Huang, Rongxin Jiang, Yaowu Chen, and Xian-Sheng Hua. 2020. Pcpl: Predicate-correlation perception learning for unbiased scene graph generation. In ACM MM. 265--273."},{"key":"e_1_3_2_2_55_1","volume-title":"Annolid: Annotate, Segment, and Track Anything You Need. arXiv preprint arXiv:2403.18690","author":"Yang Chen","year":"2024","unstructured":"Chen Yang and Thomas A Cleland. 2024. Annolid: Annotate, Segment, and Track Anything You Need. arXiv preprint arXiv:2403.18690 (2024)."},{"key":"e_1_3_2_2_56_1","doi-asserted-by":"crossref","unstructured":"Yue Yang Artemis Panagopoulou Shenghao Zhou Daniel Jin Chris Callison-Burch and Mark Yatskar. 2023. Language in a bottle: Language model guided concept bottlenecks for interpretable image classification. In CVPR. 19187--19197.","DOI":"10.1109\/CVPR52729.2023.01839"},{"key":"e_1_3_2_2_57_1","doi-asserted-by":"crossref","unstructured":"Mark Yatskar Luke Zettlemoyer and Ali Farhadi. 2016. Situation recognition: Visual semantic role labeling for image understanding. 5534--5542.","DOI":"10.1109\/CVPR.2016.597"},{"key":"e_1_3_2_2_58_1","unstructured":"Qifan Yu Juncheng Li Yu Wu Siliang Tang Wei Ji and Yueting Zhuang. 2023. Visually-prompted language model for fine-grained scene graph generation in an open world. In ICCV. 21560--21571."},{"key":"e_1_3_2_2_59_1","volume-title":"Xi Victoria Lin, et al","author":"Zhang Susan","year":"2023","unstructured":"Susan Zhang, Stephen Roller, Naman Goyal, Mikel Artetxe, Moya Chen, Shuohui Chen, Christopher Dewan, Mona Diab, Xian Li, Xi Victoria Lin, et al. 2023. Opt: Open pre-trained transformer language models, 2022. URL https:\/\/arxiv. org\/abs\/2205.01068 (2023), 19-0."},{"key":"e_1_3_2_2_60_1","volume-title":"Xiangyu Chu, Qi Dou, and KW Au.","author":"Zhang Zhen","year":"2023","unstructured":"Zhen Zhang, Anran Lin, Chun Wai Wong, Xiangyu Chu, Qi Dou, and KW Au. 2023. Interactive Navigation in Environments with Traversable Obstacles Using Large Language and Vision-Language Models. arXiv preprint arXiv:2310.08873 (2023)."},{"key":"e_1_3_2_2_61_1","volume-title":"Object detection with deep learning: A review. 30, 11","author":"Zhao Zhong-Qiu","year":"2019","unstructured":"Zhong-Qiu Zhao, Peng Zheng, Shou-tao Xu, and Xindong Wu. 2019. Object detection with deep learning: A review. 30, 11 (2019), 3212--3232."}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Melbourne VIC Australia","acronym":"MM '24"},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681036","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3681036","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:17:37Z","timestamp":1750295857000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681036"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":61,"alternative-id":["10.1145\/3664647.3681036","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3681036","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}