{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T17:46:38Z","timestamp":1777657598597,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":78,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"National Science Foundation of China under Grant","award":["62106219"],"award-info":[{"award-number":["62106219"]}]},{"name":"Zhejiang Provincial Natural Science Foundation of China under Grant","award":["LD24F020016, LZ24F030005"],"award-info":[{"award-number":["LD24F020016, LZ24F030005"]}]},{"name":"National Science Foundation for Distinguished Young Scholars under Grant","award":["62225605"],"award-info":[{"award-number":["62225605"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3680679","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:27Z","timestamp":1729925967000},"page":"2945-2954","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":6,"title":["Ego3DT: Tracking Every 3D Object in Ego-centric Videos"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8652-8556","authenticated-orcid":false,"given":"Shengyu","family":"Hao","sequence":"first","affiliation":[{"name":"College of Computer Science and Technology, Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2611-0008","authenticated-orcid":false,"given":"Wenhao","family":"Chai","sequence":"additional","affiliation":[{"name":"University of Washington, Seattle, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-6537-376X","authenticated-orcid":false,"given":"Zhonghan","family":"Zhao","sequence":"additional","affiliation":[{"name":"College of Computer Science and Technology, Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-6158-7220","authenticated-orcid":false,"given":"Meiqi","family":"Sun","sequence":"additional","affiliation":[{"name":"Zhejiang University-University of Illinois Urbana Champaign Institute, Zhejiang University, Haining, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-0889-1977","authenticated-orcid":false,"given":"Wendi","family":"Hu","sequence":"additional","affiliation":[{"name":"College of Computer Science and Technology, Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-3387-1805","authenticated-orcid":false,"given":"Jieyang","family":"Zhou","sequence":"additional","affiliation":[{"name":"Zhejiang University-University of Illinois Urbana Champaign Institute, Zhejiang University, Haining, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-8295-9630","authenticated-orcid":false,"given":"Yixian","family":"Zhao","sequence":"additional","affiliation":[{"name":"Zhejiang University-University of Illinois Urbana Champaign Institute, Zhejiang University, Haining, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-3555-3324","authenticated-orcid":false,"given":"Qi","family":"Li","sequence":"additional","affiliation":[{"name":"Zhejiang University-University of Illinois Urbana Champaign Institute, Zhejiang University, Haining, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9692-6235","authenticated-orcid":false,"given":"Yizhou","family":"Wang","sequence":"additional","affiliation":[{"name":"University of Washington, Seattle, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-4155-3166","authenticated-orcid":false,"given":"Xi","family":"Li","sequence":"additional","affiliation":[{"name":"College of Computer Science and Technology, Zhejiang University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8403-1538","authenticated-orcid":false,"given":"Gaoang","family":"Wang","sequence":"additional","affiliation":[{"name":"Zhejiang University-University of Illinois Urbana Champaign Institute, Zhejiang University &amp; College of Computer Science and Technology, Zhejiang University, Haining, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01164"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.02106"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW56347.2022.00161"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"crossref","unstructured":"Tianheng Cheng Lin Song Yixiao Ge Wenyu Liu Xinggang Wang and Ying Shan. 2024. YOLO-World: Real-Time Open-Vocabulary Object Detection. arxiv: 2401.17270 [cs.CV]","DOI":"10.1109\/CVPR52733.2024.01599"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00667"},{"key":"e_1_3_2_2_6_1","volume-title":"Scaling Egocentric Vision: The EPIC-KITCHENS Dataset. In European Conference on Computer Vision (ECCV).","author":"Damen Dima","year":"2018","unstructured":"Dima Damen, Hazel Doughty, Giovanni Maria Farinella, Sanja Fidler, Antonino Furnari, Evangelos Kazakos, Davide Moltisanti, Jonathan Munro, Toby Perrett, Will Price, and Michael Wray. 2018. Scaling Egocentric Vision: The EPIC-KITCHENS Dataset. In European Conference on Computer Vision (ECCV)."},{"key":"e_1_3_2_2_7_1","volume-title":"EPIC-KITCHENS VISOR Benchmark: VIdeo Segmentations and Object Relations. In Thirty-sixth Conference on Neural Information Processing Systems Datasets and Benchmarks Track.","author":"Darkhalil Ahmad","year":"2022","unstructured":"Ahmad Darkhalil, Dandan Shan, Bin Zhu, Jian Ma, Amlan Kar, Richard Ely Locke Higgins, Sanja Fidler, David Fouhey, and Dima Damen. 2022. EPIC-KITCHENS VISOR Benchmark: VIdeo Segmentations and Object Relations. In Thirty-sixth Conference on Neural Information Processing Systems Datasets and Benchmarks Track."},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-020-01393-0"},{"key":"e_1_3_2_2_9_1","volume-title":"CityGen: Infinite and Controllable 3D City Layout Generation. arXiv preprint arXiv:2312.01508","author":"Deng Jie","year":"2023","unstructured":"Jie Deng, Wenhao Chai, Jianshu Guo, Qixuan Huang, Wenhao Hu, Jenq-Neng Hwang, and Gaoang Wang. 2023. CityGen: Infinite and Controllable 3D City Layout Generation. arXiv preprint arXiv:2312.01508 (2023)."},{"key":"e_1_3_2_2_10_1","unstructured":"Jie Deng Wenhao Chai Junsheng Huang Zhonghan Zhao Qixuan Huang Mingyan Gao Jianshu Guo Shengyu Hao Wenhao Hu Jenq-Neng Hwang et al. 2024. CityCraft: A Real Crafter for 3D City Generation. arXiv preprint arXiv:2406.04983 (2024)."},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"crossref","unstructured":"Yu Du Fangyun Wei Zihe Zhang Miaojing Shi Yue Gao and Guoqi Li. 2022. Learning to Prompt for Open-Vocabulary Object Detection with Vision-Language Model. In CVPR. 14064--14073.","DOI":"10.1109\/CVPR52688.2022.01369"},{"key":"e_1_3_2_2_12_1","volume-title":"Hongyuan Zhu, and Cheston Tan.","author":"Duan Jiafei","year":"2022","unstructured":"Jiafei Duan, Samson Yu, Hui Li Tan, Hongyuan Zhu, and Cheston Tan. 2022. A survey of embodied ai: From simulators to research tasks. IEEE Transactions on Emerging Topics in Computational Intelligence (2022)."},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW54120.2021.00304"},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-022-01694-6"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00552"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01842"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"crossref","unstructured":"Kristen Grauman Andrew Westbury Lorenzo Torresani Kris Kitani Jitendra Malik Triantafyllos Afouras Kumar Ashutosh Vijay Baiyya Siddhant Bansal Bikram Boote et al. 2023. Ego-exo4d: Understanding skilled human activity from first-and third-person perspectives. arXiv preprint arXiv:2311.18259 (2023).","DOI":"10.1109\/CVPR52733.2024.01834"},{"key":"e_1_3_2_2_18_1","unstructured":"Xiuye Gu Tsung-Yi Lin Weicheng Kuo and Yin Cui. 2022. Open-vocabulary Object Detection via Vision and Language Knowledge Distillation. In ICLR."},{"key":"e_1_3_2_2_19_1","volume-title":"SPARF: Large-Scale Learning of 3D Sparse Radiance Fields from Few Input Images. arXiv preprint arXiv:2212.09100","author":"Hamdi Abdullah","year":"2022","unstructured":"Abdullah Hamdi, Bernard Ghanem, and Matthias Nie\u00dfner. 2022. SPARF: Large-Scale Learning of 3D Sparse Radiance Fields from Few Input Images. arXiv preprint arXiv:2212.09100 (2022)."},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-023-01922-7"},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.cviu.2021.103261"},{"key":"e_1_3_2_2_22_1","volume-title":"Exploring Learning-based Motion Models in Multi-Object Tracking. arXiv preprint arXiv:2403.10826","author":"Huang Hsiang-Wei","year":"2024","unstructured":"Hsiang-Wei Huang, Cheng-Yen Yang, Wenhao Chai, Zhongyu Jiang, and Jenq-Neng Hwang. 2024. Exploring Learning-based Motion Models in Multi-Object Tracking. arXiv preprint arXiv:2403.10826 (2024)."},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2019.2957464"},{"key":"e_1_3_2_2_24_1","unstructured":"Chao Jia Yinfei Yang Ye Xia Yi-Ting Chen Zarana Parekh Hieu Pham Quoc V. Le Yun-Hsuan Sung Zhen Li and Tom Duerig. 2021. Scaling Up Visual and Vision-Language Representation Learning With Noisy Text Supervision. In ICML. 4904--4916."},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.336"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1002\/rob.21831"},{"key":"e_1_3_2_2_28_1","volume-title":"SCTN: Sparse Convolution-Transformer Network for Scene Flow Estimation. In AAAI.","author":"Li Bing","year":"2022","unstructured":"Bing Li, Cheng Zheng, Silvio Giancola, and Bernard Ghanem. 2022. SCTN: Sparse Convolution-Transformer Network for Scene Flow Estimation. In AAAI."},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01644"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"crossref","unstructured":"Liunian Harold Li Pengchuan Zhang Haotian Zhang Jianwei Yang Chunyuan Li Yiwu Zhong Lijuan Wang Lu Yuan Lei Zhang Jenq-Neng Hwang Kai-Wei Chang and Jianfeng Gao. 2022. Grounded Language-Image Pre-training. In CVPR. 10955--10965.","DOI":"10.1109\/CVPR52688.2022.01069"},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20047-2_29"},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00539"},{"key":"e_1_3_2_2_33_1","unstructured":"Yanghao Li Tushar Nagarajan Bo Xiong and Kristen Grauman. 2021. Ego-Exo: Transferring Visual Representations from Third-person to First-person Videos. In CVPR."},{"key":"e_1_3_2_2_34_1","volume-title":"MonoTAKD: Teaching Assistant Knowledge Distillation for Monocular 3D Object Detection. arXiv preprint arXiv:2404.04910","author":"Liu I","year":"2024","unstructured":"Hou-I Liu, Christine Wu, Jen-Hao Cheng, Wenhao Chai, Shian-Yun Wang, Gaowen Liu, Jenq-Neng Hwang, Hong-Han Shuai, and Wen-Huang Cheng. 2024. MonoTAKD: Teaching Assistant Knowledge Distillation for Monocular 3D Object Detection. arXiv preprint arXiv:2404.04910 (2024)."},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01551"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2303.05499"},{"key":"e_1_3_2_2_37_1","volume-title":"Hota: A higher order metric for evaluating multi-object tracking. International journal of computer vision","author":"Luiten Jonathon","year":"2021","unstructured":"Jonathon Luiten, Aljosa Osep, Patrick Dendorfer, Philip Torr, Andreas Geiger, Laura Leal-Taix\u00e9, and Bastian Leibe. 2021. Hota: A higher order metric for evaluating multi-object tracking. International journal of computer vision, Vol. 129 (2021), 548--578."},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01246-5_19"},{"key":"e_1_3_2_2_39_1","volume-title":"Jose Maria Martinez Montiel, and Juan D Tardos","author":"Mur-Artal Raul","year":"2015","unstructured":"Raul Mur-Artal, Jose Maria Martinez Montiel, and Juan D Tardos. 2015. ORB-SLAM: a versatile and accurate monocular SLAM system. IEEE transactions on robotics, Vol. 31, 5 (2015), 1147--1163."},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1017\/S096249291700006X"},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3580305.3599312"},{"key":"e_1_3_2_2_42_1","volume-title":"Pix4point: Image pretrained transformers for 3d point cloud understanding. arXiv preprint arXiv:2208.12259","author":"Qian Guocheng","year":"2022","unstructured":"Guocheng Qian, Xingdi Zhang, Abdullah Hamdi, and Bernard Ghanem. 2022. Pix4point: Image pretrained transformers for 3d point cloud understanding. arXiv preprint arXiv:2208.12259 (2022)."},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/2980179.2980235"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV51458.2022.00133"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00943"},{"key":"e_1_3_2_2_46_1","volume-title":"Structure-from-Motion Revisited. In Conference on Computer Vision and Pattern Recognition (CVPR).","author":"Sch\u00f6nberger Johannes Lutz","year":"2016","unstructured":"Johannes Lutz Sch\u00f6nberger and Jan-Michael Frahm. 2016. Structure-from-Motion Revisited. In Conference on Computer Vision and Pattern Recognition (CVPR)."},{"key":"e_1_3_2_2_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00989"},{"key":"e_1_3_2_2_48_1","volume-title":"Moviechat: From dense token to sparse memory for long video understanding. arXiv preprint arXiv:2307.16449","author":"Song Enxin","year":"2023","unstructured":"Enxin Song, Wenhao Chai, Guanhong Wang, Yucheng Zhang, Haoyang Zhou, Feiyang Wu, Xun Guo, Tian Ye, Yan Lu, Jenq-Neng Hwang, et al. 2023. Moviechat: From dense token to sparse memory for long video understanding. arXiv preprint arXiv:2307.16449 (2023)."},{"key":"e_1_3_2_2_49_1","volume-title":"MovieChat: Question-aware Sparse Memory for Long Video Question Answering. arXiv preprint arXiv:2404.17176","author":"Song Enxin","year":"2024","unstructured":"Enxin Song, Wenhao Chai, Tian Ye, Jenq-Neng Hwang, Xi Li, and Gaoang Wang. 2024. MovieChat: Question-aware Sparse Memory for Long Video Question Answering. arXiv preprint arXiv:2404.17176 (2024)."},{"key":"e_1_3_2_2_50_1","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Tang Hao","year":"2024","unstructured":"Hao Tang, Kevin J Liang, Kristen Grauman, Matt Feiszli, and Weiyao Wang. 2024. Egotracks: A long-term egocentric visual object tracking dataset. Advances in Neural Information Processing Systems, Vol. 36 (2024)."},{"key":"e_1_3_2_2_51_1","volume-title":"Deepv2d: Video to depth with differentiable structure from motion. arXiv preprint arXiv:1812.04605","author":"Teed Zachary","year":"2018","unstructured":"Zachary Teed and Jia Deng. 2018. Deepv2d: Video to depth with differentiable structure from motion. arXiv preprint arXiv:1812.04605 (2018)."},{"key":"e_1_3_2_2_52_1","volume-title":"Droid-slam: Deep visual slam for monocular, stereo, and rgb-d cameras. Advances in neural information processing systems","author":"Teed Zachary","year":"2021","unstructured":"Zachary Teed and Jia Deng. 2021. Droid-slam: Deep visual slam for monocular, stereo, and rgb-d cameras. Advances in neural information processing systems, Vol. 34 (2021), 16558--16569."},{"key":"e_1_3_2_2_53_1","volume-title":"Sfm-net: Learning of structure and motion from video. arXiv preprint arXiv:1704.07804","author":"Vijayanarasimhan Sudheendra","year":"2017","unstructured":"Sudheendra Vijayanarasimhan, Susanna Ricco, Cordelia Schmid, Rahul Sukthankar, and Katerina Fragkiadaki. 2017. Sfm-net: Learning of structure and motion from video. arXiv preprint arXiv:1704.07804 (2017)."},{"key":"e_1_3_2_2_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00973"},{"key":"e_1_3_2_2_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2022.3140919"},{"key":"e_1_3_2_2_56_1","doi-asserted-by":"publisher","DOI":"10.1145\/3343031.3350853"},{"key":"e_1_3_2_2_57_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01130"},{"key":"e_1_3_2_2_58_1","volume-title":"DUSt3R: Geometric 3D Vision Made Easy. arXiv preprint arXiv:2312.14132","author":"Wang Shuzhe","year":"2023","unstructured":"Shuzhe Wang, Vincent Leroy, Yohann Cabon, Boris Chidlovskii, and Jerome Revaud. 2023. DUSt3R: Geometric 3D Vision Made Easy. arXiv preprint arXiv:2312.14132 (2023)."},{"key":"e_1_3_2_2_59_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01868"},{"key":"e_1_3_2_2_60_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICIP.2017.8296962"},{"key":"e_1_3_2_2_61_1","volume-title":"General object foundation model for images and videos at scale. arXiv preprint arXiv:2312.09158","author":"Wu Junfeng","year":"2023","unstructured":"Junfeng Wu, Yi Jiang, Qihao Liu, Zehuan Yuan, Xiang Bai, and Song Bai. 2023. General object foundation model for images and videos at scale. arXiv preprint arXiv:2312.09158 (2023)."},{"key":"e_1_3_2_2_62_1","doi-asserted-by":"crossref","unstructured":"Size Wu Wenwei Zhang Sheng Jin Wentao Liu and Chen Change Loy. 2023. Aligning Bag of Regions for Open-Vocabulary Object Detection. In CVPR. 15254--15264.","DOI":"10.1109\/CVPR52729.2023.01464"},{"key":"e_1_3_2_2_63_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2013.312"},{"key":"e_1_3_2_2_64_1","unstructured":"Lewei Yao Jianhua Han Youpeng Wen Xiaodan Liang Dan Xu Wei Zhang Zhenguo Li Chunjing Xu and Hang Xu. 2022. DetCLIP: Dictionary-Enriched Visual-Concept Paralleled Pre-training for Open-world Detection. In NeurIPS."},{"key":"e_1_3_2_2_65_1","volume-title":"Derek Hao Hu, and Shih-Fu Chang.","author":"Zareian Alireza","year":"2021","unstructured":"Alireza Zareian, Kevin Dela Rosa, Derek Hao Hu, and Shih-Fu Chang. 2021. Open-Vocabulary Object Detection Using Captions. In CVPR. 14393--14402."},{"key":"e_1_3_2_2_66_1","volume-title":"DINO: DETR with Improved DeNoising Anchor Boxes for End-to-End Object Detection. In ICLR.","author":"Zhang Hao","year":"2023","unstructured":"Hao Zhang, Feng Li, Shilong Liu, Lei Zhang, Hang Su, Jun Zhu, Lionel M. Ni, and Heung-Yeung Shum. 2023. DINO: DETR with Improved DeNoising Anchor Boxes for End-to-End Object Detection. In ICLR."},{"key":"e_1_3_2_2_67_1","doi-asserted-by":"publisher","DOI":"10.1145\/3343031.3350933"},{"key":"e_1_3_2_2_68_1","unstructured":"Haotian Zhang Pengchuan Zhang Xiaowei Hu Yen-Chun Chen Liunian Harold Li Xiyang Dai Lijuan Wang Lu Yuan Jenq-Neng Hwang and Jianfeng Gao. 2022 d. GLIPv2: Unifying Localization and Vision-Language Understanding. In NeurIPS."},{"key":"e_1_3_2_2_69_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20068-7_11"},{"key":"e_1_3_2_2_70_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20047-2_1"},{"key":"e_1_3_2_2_71_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19827-4_2"},{"key":"e_1_3_2_2_72_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19824-3_31"},{"key":"e_1_3_2_2_73_1","volume-title":"See and think: Embodied agent in virtual environment. arXiv preprint arXiv:2311.15209","author":"Zhao Zhonghan","year":"2023","unstructured":"Zhonghan Zhao, Wenhao Chai, Xuan Wang, Li Boyi, Shengyu Hao, Shidong Cao, Tian Ye, Jenq-Neng Hwang, and Gaoang Wang. 2023. See and think: Embodied agent in virtual environment. arXiv preprint arXiv:2311.15209 (2023)."},{"key":"e_1_3_2_2_74_1","volume-title":"STEVE Series: Step-by-Step Construction of Agent Systems in Minecraft. arXiv preprint arXiv:2406.11247","author":"Zhao Zhonghan","year":"2024","unstructured":"Zhonghan Zhao, Wenhao Chai, Xuan Wang, Ke Ma, Kewei Chen, Dongxu Guo, Tian Ye, Yanting Zhang, Hongwei Wang, and Gaoang Wang. 2024. STEVE Series: Step-by-Step Construction of Agent Systems in Minecraft. arXiv preprint arXiv:2406.11247 (2024)."},{"key":"e_1_3_2_2_75_1","volume-title":"Hierarchical auto-organizing system for open-ended multi-agent navigation. arXiv preprint arXiv:2403.08282","author":"Zhao Zhonghan","year":"2024","unstructured":"Zhonghan Zhao, Kewei Chen, Dongxu Guo, Wenhao Chai, Tian Ye, Yanting Zhang, and Gaoang Wang. 2024. Hierarchical auto-organizing system for open-ended multi-agent navigation. arXiv preprint arXiv:2403.08282 (2024)."},{"key":"e_1_3_2_2_76_1","volume-title":"Do We Really Need a Complex Agent System? Distill Embodied Agent into a Single Model. arXiv preprint arXiv:2404.04619","author":"Zhao Zhonghan","year":"2024","unstructured":"Zhonghan Zhao, Ke Ma, Wenhao Chai, Xuan Wang, Kewei Chen, Dongxu Guo, Yanting Zhang, Hongwei Wang, and Gaoang Wang. 2024. Do We Really Need a Complex Agent System? Distill Embodied Agent into a Single Model. arXiv preprint arXiv:2404.04619 (2024)."},{"key":"e_1_3_2_2_77_1","volume-title":"Luowei Zhou, Xiyang Dai, Lu Yuan, Yin Li, and Jianfeng Gao.","author":"Zhong Yiwu","year":"2022","unstructured":"Yiwu Zhong, Jianwei Yang, Pengchuan Zhang, Chunyuan Li, Noel Codella, Liunian Harold Li, Luowei Zhou, Xiyang Dai, Lu Yuan, Yin Li, and Jianfeng Gao. 2022. RegionCLIP: Region-based Language-Image Pretraining. In CVPR. 16772--16782."},{"key":"e_1_3_2_2_78_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.700"}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","location":"Melbourne VIC Australia","acronym":"MM '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3680679","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3680679","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:17:57Z","timestamp":1750295877000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3680679"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":78,"alternative-id":["10.1145\/3664647.3680679","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3680679","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}