{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,13]],"date-time":"2026-07-13T19:51:58Z","timestamp":1783972318765,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":76,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3758180","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T05:44:48Z","timestamp":1761371088000},"page":"12519-12528","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["AirScape: An Aerial Generative World Model with Motion Controllability"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9653-3316","authenticated-orcid":false,"given":"Baining","family":"Zhao","sequence":"first","affiliation":[{"name":"Shenzhen International Graduate School, Tsinghua University, Shenzhen, China and Pengcheng Laboratory, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-4625-2315","authenticated-orcid":false,"given":"Rongze","family":"Tang","sequence":"additional","affiliation":[{"name":"School of Computer Science &amp; Technology, Beijing Institute of Technology, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-4890-7794","authenticated-orcid":false,"given":"Mingyuan","family":"Jia","sequence":"additional","affiliation":[{"name":"Department of Automation, Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-2297-1830","authenticated-orcid":false,"given":"Ziyou","family":"Wang","sequence":"additional","affiliation":[{"name":"Computer and Communication Engineering College, Northeastern University, Qinhuangdao, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5830-0685","authenticated-orcid":false,"given":"Fanhang","family":"Man","sequence":"additional","affiliation":[{"name":"Shenzhen International Graduate School, Tsinghua University, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2506-7370","authenticated-orcid":false,"given":"Xin","family":"Zhang","sequence":"additional","affiliation":[{"name":"Manifold AI, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-9049-8483","authenticated-orcid":false,"given":"Yu","family":"Shang","sequence":"additional","affiliation":[{"name":"Department of Electronic Engineering, Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-8892-9389","authenticated-orcid":false,"given":"Weichen","family":"Zhang","sequence":"additional","affiliation":[{"name":"Shenzhen International Graduate School, Tsinghua University, Shenzhen, China and Pengcheng Laboratory, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5757-4476","authenticated-orcid":false,"given":"Wei","family":"Wu","sequence":"additional","affiliation":[{"name":"Manifold AI, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7561-5646","authenticated-orcid":false,"given":"Chen","family":"Gao","sequence":"additional","affiliation":[{"name":"BNRist, Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8271-5023","authenticated-orcid":false,"given":"Xinlei","family":"Chen","sequence":"additional","affiliation":[{"name":"Shenzhen International Graduate School, Tsinghua University, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5617-1659","authenticated-orcid":false,"given":"Yong","family":"Li","sequence":"additional","affiliation":[{"name":"Department of Electronic Engineering, Tsinghua University, Beijing, China and BNRist, Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_2_1_1","unstructured":"Niket Agarwal Arslan Ali Maciej Bala Yogesh Balaji Erik Barker Tiffany Cai Prithvijit Chattopadhyay Yongxin Chen Yin Cui Yifan Ding et al. 2025. Cosmos world foundation model platform for physical ai. arXiv preprint arXiv:2501.03575 (2025)."},{"key":"e_1_3_2_2_2_1","unstructured":"Google AI. 2023. Gemini-2.0-Flash. https:\/\/ai.google.dev\/. Accessed: 2025-05-30."},{"key":"e_1_3_2_2_3_1","volume-title":"Etpnav: Evolving topological planning for vision-language navigation in continuous environments","author":"An Dong","year":"2024","unstructured":"Dong An, Hanqing Wang, Wenguan Wang, Zun Wang, Yan Huang, Keji He, and Liang Wang. 2024. Etpnav: Evolving topological planning for vision-language navigation in continuous environments. IEEE Transactions on Pattern Analysis and Machine Intelligence (2024)."},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/3680528.3687614"},{"key":"e_1_3_2_2_5_1","unstructured":"Andreas Blattmann Tim Dockhorn Sumith Kulal Daniel Mendelevitch Maciej Kilian Dominik Lorenz Yam Levi Zion English Vikram Voleti Adam Letts et al. 2023. Stable video diffusion: Scaling latent video diffusion models to large datasets. arXiv preprint arXiv:2311.15127 (2023)."},{"key":"e_1_3_2_2_6_1","volume-title":"Random forests. Machine learning","author":"Breiman Leo","year":"2001","unstructured":"Leo Breiman. 2001. Random forests. Machine learning, Vol. 45, 1 (2001), 5-32."},{"key":"e_1_3_2_2_7_1","volume-title":"Forty-first International Conference on Machine Learning.","author":"Bruce Jake","year":"2024","unstructured":"Jake Bruce, Michael D Dennis, Ashley Edwards, Jack Parker-Holder, Yuge Shi, Edward Hughes, Matthew Lai, Aditi Mavalankar, Richie Steigerwald, Chris Apps, et al., 2024. Genie: Generative interactive environments. In Forty-first International Conference on Machine Learning."},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01164"},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/TKDE.2024.3361474"},{"key":"e_1_3_2_2_10_1","volume-title":"International conference on machine learning. PMLR, 1691-1703","author":"Chen Mark","year":"2020","unstructured":"Mark Chen, Alec Radford, Rewon Child, Jeffrey Wu, Heewoo Jun, David Luan, and Ilya Sutskever. 2020. Generative pretraining from pixels. In International conference on machine learning. PMLR, 1691-1703."},{"key":"e_1_3_2_2_11_1","volume-title":"Ddl: Empowering delivery drones with large-scale urban sensing capability","author":"Chen Xuecheng","year":"2024","unstructured":"Xuecheng Chen, Haoyang Wang, Yuhan Cheng, Haohao Fu, Yuxuan Liu, Fan Dang, Yunhao Liu, Jinqiang Cui, and Xinlei Chen. 2024a. Ddl: Empowering delivery drones with large-scale urban sensing capability. IEEE Journal of Selected Topics in Signal Processing (2024)."},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/JIOT.2024.3389771"},{"key":"e_1_3_2_2_13_1","unstructured":"Zesen Cheng Sicong Leng Hang Zhang Yifei Xin Xin Li Guanzheng Chen Yongxin Zhu Wenqi Zhang Ziyang Luo Deli Zhao et al. 2024. Videollama 2: Advancing spatial-temporal modeling and audio understanding in video-llms. arXiv preprint arXiv:2406.07476 (2024)."},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3386569.3392457"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2023.3261988"},{"key":"e_1_3_2_2_16_1","volume-title":"Robonet: Large-scale multi-robot learning. arXiv preprint arXiv:1910.11215","author":"Dasari Sudeep","year":"2019","unstructured":"Sudeep Dasari, Frederik Ebert, Stephen Tian, Suraj Nair, Bernadette Bucher, Karl Schmeckpeper, Siddharth Singh, Sergey Levine, and Chelsea Finn. 2019. Robonet: Large-scale multi-robot learning. arXiv preprint arXiv:1910.11215 (2019)."},{"key":"e_1_3_2_2_17_1","unstructured":"Jingtao Ding Yunke Zhang Yu Shang Yuheng Zhang Zefang Zong Jie Feng Yuan Yuan Hongyuan Su Nian Li Nicholas Sukiennik et al. 2024. Understanding world or predicting future? a comprehensive survey of world models. Comput. Surveys (2024)."},{"key":"e_1_3_2_2_18_1","unstructured":"Chen Gao Baining Zhao Weichen Zhang Jinzhu Mao Jun Zhang Zhiheng Zheng Fanhang Man Jianjie Fang Zile Zhou Jinqiang Cui et al. 2024c. EmbodiedCity: A Benchmark Platform for Embodied Agent in Real-world City Environment. arXiv preprint arXiv:2410.09604 (2024)."},{"key":"e_1_3_2_2_19_1","volume-title":"Magicdrive3d: Controllable 3d generation for any-view rendering in street scenes. arXiv preprint arXiv:2405.14475","author":"Gao Ruiyuan","year":"2024","unstructured":"Ruiyuan Gao, Kai Chen, Zhihao Li, Lanqing Hong, Zhenguo Li, and Qiang Xu. 2024a. Magicdrive3d: Controllable 3d generation for any-view rendering in street scenes. arXiv preprint arXiv:2405.14475 (2024)."},{"key":"e_1_3_2_2_20_1","volume-title":"Magicdrive: Street view generation with diverse 3d geometry control. arXiv preprint arXiv:2310.02601","author":"Gao Ruiyuan","year":"2023","unstructured":"Ruiyuan Gao, Kai Chen, Enze Xie, Lanqing Hong, Zhenguo Li, Dit-Yan Yeung, and Qiang Xu. 2023. Magicdrive: Street view generation with diverse 3d geometry control. arXiv preprint arXiv:2310.02601 (2023)."},{"key":"e_1_3_2_2_21_1","volume-title":"Vista: A generalizable driving world model with high fidelity and versatile controllability. arXiv preprint arXiv:2405.17398","author":"Gao Shenyuan","year":"2024","unstructured":"Shenyuan Gao, Jiazhi Yang, Li Chen, Kashyap Chitta, Yihang Qiu, Andreas Geiger, Jun Zhang, and Hongyang Li. 2024b. Vista: A generalizable driving world model with high fidelity and versatile controllability. arXiv preprint arXiv:2405.17398 (2024)."},{"key":"e_1_3_2_2_22_1","volume-title":"Advancing humanoid locomotion: Mastering challenging terrains with denoising world model learning. arXiv preprint arXiv:2408.14472","author":"Gu Xinyang","year":"2024","unstructured":"Xinyang Gu, Yen-Jen Wang, Xiang Zhu, Chengming Shi, Yanjiang Guo, Yichen Liu, and Jianyu Chen. 2024. Advancing humanoid locomotion: Mastering challenging terrains with denoising world model learning. arXiv preprint arXiv:2408.14472 (2024)."},{"key":"e_1_3_2_2_23_1","volume-title":"World models for autonomous driving: An initial survey","author":"Guan Yanchen","year":"2024","unstructured":"Yanchen Guan, Haicheng Liao, Zhenning Li, Jia Hu, Runze Yuan, Guohui Zhang, and Chengzhong Xu. 2024. World models for autonomous driving: An initial survey. IEEE Transactions on Intelligent Vehicles (2024)."},{"key":"e_1_3_2_2_24_1","volume-title":"Embodied intelligence via learning and evolution. Nature communications","author":"Gupta Agrim","year":"2021","unstructured":"Agrim Gupta, Silvio Savarese, Surya Ganguli, and Li Fei-Fei. 2021. Embodied intelligence via learning and evolution. Nature communications, Vol. 12, 1 (2021), 5721."},{"key":"e_1_3_2_2_25_1","volume-title":"World models. arXiv preprint arXiv:1803.10122","author":"Ha David","year":"2018","unstructured":"David Ha and J\u00fcrgen Schmidhuber. 2018. World models. arXiv preprint arXiv:1803.10122 (2018)."},{"key":"e_1_3_2_2_26_1","volume-title":"Ltx-video: Realtime video latent diffusion. arXiv preprint arXiv:2501.00103","author":"HaCohen Yoav","year":"2024","unstructured":"Yoav HaCohen, Nisan Chiprut, Benny Brazowski, Daniel Shalem, Dudu Moshe, Eitan Richardson, Eran Levin, Guy Shiran, Nir Zabari, Ori Gordon, et al., 2024. Ltx-video: Realtime video latent diffusion. arXiv preprint arXiv:2501.00103 (2024)."},{"key":"e_1_3_2_2_27_1","volume-title":"Gans trained by a two time-scale update rule converge to a local nash equilibrium. Advances in neural information processing systems","author":"Heusel Martin","year":"2017","unstructured":"Martin Heusel, Hubert Ramsauer, Thomas Unterthiner, Bernhard Nessler, and Sepp Hochreiter. 2017. Gans trained by a two time-scale update rule converge to a local nash equilibrium. Advances in neural information processing systems, Vol. 30 (2017)."},{"key":"e_1_3_2_2_28_1","first-page":"6840","volume-title":"Lin (Eds.)","volume":"33","author":"Ho Jonathan","year":"2020","unstructured":"Jonathan Ho, Ajay Jain, and Pieter Abbeel. 2020. Denoising Diffusion Probabilistic Models. In Advances in Neural Information Processing Systems, H. Larochelle, M. Ranzato, R. Hadsell, M.F. Balcan, and H. Lin (Eds.), Vol. 33. Curran Associates, Inc., 6840-6851. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2020\/file\/4c5bcfec8584af0d967f1ab10179ca4b-Paper.pdf"},{"key":"e_1_3_2_2_29_1","volume-title":"Video diffusion models. Advances in neural information processing systems","author":"Ho Jonathan","year":"2022","unstructured":"Jonathan Ho, Tim Salimans, Alexey Gritsenko, William Chan, Mohammad Norouzi, and David J Fleet. 2022. Video diffusion models. Advances in neural information processing systems, Vol. 35 (2022), 8633-8646."},{"key":"e_1_3_2_2_30_1","volume-title":"Cogvideo: Large-scale pretraining for text-to-video generation via transformers. arXiv preprint arXiv:2205.15868","author":"Hong Wenyi","year":"2022","unstructured":"Wenyi Hong, Ming Ding, Wendi Zheng, Xinghan Liu, and Jie Tang. 2022. Cogvideo: Large-scale pretraining for text-to-video generation via transformers. arXiv preprint arXiv:2205.15868 (2022)."},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01768"},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02060"},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00639"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00510"},{"key":"e_1_3_2_2_35_1","volume-title":"Videopoet: A large language model for zero-shot video generation. arXiv preprint arXiv:2312.14125","author":"Kondratyuk Dan","year":"2023","unstructured":"Dan Kondratyuk, Lijun Yu, Xiuye Gu, Jos\u00e9 Lezama, Jonathan Huang, Grant Schindler, Rachel Hornung, Vighnesh Birodkar, Jimmy Yan, Ming-Chang Chiu, et al., 2023. Videopoet: A large language model for zero-shot video generation. arXiv preprint arXiv:2312.14125 (2023)."},{"key":"e_1_3_2_2_36_1","volume-title":"Hunyuanvideo: A systematic framework for large video generative models. arXiv preprint arXiv:2412.03603","author":"Kong Weijie","year":"2024","unstructured":"Weijie Kong, Qi Tian, Zijian Zhang, Rox Min, Zuozhuo Dai, Jin Zhou, Jiangfeng Xiong, Xin Li, Bo Wu, Jianwei Zhang, et al., 2024. Hunyuanvideo: A systematic framework for large video generative models. arXiv preprint arXiv:2412.03603 (2024)."},{"key":"e_1_3_2_2_37_1","unstructured":"LAION-AI. 2022. aesthetic-predictor. https:\/\/github.com\/LAION-AI\/aesthetic-predictor."},{"key":"e_1_3_2_2_38_1","first-page":"1","article-title":"A path towards autonomous machine intelligence version 0.9. 2, 2022-06-27","volume":"62","author":"LeCun Yann","year":"2022","unstructured":"Yann LeCun. 2022. A path towards autonomous machine intelligence version 0.9. 2, 2022-06-27. Open Review, Vol. 62, 1 (2022), 1-62.","journal-title":"Open Review"},{"key":"e_1_3_2_2_39_1","volume-title":"Robotic world model: A neural network simulator for robust policy optimization in robotics. arXiv preprint arXiv:2501.10100","author":"Li Chenhao","year":"2025","unstructured":"Chenhao Li, Andreas Krause, and Marco Hutter. 2025. Robotic world model: A neural network simulator for robust policy optimization in robotics. arXiv preprint arXiv:2501.10100 (2025)."},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00945"},{"key":"e_1_3_2_2_41_1","first-page":"413","article-title":"Isolation forest. In 2008 eighth ieee international conference on data mining","author":"Liu Fei Tony","year":"2008","unstructured":"Fei Tony Liu, Kai Ming Ting, and Zhi-Hua Zhou. 2008. Isolation forest. In 2008 eighth ieee international conference on data mining. IEEE, 413-422.","journal-title":"IEEE"},{"key":"e_1_3_2_2_42_1","volume-title":"Mardini: Masked autoregressive diffusion for video generation at scale. arXiv preprint arXiv:2410.20280","author":"Liu Haozhe","year":"2024","unstructured":"Haozhe Liu, Shikun Liu, Zijian Zhou, Mengmeng Xu, Yanping Xie, Xiao Han, Juan C P\u00e9rez, Ding Liu, Kumara Kahatapitiya, Menglin Jia, et al., 2024a. Mardini: Masked autoregressive diffusion for video generation at scale. arXiv preprint arXiv:2410.20280 (2024)."},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/3643832.3661872"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASE.2025.3534143"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"crossref","unstructured":"NVIDIA. 2025. COSMOS Predict2. https:\/\/github.com\/nvidia-cosmos\/cosmos-predict2. Accessed: 2025-08-26.","DOI":"10.70558\/COSMOS.2025.v2.i4.25434"},{"key":"e_1_3_2_2_46_1","volume-title":"Sora: High-Fidelity Video Generation. https:\/\/openai.com\/sora\/. Accessed: 2025-05-30.","author":"AI.","year":"2023","unstructured":"OpenAI. 2023. Sora: High-Fidelity Video Generation. https:\/\/openai.com\/sora\/. Accessed: 2025-05-30."},{"key":"e_1_3_2_2_47_1","doi-asserted-by":"publisher","DOI":"10.3390\/drones8100572"},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00387"},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.scitotenv.2020.136546"},{"key":"e_1_3_2_2_50_1","volume-title":"Gaia-2: A controllable multi-view generative world model for autonomous driving. arXiv preprint arXiv:2503.20523","author":"Russell Lloyd","year":"2025","unstructured":"Lloyd Russell, Anthony Hu, Lorenzo Bertoni, George Fedoseev, Jamie Shotton, Elahe Arani, and Gianluca Corrado. 2025. Gaia-2: A controllable multi-view generative world model for autonomous driving. arXiv preprint arXiv:2503.20523 (2025)."},{"key":"e_1_3_2_2_51_1","doi-asserted-by":"publisher","DOI":"10.3389\/frobt.2023.1253049"},{"key":"e_1_3_2_2_52_1","doi-asserted-by":"publisher","DOI":"10.1126\/scirobotics.abg1188"},{"key":"e_1_3_2_2_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00723"},{"key":"e_1_3_2_2_54_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58536-5_24"},{"key":"e_1_3_2_2_55_1","volume-title":"Tarun Gupta, Darren Gehring, et al.","author":"Tot Marko","year":"2025","unstructured":"Marko Tot, Shu Ishida, Abdelhak Lemkhenter, David Bignell, Pallavi Choudhury, Chris Lovett, Luis Fran\u00e7a, Matheus Ribeiro Furtado de Mendon\u00e7a, Tarun Gupta, Darren Gehring, et al., 2025. Adapting a World Model for Trajectory Following in a 3D Game. arXiv preprint arXiv:2504.12299 (2025)."},{"key":"e_1_3_2_2_56_1","volume-title":"Karol Kurach, Raphael Marinier, Marcin Michalski, and Sylvain Gelly.","author":"Unterthiner Thomas","year":"2018","unstructured":"Thomas Unterthiner, Sjoerd Van Steenkiste, Karol Kurach, Raphael Marinier, Marcin Michalski, and Sylvain Gelly. 2018. Towards accurate generative models of video: A new metric & challenges. arXiv preprint arXiv:1812.01717 (2018)."},{"key":"e_1_3_2_2_57_1","volume-title":"Wan: Open and advanced large-scale video generative models. arXiv preprint arXiv:2503.20314","author":"Wan Team","year":"2025","unstructured":"Team Wan, Ang Wang, Baole Ai, Bin Wen, Chaojie Mao, Chen-Wei Xie, Di Chen, Feiwu Yu, Haiming Zhao, Jianxiao Yang, et al., 2025. Wan: Open and advanced large-scale video generative models. arXiv preprint arXiv:2503.20314 (2025)."},{"key":"e_1_3_2_2_58_1","doi-asserted-by":"publisher","DOI":"10.1145\/3715014.3722048"},{"key":"e_1_3_2_2_59_1","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM52122.2024.10621375"},{"key":"e_1_3_2_2_60_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00499"},{"key":"e_1_3_2_2_61_1","volume-title":"Videofactory: Swap attention in spatiotemporal diffusions for text-to-video generation.","author":"Wang Wenjing","year":"2023","unstructured":"Wenjing Wang, Huan Yang, Zixi Tuo, Huiguo He, Junchen Zhu, Jianlong Fu, and Jiaying Liu. 2023. Videofactory: Swap attention in spatiotemporal diffusions for text-to-video generation. (2023)."},{"key":"e_1_3_2_2_62_1","volume-title":"DriveDreamer: Towards Real-World-Drive World Models for Autonomous Driving. In European Conference on Computer Vision. Springer, 55-72","author":"Wang Xiaofeng","year":"2024","unstructured":"Xiaofeng Wang, Zheng Zhu, Guan Huang, Xinze Chen, Jiagang Zhu, and Jiwen Lu. 2024b. DriveDreamer: Towards Real-World-Drive World Models for Autonomous Driving. In European Conference on Computer Vision. Springer, 55-72."},{"key":"e_1_3_2_2_63_1","volume-title":"Worlddreamer: Towards general world models for video generation via predicting masked tokens. arXiv preprint arXiv:2401.09985","author":"Wang Xiaofeng","year":"2024","unstructured":"Xiaofeng Wang, Zheng Zhu, Guan Huang, Boyuan Wang, Xinze Chen, and Jiwen Lu. 2024c. Worlddreamer: Towards general world models for video generation via predicting masked tokens. arXiv preprint arXiv:2401.09985 (2024)."},{"key":"e_1_3_2_2_64_1","volume-title":"Conference on robot learning. PMLR, 2226-2240","author":"Wu Philipp","year":"2023","unstructured":"Philipp Wu, Alejandro Escontrela, Danijar Hafner, Pieter Abbeel, and Ken Goldberg. 2023. Daydreamer: World models for physical robot learning. In Conference on robot learning. PMLR, 2226-2240."},{"key":"e_1_3_2_2_65_1","doi-asserted-by":"publisher","DOI":"10.1145\/3636534.3694730"},{"key":"e_1_3_2_2_66_1","volume-title":"Cogvideox: Text-to-video diffusion models with an expert transformer. arXiv preprint arXiv:2408.06072","author":"Yang Zhuoyi","year":"2024","unstructured":"Zhuoyi Yang, Jiayan Teng, Wendi Zheng, Ming Ding, Shiyu Huang, Jiazheng Xu, Yuanming Yang, Wenyi Hong, Xiaohan Zhang, Guanyu Feng, et al., 2024. Cogvideox: Text-to-video diffusion models with an expert transformer. arXiv preprint arXiv:2408.06072 (2024)."},{"key":"e_1_3_2_2_67_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00869"},{"key":"e_1_3_2_2_68_1","volume-title":"Thang Luong, Gunjan Baid, Zirui Wang, Vijay Vasudevan, Alexander Ku, Yinfei Yang, Burcu Karagol Ayan, et al.","author":"Yu Jiahui","year":"2022","unstructured":"Jiahui Yu, Yuanzhong Xu, Jing Yu Koh, Thang Luong, Gunjan Baid, Zirui Wang, Vijay Vasudevan, Alexander Ku, Yinfei Yang, Burcu Karagol Ayan, et al., 2022. Scaling autoregressive models for content-rich text-to-image generation. arXiv preprint arXiv:2206.10789, Vol. 2, 3 (2022), 5."},{"key":"e_1_3_2_2_69_1","volume-title":"How to enable llm with 3d capacity? a survey of spatial reasoning in llm. arXiv preprint arXiv:2504.05786","author":"Zha Jirong","year":"2025","unstructured":"Jirong Zha, Yuxuan Fan, Xiao Yang, Chen Gao, and Xinlei Chen. 2025. How to enable llm with 3d capacity? a survey of spatial reasoning in llm. arXiv preprint arXiv:2504.05786 (2025)."},{"key":"e_1_3_2_2_70_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2022.3232854"},{"key":"e_1_3_2_2_71_1","volume-title":"CityNavAgent: Aerial Vision-and-Language Navigation with Hierarchical Semantic Planning and Global Memory. arXiv preprint arXiv:2505.05622","author":"Zhang Weichen","year":"2025","unstructured":"Weichen Zhang, Chen Gao, Shiquan Yu, Ruiying Peng, Baining Zhao, Qian Zhang, Jinqiang Cui, Xinlei Chen, and Yong Li. 2025. CityNavAgent: Aerial Vision-and-Language Navigation with Hierarchical Semantic Planning and Global Memory. arXiv preprint arXiv:2505.05622 (2025)."},{"key":"e_1_3_2_2_72_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.acl-long.1558"},{"key":"e_1_3_2_2_73_1","doi-asserted-by":"crossref","unstructured":"Baining Zhao Ziyou Wang Jianjie Fang Chen Gao Fanhang Man Jinqiang Cui Xin Wang Xinlei Chen Yong Li and Wenwu Zhu. 2025b. Embodied-R: Collaborative Framework for Activating Embodied Spatial Reasoning in Foundation Models via Reinforcement Learning. arXiv:2504.12680 [cs.AI] https:\/\/arxiv.org\/abs\/2504.12680","DOI":"10.1145\/3746027.3755703"},{"key":"e_1_3_2_2_74_1","volume-title":"Identifying and solving conditional image leakage in image-to-video diffusion model. arXiv preprint arXiv:2406.15735","author":"Zhao Min","year":"2024","unstructured":"Min Zhao, Hongzhou Zhu, Chendong Xiang, Kaiwen Zheng, Chongxuan Li, and Jun Zhu. 2024. Identifying and solving conditional image leakage in image-to-video diffusion model. arXiv preprint arXiv:2406.15735 (2024)."},{"key":"e_1_3_2_2_75_1","doi-asserted-by":"crossref","unstructured":"Nan Zhou Yuxuan Liu Haoyang Wang Fanhang Man Jingao Xu Fan Dang Chaopeng Hong Yunhao Liu Xiao-Ping Zhang Yali Song et al. 2025. CatUA: Catalyzing Urban Air Quality Intelligence through Mobile Crowd-sensing. IEEE Transactions on Mobile Computing (2025).","DOI":"10.1109\/TMC.2025.3560120"},{"key":"e_1_3_2_2_76_1","volume-title":"Robodreamer: Learning compositional world models for robot imagination. arXiv preprint arXiv:2404.12377","author":"Zhou Siyuan","year":"2024","unstructured":"Siyuan Zhou, Yilun Du, Jiaben Chen, Yandong Li, Dit-Yan Yeung, and Chuang Gan. 2024. Robodreamer: Learning compositional world models for robot imagination. arXiv preprint arXiv:2404.12377 (2024)."}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3758180","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:15:45Z","timestamp":1765307745000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3758180"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":76,"alternative-id":["10.1145\/3746027.3758180","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3758180","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}