{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T14:25:11Z","timestamp":1784643911477,"version":"3.55.0"},"reference-count":53,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2022YFA1004000"],"award-info":[{"award-number":["2022YFA1004000"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100004826","name":"Beijing Natural Science Foundation","doi-asserted-by":"publisher","award":["L253007,4242052"],"award-info":[{"award-number":["L253007,4242052"]}],"id":[{"id":"10.13039\/501100004826","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62173325"],"award-info":[{"award-number":["62173325"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,10,19]]},"DOI":"10.1109\/iccv51701.2025.02659","type":"proceedings-article","created":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T19:45:49Z","timestamp":1777491949000},"page":"28632-28642","source":"Crossref","is-referenced-by-count":5,"title":["World4Drive: End-to-End Autonomous Driving via Intention-Aware Physical Latent World Model"],"prefix":"10.1109","author":[{"given":"Yupeng","family":"Zheng","sequence":"first","affiliation":[{"name":"Institute of Automation, Chinese Academy of Sciences"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Pengxuan","family":"Yang","sequence":"additional","affiliation":[{"name":"Institute of Automation, Chinese Academy of Sciences"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zebin","family":"Xing","sequence":"additional","affiliation":[{"name":"Institute of Automation, Chinese Academy of Sciences"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qichao","family":"Zhang","sequence":"additional","affiliation":[{"name":"Institute of Automation, Chinese Academy of Sciences"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuhang","family":"Zheng","sequence":"additional","affiliation":[{"name":"National University of Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yinfeng","family":"Gao","sequence":"additional","affiliation":[{"name":"Institute of Automation, Chinese Academy of Sciences"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Pengfei","family":"Li","sequence":"additional","affiliation":[{"name":"Tsinghua University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Teng","family":"Zhang","sequence":"additional","affiliation":[{"name":"School of Artificial Intelligence, UCAS"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhongpu","family":"Xia","sequence":"additional","affiliation":[{"name":"Institute of Automation, Chinese Academy of Sciences"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Peng","family":"Jia","sequence":"additional","affiliation":[{"name":"Li Auto"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"XianPeng","family":"Lang","sequence":"additional","affiliation":[{"name":"Li Auto"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dongbin","family":"Zhao","sequence":"additional","affiliation":[{"name":"Institute of Automation, Chinese Academy of Sciences"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Vavim and vavam: Autonomous driving through video generative modeling","author":"Bartoccioni","year":"2025","journal-title":"arXiv preprint"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01164"},{"key":"ref3","article-title":"Vadv2: End-to-end vectorized autonomous driving via probabilistic planning","author":"Chen","year":"2024","journal-title":"arXiv preprint"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72761-0_14"},{"key":"ref5","volume-title":"Openscene: The largest up-todate 3d occupancy prediction benchmark in autonomous driving","author":"Contributors","year":"2023"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.52202\/079017-0902"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.52202\/079017-2906"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/TIV.2024.3408830"},{"key":"ref9","article-title":"Dome: Taming diffusion model into high-fidelity controllable occupancy world model","author":"Gu","year":"2024","journal-title":"arXiv preprint"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref11","article-title":"Denoising diffusion probabilistic models","author":"Ho","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref12","article-title":"Gaia-1: A generative world model for autonomous driving","author":"Hu","year":"2023","journal-title":"arXiv preprint"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2024.3444912"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19839-7_31"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01712"},{"key":"ref16","article-title":"Drivetransformer: Unified transformer for scalable end-toend autonomous driving","volume-title":"The Thirteenth International Conference on Learning Representations","author":"Jia","year":"2025"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00766"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10160326"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72649-1_21"},{"key":"ref20","article-title":"Semi-supervised vision-centric 3d occupancy world model for autonomous driving","author":"Li","year":"2025","journal-title":"arXiv preprint"},{"key":"ref21","article-title":"Enhancing end-to-end autonomous driving with latent world model","volume-title":"The Thirteenth International Conference on Learning Representations","author":"Li","year":"2025"},{"key":"ref22","article-title":"Hydra-mdp: End-to-end multimodal planning with multitarget hydra-distillation","author":"Li","year":"2024","journal-title":"arXiv preprint"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2024.3515454"},{"key":"ref24","volume-title":"Is ego status all you need for openloop end-to-end autonomous driving? In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Li"},{"key":"ref25","article-title":"Maptr: Structured modeling and learning for online vectorized hd map construction","volume-title":"International Conference on Learning Representations","author":"Liao","year":"2023"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01124"},{"key":"ref27","article-title":"Sparse4d: Multi-view 3d object detection with sparse spatial-temporal fusion","author":"Lin","year":"2022","journal-title":"arXiv preprint"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19812-0_31"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72980-5_15"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01470"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01398"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00700"},{"key":"ref34","article-title":"Grounded sam: Assembling open-world models for diverse visual tasks","author":"Ren","year":"2024","journal-title":"arXiv preprint"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72943-0_15"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA55743.2025.11128800"},{"key":"ref37","article-title":"Tokenize the world into object-level knowledge to address long-tail events in autonomous driving","author":"Tian","year":"2024","journal-title":"arXiv preprint"},{"key":"ref38","article-title":"Drivevlm: The convergence of autonomous driving and large vision-language models","volume-title":"The Conference on Robot Learning","author":"Tian","year":"2024"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00772"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73195-2_4"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01397"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01463"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00157"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01710"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA55743.2025.11128148"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00830"},{"key":"ref47","article-title":"Copilot4d: Learning unsupervised world models for autonomous driving via discrete diffusion","volume-title":"The Twelfth International Conference on Learning Representations","author":"Zhang","year":"2024"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i10.33130"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72624-8_4"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73650-6_6"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA57147.2024.10611261"},{"key":"ref52","article-title":"Preliminary investigation into data scaling laws for imitation learning-based end-to-end autonomous driving","author":"Zheng","year":"2024","journal-title":"arXiv preprint"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/tcds.2026.3664120"}],"event":{"name":"2025 IEEE\/CVF International Conference on Computer Vision (ICCV)","location":"Honolulu, HI, USA","start":{"date-parts":[[2025,10,19]]},"end":{"date-parts":[[2025,10,25]]}},"container-title":["2025 IEEE\/CVF International Conference on Computer Vision (ICCV)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11443115\/11443287\/11444588.pdf?arnumber=11444588","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T05:00:00Z","timestamp":1777611600000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11444588\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,19]]},"references-count":53,"URL":"https:\/\/doi.org\/10.1109\/iccv51701.2025.02659","relation":{},"subject":[],"published":{"date-parts":[[2025,10,19]]}}}