{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,11]],"date-time":"2026-04-11T04:54:22Z","timestamp":1775883262386,"version":"3.50.1"},"reference-count":32,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"5","license":[{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62503033"],"award-info":[{"award-number":["62503033"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["52272327"],"award-info":[{"award-number":["52272327"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Robot. Autom. Lett."],"published-print":{"date-parts":[[2026,5]]},"DOI":"10.1109\/lra.2026.3678842","type":"journal-article","created":{"date-parts":[[2026,3,30]],"date-time":"2026-03-30T20:08:28Z","timestamp":1774901308000},"page":"6336-6343","source":"Crossref","is-referenced-by-count":0,"title":["Learning Rollout from Sampling: An R1-Style Tokenized Traffic Simulation Model"],"prefix":"10.1109","volume":"11","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-7170-0729","authenticated-orcid":false,"given":"Ziyan","family":"Wang","sequence":"first","affiliation":[{"name":"State Key Laboratory of Intelligent Transportation System, Key Laboratory of Autonomous Transportation Technology for Special Vehicles, Ministry of Industry and Information Technology, School of Transportation Science and Engineering, Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8076-8989","authenticated-orcid":false,"given":"Peng","family":"Chen","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Intelligent Transportation System, Key Laboratory of Autonomous Transportation Technology for Special Vehicles, Ministry of Industry and Information Technology, School of Transportation Science and Engineering, Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-4613-8610","authenticated-orcid":false,"given":"Ding","family":"Li","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Intelligent Transportation System, Key Laboratory of Autonomous Transportation Technology for Special Vehicles, Ministry of Industry and Information Technology, School of Transportation Science and Engineering, Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chiwei","family":"Li","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Intelligent Transportation System, Key Laboratory of Autonomous Transportation Technology for Special Vehicles, Ministry of Industry and Information Technology, School of Transportation Science and Engineering, Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9747-391X","authenticated-orcid":false,"given":"Qichao","family":"Zhang","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Multimodal Artificial Intelligence Systems, Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhongpu","family":"Xia","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Multimodal Artificial Intelligence Systems, Institute of Automation, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8374-7422","authenticated-orcid":false,"given":"Guizhen","family":"Yu","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Intelligent Transportation System, Key Laboratory of Autonomous Transportation Technology for Special Vehicles, Ministry of Industry and Information Technology, School of Transportation Science and Engineering, Beihang University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","first-page":"26933","article-title":"DriveArena: A closed-loop generative simulation platform for autonomous driving","volume-title":"Proc. IEEE\/CVF Int. Conf. Comput. Vis.","author":"Yang","year":"2025"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1038\/s41467-021-21007-8"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2022.3144501"},{"key":"ref4","article-title":"GPT-4 technical report","author":"Achiam","year":"2024"},{"key":"ref5","first-page":"114048","article-title":"SMART: Scalable multi-agent real-time simulation via next-token prediction","volume-title":"Proc. Annu. Conf. Neural Inf. Process. Syst.","volume":"37","author":"Wu","year":"2024"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00510"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1082"},{"key":"ref8","article-title":"Revisit mixture models for multi-agent simulation: Experimental study within a unified framework","author":"Lin","year":"2025"},{"key":"ref9","article-title":"The entropy mechanism of reinforcement learning for reasoning language models","author":"Cui","year":"2025"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4419-7970-4"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00536"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-025-09422-z"},{"key":"ref13","article-title":"DeepSeekMath: Pushing the limits of mathematical reasoning in open language models","author":"Shao","year":"2024"},{"key":"ref14","article-title":"Trajeglish: Traffic modeling as next-token prediction","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Philion","year":"2024"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00788"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.52202\/079017-2527"},{"key":"ref17","article-title":"Causal confusion in imitation learning","volume-title":"Proc. Annu. Conf. Neural Inf. Process. Syst.","volume":"32","author":"Haan","year":"2019"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/IROS55552.2023.10342038"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA55743.2025.11127286"},{"key":"ref20","article-title":"Finetuning generative trajectory model with reinforcement learning from human feedback","author":"Li","year":"2025"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01607"},{"key":"ref22","article-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.emnlp-main.1087"},{"key":"ref24","article-title":"Plan then action: High-level planning guidance reinforcement learning for LLM reasoning","author":"Dou","year":"2025"},{"key":"ref25","article-title":"AlphaDrive: Unleashing the power of VLMs in autonomous driving via reinforcement learning and reasoning","author":"Jiang","year":"2025"},{"key":"ref26","article-title":"Plan-R1: Safe and feasible trajectory planning as language modeling","author":"Tang","year":"2025"},{"key":"ref27","article-title":"Beyond the 80\/20 rule: High-entropy minority tokens drive effective reinforcement learning for LLM reasoning","author":"Wang","year":"2025"},{"key":"ref28","article-title":"DROPE: Directional rotary position embedding for efficient agent interaction modeling","author":"Zhao","year":"2025"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2024.3518308"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72946-1_22"},{"key":"ref31","first-page":"59151","article-title":"The waymo open sim agents challenge","volume-title":"Proc. Annu. Conf. Neural Inf. Process. Syst.","volume":"36","author":"Montali","year":"2023"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00957"}],"container-title":["IEEE Robotics and Automation Letters"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/7083369\/11435997\/11457596.pdf?arnumber=11457596","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,11]],"date-time":"2026-04-11T04:23:07Z","timestamp":1775881387000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11457596\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5]]},"references-count":32,"journal-issue":{"issue":"5"},"URL":"https:\/\/doi.org\/10.1109\/lra.2026.3678842","relation":{},"ISSN":["2377-3766","2377-3774"],"issn-type":[{"value":"2377-3766","type":"electronic"},{"value":"2377-3774","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,5]]}}}