{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,1]],"date-time":"2025-10-01T15:13:06Z","timestamp":1759331586688,"version":"build-2065373602"},"reference-count":46,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,8,17]],"date-time":"2025-08-17T00:00:00Z","timestamp":1755388800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,8,17]],"date-time":"2025-08-17T00:00:00Z","timestamp":1755388800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,8,17]]},"DOI":"10.1109\/case58245.2025.11164052","type":"proceedings-article","created":{"date-parts":[[2025,9,23]],"date-time":"2025-09-23T17:24:07Z","timestamp":1758648247000},"page":"2766-2773","source":"Crossref","is-referenced-by-count":0,"title":["Transformer-Based World Interaction Modeling for Humanoid Locomotion Control"],"prefix":"10.1109","author":[{"given":"Han","family":"Zheng","sequence":"first","affiliation":[{"name":"Tsinghua University,Tsinghua Shenzhen International Graduate School,Shenzhen,China,518055"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yi","family":"Cheng","sequence":"additional","affiliation":[{"name":"Tsinghua University,Tsinghua Shenzhen International Graduate School,Shenzhen,China,518055"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hang","family":"Liu","sequence":"additional","affiliation":[{"name":"University of Michigan,Ann Arbor,MI,USA,48109"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiayi","family":"Li","sequence":"additional","affiliation":[{"name":"Tsinghua University,Tsinghua Shenzhen International Graduate School,Shenzhen,China,518055"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yizhe","family":"Li","sequence":"additional","affiliation":[{"name":"Tsinghua University,Tsinghua Shenzhen International Graduate School,Shenzhen,China,518055"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Linqi","family":"Ye","sequence":"additional","affiliation":[{"name":"Shanghai University,Shanghai,China,200444"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Houde","family":"Liu","sequence":"additional","affiliation":[{"name":"Tsinghua University,Tsinghua Shenzhen International Graduate School,Shenzhen,China,518055"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2014.6906613"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/HUMANOIDS47582.2021.9555782"},{"volume-title":"Development and Real-Time Optimization-based Control of a Full-sized Humanoid for Dynamic Walking and Running","year":"2023","author":"Ahn","key":"ref3"},{"key":"ref4","first-page":"22","article-title":"Walk these ways: Tuning robot control for generalization with multiplicity of behavior","volume-title":"Conference on Robot Learning","author":"Margolis"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1177\/02783649241285161"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/icra57147.2024.10610978"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2023.3290509"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA40945.2020.9196777"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1285"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/IROS47612.2022.9981973"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01000"},{"article-title":"Humanoid locomotion as next token prediction","year":"2024","author":"Radosavovic","key":"ref12"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1126\/scirobotics.abc5986"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2021.XVII.011"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10160497"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/Humanoids57100.2023.10375167"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2022.3151396"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10161144"},{"article-title":"Hybrid internal model: A simple and efficient learner for agile legged locomotion","year":"2023","author":"Long","key":"ref19"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1007\/3-540-31635-3_22"},{"key":"ref21","first-page":"2226","article-title":"Day-dreamer: World models for physical robot learning","volume-title":"Conference on robot learning","author":"Wu"},{"article-title":"Mastering atari with discrete world models","year":"2020","author":"Hafner","key":"ref22"},{"article-title":"Mastering diverse domains through world models","year":"2023","author":"Hafner","key":"ref23"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.15607\/rss.2024.xx.058"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","year":"2020","author":"Dosovitskiy","key":"ref26"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"ref28","first-page":"10 347","article-title":"Training data-efficient image transformers & distillation through attention","volume-title":"International conference on machine learning","author":"Touvron"},{"article-title":"Bert: Pre-training of deep bidirectional transformers for language understanding","year":"2018","author":"Devlin","key":"ref29"},{"key":"ref30","article-title":"Improving language understanding by generative pre-training","author":"Radford","year":"2018","journal-title":"None"},{"issue":"8","key":"ref31","first-page":"9","article-title":"Language models are unsupervised multitask learners","volume":"1","author":"Radford","year":"2019","journal-title":"OpenAI blog"},{"article-title":"Language models are few-shot learners","year":"2020","author":"Brown","key":"ref32"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/ICPR48806.2021.9412190"},{"key":"ref34","first-page":"15 084","article-title":"Decision transformer: Reinforcement learning via sequence modeling","volume":"34","author":"Chen","year":"2021","journal-title":"Advances in neural information processing systems"},{"article-title":"Learning vision-guided quadrupedal locomotion end-to-end with cross-modal transformers","year":"2021","author":"Yang","key":"ref35"},{"article-title":"Humanplus: Humanoid shadowing and imitation from humans","year":"2024","author":"Fu","key":"ref36"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1126\/scirobotics.adi9579"},{"article-title":"Transdreamer: Reinforcement learning with transformer world models","year":"2022","author":"Chen","key":"ref38"},{"article-title":"Transformers are sample-efficient world models","year":"2022","author":"Micheli","key":"ref39"},{"article-title":"Transformer-based world models are happy with 100k interactions","year":"2023","author":"Robine","key":"ref40"},{"key":"ref41","article-title":"Storm: Efficient stochastic transformer based world models for reinforcement learning","volume":"36","author":"Zhang","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref42","article-title":"Facing off world model backbones: Rnns, transformers, and s4","volume":"36","author":"Deng","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"article-title":"Rethinking transformers in solving pomdps","year":"2024","author":"Lu","key":"ref43"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1145\/1553374.1553380"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48506.2021.9561814"},{"key":"ref46","doi-asserted-by":"crossref","DOI":"10.15607\/RSS.2020.XVI.031","article-title":"Learning memory-based control for human-scale bipedal locomotion","author":"Siekmann","year":"2020"}],"event":{"name":"2025 IEEE 21st International Conference on Automation Science and Engineering (CASE)","start":{"date-parts":[[2025,8,17]]},"location":"Los Angeles, CA, USA","end":{"date-parts":[[2025,8,21]]}},"container-title":["2025 IEEE 21st International Conference on Automation Science and Engineering (CASE)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11163731\/11163732\/11164052.pdf?arnumber=11164052","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,30]],"date-time":"2025-09-30T13:24:55Z","timestamp":1759238695000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11164052\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,8,17]]},"references-count":46,"URL":"https:\/\/doi.org\/10.1109\/case58245.2025.11164052","relation":{},"subject":[],"published":{"date-parts":[[2025,8,17]]}}}