{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T23:53:57Z","timestamp":1783641237473,"version":"3.55.0"},"reference-count":27,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,5,19]],"date-time":"2025-05-19T00:00:00Z","timestamp":1747612800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,5,19]],"date-time":"2025-05-19T00:00:00Z","timestamp":1747612800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,5,19]]},"DOI":"10.1109\/icra55743.2025.11127521","type":"proceedings-article","created":{"date-parts":[[2025,9,2]],"date-time":"2025-09-02T17:28:56Z","timestamp":1756834136000},"page":"3647-3653","source":"Crossref","is-referenced-by-count":1,"title":["Dynamic Non-Prehensile Object Transport via Model-Predictive Reinforcement Learning"],"prefix":"10.1109","author":[{"given":"Neel","family":"Jawale","sequence":"first","affiliation":[{"name":"University of Washington"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Byron","family":"Boots","sequence":"additional","affiliation":[{"name":"University of Washington"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Balakumar","family":"Sundaralingam","sequence":"additional","affiliation":[{"name":"NVIDIA,USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mohak","family":"Bhardwaj","sequence":"additional","affiliation":[{"name":"University of Washington"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Human-in-the-loop imitation learning using remote teleoperation","author":"Mandlekar","year":"2020","journal-title":"arXiv preprint"},{"key":"ref2","first-page":"879","article-title":"Roboturk: A crowdsourcing platform for robotic skill learning through imitation","volume-title":"Conference on Robot Learning. PMLR","author":"Mandlekar"},{"key":"ref3","first-page":"627","article-title":"A reduction of imitation learning and structured prediction to no-regret online learning","volume-title":"Proceedings of the fourteenth international conference on artificial intelligence and statistics. JMLR Workshop and Conference Proceedings","author":"Ross"},{"key":"ref4","article-title":"Aloha unleashed: A simple recipe for robot dexterity","volume-title":"8th Annual Conference on Robot Learning","author":"Zhao"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/TRO.2021.3086773"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/TCST.2023.3277224"},{"key":"ref7","first-page":"750","article-title":"Storm: An integrated framework for fast joint-space model-predictive control for reactive manipulation","volume-title":"Conference on Robot Learning. PMLR","author":"Bhardwaj"},{"key":"ref8","article-title":"Sampling-based MPC using a GPU-parallelizable physics simulator as dynamic model: an open source implementation with isaacgym","volume-title":"Embracing Contacts - Workshop at ICRA 2023","author":"Pezzato","year":"2023"},{"key":"ref9","article-title":"Plan online, learn offline: Efficient learning and exploration via model-based control","author":"Lowrey","year":"2018","journal-title":"arXiv preprint"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.7551\/mitpress\/9481.003.0015"},{"key":"ref11","article-title":"Blending mpc & value function approximation for efficient reinforcement learning","author":"Bhardwaj","year":"2020","journal-title":"arXiv preprint"},{"key":"ref12","article-title":"Offline reinforcement learning: Tutorial, review, and perspectives on open problems","author":"Levine","year":"2020","journal-title":"arXiv preprint"},{"key":"ref13","first-page":"1179","article-title":"Conservative qlearning for offline reinforcement learning","volume":"33","author":"Kumar","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1177\/0278364910371999"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2016.7487277"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2018.8594448"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2023.3324520"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA46639.2022.9812424"},{"key":"ref19","first-page":"19360","article-title":"Mahalo: Unifying offline reinforcement learning and imitation learning from observations","volume-title":"International Conference on Machine Learning. PMLR","author":"Li"},{"key":"ref20","article-title":"Reinforcement learning: An introduction","author":"Sutton","year":"2018","journal-title":"A Bradford Book"},{"key":"ref21","article-title":"Neuro-dynamic programming","author":"Bertsekas","year":"1996","journal-title":"Athena Scientific"},{"key":"ref22","first-page":"21810","article-title":"Morel: Model-based offline reinforcement learning","volume":"33","author":"Kidambi","year":"2020","journal-title":"Advances in neural information processing systems"},{"key":"ref23","first-page":"7436","article-title":"Uncertainty-based offline reinforcement learning with diversified q-ensemble","volume":"34","author":"An","year":"2021","journal-title":"Advances in neural information processing systems"},{"key":"ref24","first-page":"3852","article-title":"Adversarially trained actor critic for offline reinforcement learning","volume-title":"International Conference on Machine Learning. PMLR","author":"Cheng"},{"key":"ref25","first-page":"6683","article-title":"Bellmanconsistent pessimism for offline reinforcement learning","volume":"34","author":"Xie","year":"2021","journal-title":"Advances in neural information processing systems"},{"key":"ref26","first-page":"1622","article-title":"Learning off-policy with online planning","volume-title":"Conference on Robot Learning","author":"Sikchi"},{"key":"ref27","article-title":"Model-based offline planning","author":"Argenson","year":"2020","journal-title":"arXiv preprint"}],"event":{"name":"2025 IEEE International Conference on Robotics and Automation (ICRA)","location":"Atlanta, GA, USA","start":{"date-parts":[[2025,5,19]]},"end":{"date-parts":[[2025,5,23]]}},"container-title":["2025 IEEE International Conference on Robotics and Automation (ICRA)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11127273\/11127223\/11127521.pdf?arnumber=11127521","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,3]],"date-time":"2025-09-03T06:15:18Z","timestamp":1756880118000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11127521\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,19]]},"references-count":27,"URL":"https:\/\/doi.org\/10.1109\/icra55743.2025.11127521","relation":{},"subject":[],"published":{"date-parts":[[2025,5,19]]}}}