{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T15:46:34Z","timestamp":1783611994893,"version":"3.55.0"},"reference-count":66,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"11","license":[{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100002855","name":"National Science and Technology Major Project of the Ministry of Science and Technology of China","doi-asserted-by":"publisher","award":["2019YFC1408703"],"award-info":[{"award-number":["2019YFC1408703"]}],"id":[{"id":"10.13039\/501100002855","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62022048"],"award-info":[{"award-number":["62022048"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62276150"],"award-info":[{"award-number":["62276150"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100020721","name":"Guoqiang Institute, Tsinghua University","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100020721","id-type":"DOI","asserted-by":"publisher"}]},{"name":"National Defense Basic Science and Technology Strengthening Program of China"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Neural Netw. Learning Syst."],"published-print":{"date-parts":[[2024,11]]},"DOI":"10.1109\/tnnls.2023.3293508","type":"journal-article","created":{"date-parts":[[2023,11,7]],"date-time":"2023-11-07T19:12:02Z","timestamp":1699384322000},"page":"16288-16300","source":"Crossref","is-referenced-by-count":13,"title":["Hundreds Guide Millions: Adaptive Offline Reinforcement Learning With Expert Guidance"],"prefix":"10.1109","volume":"35","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2587-2660","authenticated-orcid":false,"given":"Qisen","family":"Yang","sequence":"first","affiliation":[{"name":"Department of Automation, Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-0314-4243","authenticated-orcid":false,"given":"Shenzhi","family":"Wang","sequence":"additional","affiliation":[{"name":"Department of Automation, Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-7094-4694","authenticated-orcid":false,"given":"Qihang","family":"Zhang","sequence":"additional","affiliation":[{"name":"Department of Automation, Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7251-0988","authenticated-orcid":false,"given":"Gao","family":"Huang","sequence":"additional","affiliation":[{"name":"Department of Automation, Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0858-1770","authenticated-orcid":false,"given":"Shiji","family":"Song","sequence":"additional","affiliation":[{"name":"Department of Automation, Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Offline reinforcement learning: Tutorial, review, and perspectives on open problems","author":"Levine","year":"2020","journal-title":"arXiv:2005.01643"},{"key":"ref2","article-title":"A survey on offline reinforcement learning: Taxonomy, review, and open problems","author":"Prudencio","year":"2022","journal-title":"arXiv:2203.01387"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-27645-3_2"},{"key":"ref4","first-page":"1","article-title":"Scalable deep reinforcement learning for vision-based robotic manipulation","volume-title":"Proc. Conf. Robot Learn.","author":"Kalashnikov"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2016.7487517"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1177\/0278364917710318"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2018.8593986"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.2352\/ISSN.2470-1173.2017.19.AVM-023"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2019.8793742"},{"key":"ref10","article-title":"Deep reinforcement learning for sepsis treatment","author":"Raghu","year":"2017","journal-title":"arXiv:1711.09602"},{"key":"ref11","first-page":"1","article-title":"A reinforcement learning approach to weaning of mechanical ventilation in intensive care units","volume-title":"Proc. 33rd Conf. Uncertainty Artif. Intell.","author":"Prasad"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1145\/3219819.3219961"},{"key":"ref13","first-page":"1","article-title":"A minimalist approach to offline reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Fujimoto"},{"key":"ref14","first-page":"1","article-title":"Stabilizing off-policy Q-learning via bootstrapping error reduction","author":"Kumar","year":"2019","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref15","article-title":"Behavior regularized offline reinforcement learning","author":"Wu","year":"2019","journal-title":"arXiv:1911.11361"},{"key":"ref16","first-page":"1","article-title":"Conservative Q-learning for offline reinforcement learning","author":"Kumar","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref17","first-page":"1","article-title":"Offline reinforcement learning with implicit Q-learning","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Kostrikov"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2018.8462901"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11757"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/tnnls.2022.3142822"},{"key":"ref21","first-page":"1","article-title":"Model-agnostic meta-learning for fast adaptation of deep networks","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Finn"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/tnnls.2021.3122579"},{"key":"ref23","article-title":"D4RL: Datasets for deep data-driven reinforcement learning","author":"Fu","year":"2020","journal-title":"arXiv:2004.07219"},{"key":"ref24","first-page":"1","article-title":"Visualizing data using t-SNE","volume":"9","author":"Van der Maaten","year":"2008","journal-title":"J. Mach. Learn. Res."},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2012.6386109"},{"key":"ref26","first-page":"1","article-title":"Off-policy deep reinforcement learning without exploration","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Fujimoto"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00032"},{"key":"ref28","first-page":"1","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Haarnoja"},{"key":"ref29","first-page":"1","article-title":"Meta-weight-net: Learning an explicit mapping for sample weighting","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Shu"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1006\/jcss.1997.1504"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2011.6126229"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2018.2858826"},{"key":"ref33","first-page":"1","article-title":"Addressing function approximation error in actor-critic methods","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Fujimoto"},{"key":"ref34","article-title":"Openai gym","author":"Brockman","year":"2016","journal-title":"arXiv:1606.01540"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2018.XIV.049"},{"key":"ref36","first-page":"1","article-title":"Decision transformer: Reinforcement learning via sequence modeling","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Chen"},{"key":"ref37","first-page":"1","article-title":"RVS: What is essential for offline RL via supervised learning?","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Emmons"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/tnnls.2022.3207346"},{"key":"ref39","first-page":"1","volume-title":"BC+RL: Imitation learning from non-optimal demonstrations","author":"Booher","year":"2019"},{"key":"ref40","article-title":"Accelerating online reinforcement learning with offline datasets","author":"Nair","year":"2020","journal-title":"arXiv:2006.09359"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2018.8463162"},{"key":"ref42","article-title":"Algaedice: Policy gradient from arbitrary experience","author":"Nachum","year":"2019","journal-title":"arXiv:1912.02074"},{"key":"ref43","article-title":"The importance of pessimism in fixed-dataset policy optimization","volume-title":"Int. Conf. Learn. Representations","author":"Buckman"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA.2018.8460487"},{"key":"ref45","first-page":"1","article-title":"Learning to reach goals via iterated supervised learning","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Ghosh"},{"key":"ref46","first-page":"1","article-title":"Learning latent plans from play","volume-title":"Proc. Conf. Robot Learn.","author":"Lynch"},{"key":"ref47","article-title":"Reward-conditioned policies","author":"Kumar","year":"2019","journal-title":"arXiv:1912.13465"},{"key":"ref48","article-title":"Training agents using upside-down reinforcement learning","author":"Srivastava","year":"2019","journal-title":"arXiv:1912.02877"},{"key":"ref49","first-page":"1","article-title":"Rewriting history with inverse RL: hindsight inference for policy improvement","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Eysenbach"},{"key":"ref50","first-page":"1","article-title":"Fitted Q-iteration by advantage weighted regression","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Neumann"},{"key":"ref51","article-title":"Advantage-weighted regression: Simple and scalable off-policy reinforcement learning","author":"Peng","year":"2019","journal-title":"arXiv:1910.00177"},{"key":"ref52","first-page":"1","article-title":"Exponentially weighted imitation learning for batched historical data","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Wang"},{"key":"ref53","first-page":"1","article-title":"BAIL: Best-action imitation learning for batch deep reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Chen"},{"key":"ref54","first-page":"1","article-title":"Self-paced learning for latent variable models","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Kumar"},{"key":"ref55","first-page":"1","article-title":"Active bias: Training more accurate neural networks by emphasizing high variance samples","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Chang"},{"issue":"1","key":"ref56","doi-asserted-by":"crossref","first-page":"117","DOI":"10.1023\/A:1023709501986","article-title":"A framework for robust subspace learning","volume":"54","author":"De la Torre","year":"2003","journal-title":"Int. J. Comput. Vis."},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00131"},{"key":"ref58","first-page":"1","article-title":"Meta networks","volume-title":"Proc. 34th Int. Conf. Mach. Learn.","author":"Munkhdalai"},{"key":"ref59","first-page":"1","article-title":"Optimization as a model for few-shot learning","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Ravi"},{"key":"ref60","first-page":"1","article-title":"Fidelity-weighted learning","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Dehghani"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.4324\/9780203209332-7"},{"key":"ref62","first-page":"1","article-title":"Learning to teach with dynamic loss functions","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Wu"},{"key":"ref63","first-page":"1","article-title":"MentorNet: Learning data-driven curriculum for very deep neural networks on corrupted labels","volume-title":"Proc. 35th Int. Conf. Mach. Learn.","author":"Jiang"},{"key":"ref64","first-page":"1","article-title":"Learning to reweight examples for robust deep learning","volume-title":"Proc. 35th Int. Conf. Mach. Learn.","author":"Ren"},{"key":"ref65","first-page":"1","article-title":"Pytorch: An imperative style, high-performance deep learning library","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Paszke"},{"key":"ref66","first-page":"1","article-title":"Adam: A method for stochastic optimization","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Kingma"}],"container-title":["IEEE Transactions on Neural Networks and Learning Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/5962385\/10737991\/10310284.pdf?arnumber=10310284","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,27]],"date-time":"2024-11-27T19:27:04Z","timestamp":1732735624000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10310284\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11]]},"references-count":66,"journal-issue":{"issue":"11"},"URL":"https:\/\/doi.org\/10.1109\/tnnls.2023.3293508","relation":{},"ISSN":["2162-237X","2162-2388"],"issn-type":[{"value":"2162-237X","type":"print"},{"value":"2162-2388","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11]]}}}