{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,16]],"date-time":"2026-01-16T04:19:39Z","timestamp":1768537179838,"version":"3.49.0"},"publisher-location":"New York, NY, USA","reference-count":52,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,12,8]],"date-time":"2023-12-08T00:00:00Z","timestamp":1701993600000},"content-version":"vor","delay-in-days":394,"URL":"http:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/100000001","name":"NSF (National Science Foundation)","doi-asserted-by":"publisher","award":["1834701, 2038853"],"award-info":[{"award-number":["1834701, 2038853"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000015","name":"DOE U.S. Department of Energy","doi-asserted-by":"publisher","award":["DE-EE0009150"],"award-info":[{"award-number":["DE-EE0009150"]}],"id":[{"id":"10.13039\/100000015","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,11,9]]},"DOI":"10.1145\/3563357.3564064","type":"proceedings-article","created":{"date-parts":[[2022,12,8]],"date-time":"2022-12-08T13:31:36Z","timestamp":1670506296000},"page":"89-98","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":9,"title":["Accelerate online reinforcement learning for building HVAC control with heterogeneous expert guidances"],"prefix":"10.1145","author":[{"given":"Shichao","family":"Xu","sequence":"first","affiliation":[{"name":"Northwestern University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yangyang","family":"Fu","sequence":"additional","affiliation":[{"name":"Texas A&amp;M University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yixuan","family":"Wang","sequence":"additional","affiliation":[{"name":"Northwestern University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhuoran","family":"Yang","sequence":"additional","affiliation":[{"name":"Yale University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zheng","family":"O'Neill","sequence":"additional","affiliation":[{"name":"Texas A&amp;M University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhaoran","family":"Wang","sequence":"additional","affiliation":[{"name":"Northwestern University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qi","family":"Zhu","sequence":"additional","affiliation":[{"name":"Northwestern University"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2022,12,8]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Ahmad Parvaresh, Arman Fathollahi, Meysam Gheisarnejad, and Mohammad-Hassan Khooban.","author":"Abrazeh Saber","year":"2022","unstructured":"Saber Abrazeh, Saeid-Reza Mohseni, Meisam Jahanshahi Zeitouni, Ahmad Parvaresh, Arman Fathollahi, Meysam Gheisarnejad, and Mohammad-Hassan Khooban. 2022. Virtual Hardware-in-the-Loop FMU Co-Simulation Based Digital Twins for Heating, Ventilation, and Air-Conditioning (HVAC) Systems. IEEE Transactions on Emerging Topics in Computational Intelligence (2022)."},{"key":"e_1_3_2_1_2_1","unstructured":"Rishabh Agarwal Dale Schuurmans and Mohammad Norouzi. 2020. An optimistic perspective on offline reinforcement learning. In ICML. PMLR."},{"key":"e_1_3_2_1_3_1","volume-title":"Openai gym. arXiv preprint arXiv:1606.01540","author":"Brockman Greg","year":"2016","unstructured":"Greg Brockman, Vicki Cheung, Ludwig Pettersson, Jonas Schneider, John Schulman, Jie Tang, and Wojciech Zaremba. 2016. Openai gym. arXiv preprint arXiv:1606.01540 (2016)."},{"key":"e_1_3_2_1_4_1","first-page":"18353","article-title":"BAIL: Best-action imitation learning for batch deep reinforcement learning","volume":"33","author":"Chen Xinyue","year":"2020","unstructured":"Xinyue Chen, Zijian Zhou, Zheng Wang, Che Wang, Yanqiu Wu, and Keith Ross. 2020. BAIL: Best-action imitation learning for batch deep reinforcement learning. Advances in Neural Information Processing Systems 33 (2020), 18353--18363.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_5_1","volume-title":"Heuristic-guided reinforcement learning. NeurIPS","author":"Cheng Ching-An","year":"2021","unstructured":"Ching-An Cheng, Andrey Kolobov, and Adith Swaminathan. 2021. Heuristic-guided reinforcement learning. NeurIPS (2021)."},{"key":"e_1_3_2_1_6_1","first-page":"49","article-title":"Energy plus: energy simulation program","volume":"42","author":"Crawley Drury B","year":"2000","unstructured":"Drury B Crawley, Linda K Lawrie, Curtis O Pedersen, and Frederick C Winkelmann. 2000. Energy plus: energy simulation program. ASHRAE journal 42, 4 (2000), 49--56.","journal-title":"ASHRAE journal"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3486611.3492412"},{"key":"e_1_3_2_1_8_1","volume-title":"Benchmarking batch deep reinforcement learning algorithms. arXiv preprint arXiv:1910.01708","author":"Fujimoto Scott","year":"2019","unstructured":"Scott Fujimoto, Edoardo Conti, Mohammad Ghavamzadeh, and Joelle Pineau. 2019. Benchmarking batch deep reinforcement learning algorithms. arXiv preprint arXiv:1910.01708 (2019)."},{"key":"e_1_3_2_1_9_1","volume-title":"A minimalist approach to offline reinforcement learning. NeurIPS","author":"Fujimoto Scott","year":"2021","unstructured":"Scott Fujimoto and Shixiang Shane Gu. 2021. A minimalist approach to offline reinforcement learning. NeurIPS (2021)."},{"key":"e_1_3_2_1_10_1","volume-title":"International Conference on Machine Learning. PMLR","author":"Fujimoto Scott","year":"2019","unstructured":"Scott Fujimoto, David Meger, and Doina Precup. 2019. Off-policy deep reinforcement learning without exploration. In International Conference on Machine Learning. PMLR, 2052--2062."},{"key":"e_1_3_2_1_11_1","volume-title":"Energy-efficient thermal comfort control in smart buildings via deep reinforcement learning. arXiv preprint arXiv:1901.04693","author":"Gao Guanyu","year":"2019","unstructured":"Guanyu Gao, Jie Li, and Yonggang Wen. 2019. Energy-efficient thermal comfort control in smart buildings via deep reinforcement learning. arXiv preprint arXiv:1901.04693 (2019)."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/JIOT.2020.2992117"},{"key":"e_1_3_2_1_13_1","volume-title":"International Conference on Learning Representations.","author":"Guo Yijie","year":"2020","unstructured":"Yijie Guo, Shengyu Feng, Nicolas Le Roux, Ed Chi, Honglak Lee, and Minmin Chen. 2020. Batch reinforcement learning through continuation method. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_14_1","volume-title":"Gaussian error linear units (gelus). arXiv preprint arXiv:1606.08415","author":"Hendrycks Dan","year":"2016","unstructured":"Dan Hendrycks and Kevin Gimpel. 2016. Gaussian error linear units (gelus). arXiv preprint arXiv:1606.08415 (2016)."},{"key":"e_1_3_2_1_15_1","unstructured":"Geoffrey Hinton Oriol Vinyals Jeff Dean et al. 2015. Distilling the knowledge in a neural network. arXiv preprint arXiv:1503.02531 2 7 (2015)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10489-021-02239-z"},{"key":"e_1_3_2_1_17_1","volume-title":"Craig Ferguson, Agata Lapedriza, Noah Jones, Shixiang Gu, and Rosalind Picard.","author":"Jaques Natasha","year":"2019","unstructured":"Natasha Jaques, Asma Ghandeharioun, Judy Hanwen Shen, Craig Ferguson, Agata Lapedriza, Noah Jones, Shixiang Gu, and Rosalind Picard. 2019. Way off-policy batch deep reinforcement learning of implicit human preferences in dialog. arXiv preprint arXiv:1907.00456 (2019)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.2172\/90674"},{"key":"e_1_3_2_1_19_1","volume-title":"Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980","author":"Kingma Diederik P","year":"2014","unstructured":"Diederik P Kingma and Jimmy Ba. 2014. Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1038\/sj.jea.7500165"},{"key":"e_1_3_2_1_21_1","first-page":"1179","article-title":"Conservative q-learning for offline reinforcement learning","volume":"33","author":"Kumar Aviral","year":"2020","unstructured":"Aviral Kumar, Aurick Zhou, George Tucker, and Sergey Levine. 2020. Conservative q-learning for offline reinforcement learning. Advances in Neural Information Processing Systems 33 (2020), 1179--1191.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_22_1","volume-title":"Offline reinforcement learning: Tutorial, review, and perspectives on open problems. arXiv preprint arXiv:2005.01643","author":"Levine Sergey","year":"2020","unstructured":"Sergey Levine, Aviral Kumar, George Tucker, and Justin Fu. 2020. Offline reinforcement learning: Tutorial, review, and perspectives on open problems. arXiv preprint arXiv:2005.01643 (2020)."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.segy.2021.100044"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1016\/0360-1323(92)90029-O"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCST.2011.2124461"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1115\/DSCC2011-6078"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.enbuild.2014.03.057"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1016\/S0967-0661(98)00047-1"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.5555\/2600239.2600241"},{"key":"e_1_3_2_1_30_1","volume-title":"Accelerating online reinforcement learning with offline datasets. arXiv preprint arXiv:2006.09359","author":"Nair Ashvin","year":"2020","unstructured":"Ashvin Nair, Murtaza Dalal, Abhishek Gupta, and Sergey Levine. 2020. Accelerating online reinforcement learning with offline datasets. arXiv preprint arXiv:2006.09359 (2020)."},{"key":"e_1_3_2_1_31_1","volume-title":"Online energy management in commercial buildings using deep reinforcement learning. In 2019 IEEE SMARTCOMP","author":"Naug Aviek","unstructured":"Aviek Naug, Ibrahim Ahmed, and Gautam Biswas. 2019. Online energy management in commercial buildings using deep reinforcement learning. In 2019 IEEE SMARTCOMP. IEEE, 249--257."},{"key":"e_1_3_2_1_32_1","unstructured":"U.S. Department of Energy. 2011. Buildings energy data book."},{"key":"e_1_3_2_1_33_1","unstructured":"United States Department of Labor. 2021. OSHA Technical Manual (OTM) Section III: Chapter 2."},{"key":"e_1_3_2_1_34_1","unstructured":"Bjarne W Olesen and Gail S Brager. 2004. A better way to predict comfort: The new ASHRAE standard 55-2004. (2004)."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.enbuild.2016.09.044"},{"key":"e_1_3_2_1_36_1","volume-title":"Autonomous Building Control Using Offline Reinforcement Learning. In International Conference on P2P, Parallel, Grid, Cloud and Internet Computing. Springer, 246--255","author":"Schepers Jorren","year":"2021","unstructured":"Jorren Schepers, Reinout Eyckerman, Furkan Elmaz, Wim Casteels, Steven Latr\u00e9, and Peter Hellinckx. 2021. Autonomous Building Control Using Offline Reinforcement Learning. In International Conference on P2P, Parallel, Grid, Cloud and Internet Computing. Springer, 246--255."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.enconman.2019.111924"},{"key":"e_1_3_2_1_38_1","unstructured":"Ikechukwu Uchendu Ted Xiao Yao Lu Banghua Zhu Mengyuan Yan Jos\u00e9phine Simon Matthew Bennice Chuyuan Fu Cong Ma Jiantao Jiao et al. 2022. Jump-Start Reinforcement Learning. arXiv preprint arXiv:2204.02372 (2022)."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.5555\/3016100.3016191"},{"key":"e_1_3_2_1_40_1","first-page":"7768","article-title":"Critic regularized regression","volume":"33","author":"Wang Ziyu","year":"2020","unstructured":"Ziyu Wang, Alexander Novikov, Konrad Zolna, Josh S Merel, Jost Tobias Springenberg, Scott E Reed, Bobak Shahriari, Noah Siegel, Caglar Gulcehre, Nicolas Heess, et al. 2020. Critic regularized regression. Advances in Neural Information Processing Systems 33 (2020), 7768--7778.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3061639.3062224"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/TC.2015.2495244"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"crossref","unstructured":"Stephen Wilcox and William Marion. 2008. Users manual for TMY3 data sets. (2008).","DOI":"10.2172\/928611"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.compchemeng.2017.02.023"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3486611.3486644"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/3408308.3427617"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/TSG.2020.3011739"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/JIOT.2019.2957289"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"crossref","unstructured":"Kyungtae Yun Rogelio Luck Pedro J Mago and Heejin Cho. 2012. Building hourly thermal load prediction using an indexed ARX model. In Energy and Buildings.","DOI":"10.1016\/j.enbuild.2012.08.007"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1145\/3427773.3427865"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.enbuild.2019.07.029"},{"key":"e_1_3_2_1_52_1","volume-title":"2018 Building Performance Analysis Conference and SimBuild","volume":"3","author":"Zhang Zhiang","year":"2018","unstructured":"Zhiang Zhang, Adrian Chong, Yuqi Pan, Chenlu Zhang, Siliang Lu, and Khee Poh Lam. 2018. A deep reinforcement learning approach to using whole building energy model for hvac optimal control. In 2018 Building Performance Analysis Conference and SimBuild, Vol. 3. 22--23."}],"event":{"name":"BuildSys '22: The 9th ACM International Conference on Systems for Energy-Efficient Buildings, Cities, and Transportation","location":"Boston Massachusetts","acronym":"BuildSys '22","sponsor":["SIGEnergy ACM Special Interest Group on Energy Systems and Informatics"]},"container-title":["Proceedings of the 9th ACM International Conference on Systems for Energy-Efficient Buildings, Cities, and Transportation"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3563357.3564064","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3563357.3564064","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3563357.3564064","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T07:27:52Z","timestamp":1755847672000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3563357.3564064"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,11,9]]},"references-count":52,"alternative-id":["10.1145\/3563357.3564064","10.1145\/3563357"],"URL":"https:\/\/doi.org\/10.1145\/3563357.3564064","relation":{},"subject":[],"published":{"date-parts":[[2022,11,9]]},"assertion":[{"value":"2022-12-08","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}