{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,1]],"date-time":"2025-10-01T15:31:05Z","timestamp":1759332665920,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":28,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,1]],"date-time":"2024-10-01T00:00:00Z","timestamp":1727740800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/100000001","name":"NSF (National Science Foundation)","doi-asserted-by":"publisher","award":["NS-2148183 and CNS-2106268"],"award-info":[{"award-number":["NS-2148183 and CNS-2106268"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000181","name":"Air Force Office of Scientific Research","doi-asserted-by":"publisher","award":["FA8702-15-D-0001"],"award-info":[{"award-number":["FA8702-15-D-0001"]}],"id":[{"id":"10.13039\/100000181","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,14]]},"DOI":"10.1145\/3641512.3686383","type":"proceedings-article","created":{"date-parts":[[2024,10,1]],"date-time":"2024-10-01T21:11:22Z","timestamp":1727817082000},"page":"301-310","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Intervention-Assisted Online Deep Reinforcement Learning for Stochastic Queuing Network Optimization"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-6814-1959","authenticated-orcid":false,"given":"Jerrod","family":"Wigmore","sequence":"first","affiliation":[{"name":"LIDS, MIT, Cambridge, Massachusetts, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-2776-0739","authenticated-orcid":false,"given":"Brooke","family":"Shrader","sequence":"additional","affiliation":[{"name":"MIT Lincoln Laboratory, Lexington, Massachusetts, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8238-8130","authenticated-orcid":false,"given":"Eytan","family":"Modiano","sequence":"additional","affiliation":[{"name":"LIDS, MIT, Cambridge, Massachusetts, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,10]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Du\u00e9\u00f1ez-Guzm\u00e1n","author":"Chow Yinlam","year":"2019","unstructured":"Yinlam Chow, Ofir Nachum, Aleksandra Faust, M. Ghavamzadeh, and Edgar A. Du\u00e9\u00f1ez-Guzm\u00e1n. 2019. Lyapunov-Based Safe Policy Optimization for Continuous Control. ArXiv (Jan. 2019)."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1287\/stsy.2021.0081"},{"key":"e_1_3_2_1_3_1","unstructured":"Jesse Farebrother Marios C. Machado and Michael Bowling. 2020. Generalization and Regularization in DQN. arXiv:1810.00123"},{"key":"e_1_3_2_1_4_1","first-page":"1471","article-title":"Variance Reduction Techniques for Gradient Estimates in Reinforcement Learning","author":"Greensmith Evan","year":"2004","unstructured":"Evan Greensmith, Peter L. Bartlett, and Jonathan Baxter. 2004. Variance Reduction Techniques for Gradient Estimates in Reinforcement Learning. Journal of Machine Learning Research 5, Nov (2004), 1471--1530.","journal-title":"Journal of Machine Learning Research 5"},{"key":"e_1_3_2_1_5_1","volume-title":"IJCNN International Joint Conference on Neural Networks","volume":"4","author":"Haley P.J.","unstructured":"P.J. Haley and D. Soloway. 1992. Extrapolation Limitations of Multilayer Feedforward Neural Networks. In [Proceedings 1992] IJCNN International Joint Conference on Neural Networks, Vol. 4. 25-30 vol.4."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1287\/opre.9.3.383"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2021\/614"},{"key":"e_1_3_2_1_8_1","volume-title":"Average-Reward Reinforcement Learning with Trust Region Methods. arXiv","author":"Ma Xiaoteng","year":"2021","unstructured":"Xiaoteng Ma, Xiaohang Tang, Li Xia, Jun Yang, and Qianchuan Zhao. 2021. Average-Reward Reinforcement Learning with Trust Region Methods. arXiv (2021)."},{"key":"e_1_3_2_1_9_1","volume-title":"Tweedie","author":"Meyn Sean","year":"2009","unstructured":"Sean Meyn and Richard L. Tweedie. 2009. Markov Chains and Stochastic Stability (2 ed.). Cambridge University Press, Cambridge."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.5555\/1941130"},{"key":"e_1_3_2_1_11_1","volume-title":"Hanna","author":"Pavse Brahma S.","year":"2023","unstructured":"Brahma S. Pavse, Yudong Chen, Qiaomin Xie, and Josiah P. Hanna. 2023. Tackling Unbounded State Spaces in Continuing Task Reinforcement Learning. arXiv:2306.01896"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i1.16123"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2021.3126658"},{"key":"e_1_3_2_1_14_1","unstructured":"John Schulman Sergey Levine Philipp Moritz Michael I. Jordan and Pieter Abbeel. 2017. Trust Region Policy Optimization. arXiv:1502.05477"},{"key":"e_1_3_2_1_15_1","unstructured":"John Schulman Philipp Moritz Sergey Levine Michael Jordan and Pieter Abbeel. 2018. High-Dimensional Continuous Control Using Generalized Advantage Estimation. arXiv:1506.02438"},{"key":"e_1_3_2_1_16_1","unstructured":"John Schulman Filip Wolski Prafulla Dhariwal Alec Radford and Oleg Klimov. 2017. Proximal Policy Optimization Algorithms. arXiv:1707.06347"},{"key":"e_1_3_2_1_17_1","unstructured":"Adam Stooke Joshua Achiam and Pieter Abbeel. 2020. Responsive Safety in Reinforcement Learning by PID Lagrangian Methods. arXiv:2007.03964arXiv:2007.03964 [cs math] 10.48550\/"},{"key":"e_1_3_2_1_18_1","volume-title":"Barto","author":"Sutton Richard S.","year":"2018","unstructured":"Richard S. Sutton and Andrew G. Barto. 2018. Reinforcement Learning: An Introduction (second edition ed.). The MIT Press, Cambridge, Massachusetts."},{"volume-title":"Advances in Neural Information Processing Systems","author":"Sutton Richard S","key":"e_1_3_2_1_19_1","unstructured":"Richard S Sutton, David McAllester, Satinder Singh, and Yishay Mansour. 1999. Policy Gradient Methods for Reinforcement Learning with Function Approximation. In Advances in Neural Information Processing Systems, Vol. 12. MIT Press."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/9.182479"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/IROS45743.2020.9341617"},{"key":"e_1_3_2_1_22_1","unstructured":"Nolan Wagener Byron Boots and Ching-An Cheng. 2021. Safe Reinforcement Learning Using Advantage-Based Intervention. arXiv:2106.09110"},{"key":"e_1_3_2_1_23_1","volume-title":"Proceedings of The 2nd Conference on Robot Learning. PMLR, 410--421","author":"Wang Fan","year":"2018","unstructured":"Fan Wang, Bo Zhou, Ke Chen, Tingxiang Fan, Xi Zhang, Jiangyong Li, Hao Tian, and Jia Pan. 2018. Intervention Aided Reinforcement Learning for Safe and Practical Policy Optimization in Navigation. In Proceedings of The 2nd Conference on Robot Learning. PMLR, 410--421."},{"key":"e_1_3_2_1_25_1","unstructured":"Keyulu Xu Mozhi Zhang Jingling Li Simon S. Du Ken-ichi Kawarabayashi and Stefanie Jegelka. 2021. How Neural Networks Extrapolate: From Feedforward to Graph Neural Networks. arXiv:2009.11848"},{"key":"e_1_3_2_1_26_1","volume-title":"Ramadge","author":"Yang Tsung-Yen","year":"2020","unstructured":"Tsung-Yen Yang, Justinian Rosca, Karthik Narasimhan, and Peter J. Ramadge. 2020. Projection-Based Constrained Policy Optimization, arXiv:2010.03152arXiv:2010.03152 [cs] 10.48550\/"},{"key":"e_1_3_2_1_27_1","unstructured":"Amy Zhang Nicolas Ballas and Joelle Pineau. 2018. A Dissection of Overfitting and Generalization in Continuous Reinforcement Learning. arXiv:1806.07937"},{"key":"e_1_3_2_1_28_1","unstructured":"Chiyuan Zhang Oriol Vinyals Remi Munos and Samy Bengio. 2018. A Study on Overfitting in Deep Reinforcement Learning. arXiv:1804.06893"},{"volume-title":"Proceedings of the 38th International Conference on Machine Learning. PMLR, 12535--12545","author":"Zhang Yiming","key":"e_1_3_2_1_29_1","unstructured":"Yiming Zhang and Keith W. Ross. 2021. On-Policy Deep Reinforcement Learning for the Average-Reward Criterion. In Proceedings of the 38th International Conference on Machine Learning. PMLR, 12535--12545."}],"event":{"name":"MobiHoc '24: Twenty-fifth International Symposium on Theory, Algorithmic Foundations, and Protocol Design for Mobile Networks and Mobile Computing","sponsor":["SIGMOBILE ACM Special Interest Group on Mobility of Systems, Users, Data and Computing"],"location":"Athens Greece","acronym":"MobiHoc '24"},"container-title":["Proceedings of the Twenty-fifth International Symposium on Theory, Algorithmic Foundations, and Protocol Design for Mobile Networks and Mobile Computing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3641512.3686383","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3641512.3686383","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3641512.3686383","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T00:03:23Z","timestamp":1750291403000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3641512.3686383"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10]]},"references-count":28,"alternative-id":["10.1145\/3641512.3686383","10.1145\/3641512"],"URL":"https:\/\/doi.org\/10.1145\/3641512.3686383","relation":{},"subject":[],"published":{"date-parts":[[2024,10]]},"assertion":[{"value":"2024-10-01","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}