{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,31]],"date-time":"2025-12-31T12:17:04Z","timestamp":1767183424549,"version":"3.28.0"},"reference-count":26,"publisher":"IEEE","license":[{"start":{"date-parts":[[2021,4,19]],"date-time":"2021-04-19T00:00:00Z","timestamp":1618790400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2021,4,19]],"date-time":"2021-04-19T00:00:00Z","timestamp":1618790400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2021,4,19]],"date-time":"2021-04-19T00:00:00Z","timestamp":1618790400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021,4,19]]},"DOI":"10.1109\/drcn51631.2021.9477375","type":"proceedings-article","created":{"date-parts":[[2021,7,12]],"date-time":"2021-07-12T21:47:52Z","timestamp":1626126472000},"page":"1-6","source":"Crossref","is-referenced-by-count":6,"title":["Constrained Policy Optimization for Load Balancing"],"prefix":"10.1109","author":[{"given":"Ahmed Yassine","family":"Kamri","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Pham Tran Anh","family":"Quang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Nicolas","family":"Huin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jeremie","family":"Leguay","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM.2018.8485853"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/TMC.2018.2868670"},{"key":"ref12","article-title":"Reward constrained policy optimization","volume":"abs 1805 11074","author":"tessler","year":"2018","journal-title":"CoRR"},{"key":"ref13","article-title":"OpenAI Gym","volume":"abs 1606 1540","author":"brockman","year":"2016","journal-title":"CoRR"},{"key":"ref14","article-title":"Continuous control wuth deep reinforcement learning","author":"lillicrap","year":"2016","journal-title":"ICLRE"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/MNET.2017.1700200"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1145\/3152434.3152441"},{"key":"ref17","first-page":"369","article-title":"Generalization in reinforcement learning: Safely approximating the value function","author":"boyan","year":"1995","journal-title":"Advances in neural information processing systems"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/SCC.2016.12"},{"article-title":"A deep-reinforcement learning approach for software-defined networking routing optimization","year":"2017","author":"stampa","key":"ref19"},{"key":"ref4","doi-asserted-by":"crossref","first-page":"14","DOI":"10.1109\/JPROC.2014.2371999","article-title":"Software-defined networking: A comprehensive survey","volume":"103","author":"kreutz","year":"2014","journal-title":"Proceedings of the IEEE"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1145\/2592798.2592803"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/GLOCOM.2016.7841861"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1145\/2716281.2836091"},{"journal-title":"Communication Nets Stochastic Message Flow and Delay","year":"2007","author":"kleinrock","key":"ref8"},{"key":"ref7","article-title":"Mathematical models of the delay constrained routing problem","volume":"1","author":"ben-ameur","year":"2006","journal-title":"Algorithmic Operations Research"},{"article-title":"RFC 2991: Multipath issues in unicast and multicast next-hop selection","year":"2000","author":"thaler","key":"ref2"},{"key":"ref9","article-title":"G&#x00E9;n&#x00E9;ration de colonnes pour le probl&#x00E8;me de routage &#x00E0; d&#x00E9;lai variable","author":"huin","year":"2020","journal-title":"AlgoTel"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/COMST.2008.4483669"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-19945-6_4"},{"key":"ref22","article-title":"Proximal policy optimization algorithms","volume":"abs 1707 6347","author":"schulman","year":"2017","journal-title":"CoRR"},{"article-title":"Gallager., r. 1992. data networks","year":"0","author":"bertsekas","key":"ref21"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/TNSM.2019.2947905"},{"key":"ref23","article-title":"Deep reinforcement learning that matters","volume":"abs 1709 6560","author":"henderson","year":"2017","journal-title":"CoRR"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33013387"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1155\/2013\/425740"}],"event":{"name":"2021 17th International Conference on the Design of Reliable Communication Networks (DRCN)","start":{"date-parts":[[2021,4,19]]},"location":"Milano, Italy","end":{"date-parts":[[2021,4,22]]}},"container-title":["2021 17th International Conference on the Design of Reliable Communication Networks (DRCN)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9477293\/9477310\/09477375.pdf?arnumber=9477375","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,5,10]],"date-time":"2022-05-10T15:43:20Z","timestamp":1652197400000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9477375\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,4,19]]},"references-count":26,"URL":"https:\/\/doi.org\/10.1109\/drcn51631.2021.9477375","relation":{},"subject":[],"published":{"date-parts":[[2021,4,19]]}}}