{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,6]],"date-time":"2026-07-06T20:21:02Z","timestamp":1783369262150,"version":"3.54.6"},"reference-count":14,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"Ministry of Economic Development of Russian Federation","award":["139-10-2025-034"],"award-info":[{"award-number":["139-10-2025-034"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Control Syst. Lett."],"published-print":{"date-parts":[[2026]]},"DOI":"10.1109\/lcsys.2026.3707170","type":"journal-article","created":{"date-parts":[[2026,6,25]],"date-time":"2026-06-25T19:46:57Z","timestamp":1782416817000},"page":"1321-1326","source":"Crossref","is-referenced-by-count":0,"title":["Benchmarking Reinforcement Learning via Stochastic Converse Optimality: Generating Systems With Known Optimal Policies"],"prefix":"10.1109","volume":"10","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-9965-8209","authenticated-orcid":false,"given":"Sinan","family":"Ibrahim","sequence":"first","affiliation":[{"name":"Skolkovo Institute of Science and Technology, Moscow, Russia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Gr\u00e9goire","family":"Ouerdane","sequence":"additional","affiliation":[{"name":"Skolkovo Institute of Science and Technology, Moscow, Russia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hadi","family":"Salloum","sequence":"additional","affiliation":[{"name":"Research Center of the Artificial Intelligence Institute, Innopolis University, Innopolis, Russia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1914-0244","authenticated-orcid":false,"given":"Henni","family":"Ouerdane","sequence":"additional","affiliation":[{"name":"Skolkovo Institute of Science and Technology, Moscow, Russia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3398-1226","authenticated-orcid":false,"given":"Stefan","family":"Streif","sequence":"additional","affiliation":[{"name":"Technische Universit&#x00E4;t Chemnitz, Chemnitz, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6184-3293","authenticated-orcid":false,"given":"Pavel","family":"Osinenko","sequence":"additional","affiliation":[{"name":"Skolkovo Institute of Science and Technology, Moscow, Russia"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1126\/science.153.3731.34"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11694"},{"key":"ref3","first-page":"29304","article-title":"Deep reinforcement learning at the edge of the statistical precipice","volume-title":"Proc. Adv. Neural Inf. Process. Syst. (NeurIPS)","volume":"34","author":"Agarwal"},{"key":"ref4","first-page":"2048","article-title":"Leveraging procedural generation to benchmark reinforcement learning","volume-title":"Proc. 37th Int. Conf. Mach. Learn.","author":"Cobbe"},{"key":"ref5","article-title":"D4RL: Datasets for deep data-driven reinforcement learning","author":"Fu","year":"2020","journal-title":"arXiv:2004.07219"},{"key":"ref6","article-title":"Open RL benchmark: Comprehensive tracked experiments for reinforcement learning","author":"Huang","year":"2024","journal-title":"arXiv:2402.03046"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/MCS.2012.2214134"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2009.03.008"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1016\/j.automatica.2011.03.005"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2019.2941425"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/LCSYS.2025.3573943"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1108\/EC-08-2024-0709"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2025.3558898"},{"issue":"268","key":"ref14","first-page":"1","article-title":"Stable-baselines3: Reliable reinforcement learning implementations","volume":"22","author":"Raffin","year":"2021","journal-title":"J. Mach. Learn. Res."}],"container-title":["IEEE Control Systems Letters"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/7782633\/11370312\/11579323.pdf?arnumber=11579323","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,6]],"date-time":"2026-07-06T19:55:42Z","timestamp":1783367742000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11579323\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"references-count":14,"URL":"https:\/\/doi.org\/10.1109\/lcsys.2026.3707170","relation":{},"ISSN":["2475-1456"],"issn-type":[{"value":"2475-1456","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]}}}