{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,29]],"date-time":"2026-07-29T21:56:25Z","timestamp":1785362185232,"version":"3.55.0"},"reference-count":30,"publisher":"Springer Science and Business Media LLC","issue":"12","license":[{"start":{"date-parts":[[2023,9,22]],"date-time":"2023-09-22T00:00:00Z","timestamp":1695340800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,9,22]],"date-time":"2023-09-22T00:00:00Z","timestamp":1695340800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"DOI":"10.1007\/s11042-023-16906-5","type":"journal-article","created":{"date-parts":[[2023,9,22]],"date-time":"2023-09-22T07:01:59Z","timestamp":1695366119000},"page":"37073-37087","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["ORLEP: an efficient offline reinforcement learning evaluation platform"],"prefix":"10.1007","volume":"83","author":[{"given":"Keming","family":"Mao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8783-3798","authenticated-orcid":false,"given":"Chen","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jinkai","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yiyang","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,9,22]]},"reference":[{"key":"16906_CR1","doi-asserted-by":"crossref","unstructured":"Alshuqayran N, Ali N, Evans R (2016) A systematic mapping study in microservice architecture. In: 2016 IEEE 9th International Conference on Service-Oriented Computing and Applications (SOCA). IEEE, pp 44\u201351","DOI":"10.1109\/SOCA.2016.15"},{"issue":"5","key":"16906_CR2","first-page":"154","volume":"25","author":"C Burch","year":"2010","unstructured":"Burch C (2010) Django, a web framework using python: Tutorial presentation. In: J Comput Sci Coll 25(5):154\u2013155","journal-title":"In: J Comput Sci Coll"},{"key":"16906_CR3","unstructured":"Cabi S, et\u00a0al (2019) A framework for data-driven robotics. In: arXiv:1909.12200"},{"issue":"1","key":"16906_CR4","first-page":"5867","volume":"22","author":"C D\u2019Eramo","year":"2021","unstructured":"D\u2019Eramo C et al (2021) Mushroomrl: Simplifying reinforcement learning research. In: J Mach Learn Res 22(1):5867\u20135871","journal-title":"In: J Mach Learn Res"},{"key":"16906_CR5","unstructured":"Denoyer L, et\u00a0al (2021) Salina: Sequential learning of agents. In: arXiv:2110.07910"},{"key":"16906_CR6","unstructured":"Fu J, et\u00a0al (2020) D4rl: Datasets for deep data-driven reinforcement learning. In: arXiv:2004.07219"},{"key":"16906_CR7","unstructured":"Fujimoto S, Meger D, Precup D (2019) Off-policy deep reinforcement learning without exploration. In: International conference on machine learning. PMLR, pp 2052\u20132062"},{"key":"16906_CR8","doi-asserted-by":"publisher","first-page":"107","DOI":"10.1016\/j.buildenv.2018.06.016","volume":"142","author":"AN Gade","year":"2018","unstructured":"Gade AN et al (2018) REDIS: A value-based decision support tool for renovation of building portfolios. Building and environment 142:107\u2013118","journal-title":"Building and environment"},{"issue":"4","key":"16906_CR9","first-page":"487","volume":"34","author":"J Henderson","year":"2008","unstructured":"Henderson J, Lemon O, Georgila K (2008) Hybrid reinforcement\/supervised learning of dialogue policies from fixed data sets. In: Comput Linguist 34(4):487\u2013511","journal-title":"In: Comput Linguist"},{"key":"16906_CR10","doi-asserted-by":"crossref","unstructured":"Ionescu VM (2015) The analysis of the performance of RabbitMQ and ActiveMQ. In: 2015 14th RoEduNet International Conference-Networking in Education and Research (RoEduNet NER). IEEE, pp 132\u2013137","DOI":"10.1109\/RoEduNet.2015.7311982"},{"key":"16906_CR11","unstructured":"Jaques N et\u00a0al (2019) Way off-policy batch deep reinforcement learning of implicit human preferences in dialog. In: arXiv:1907.00456"},{"key":"16906_CR12","unstructured":"Kuhnle A, Schaarschmidt M, Fricke K (2017) Tensorforce: a tensorflow library for applied reinforcement learning. In: Web p 9"},{"key":"16906_CR13","first-page":"1179","volume":"33","author":"A Kumar","year":"2020","unstructured":"Kumar A et al (2020) Conservative q-learning for offline reinforcement learning. Adv Neural Inf Process Syst 33:1179\u20131191","journal-title":"Adv Neural Inf Process Syst"},{"key":"16906_CR14","unstructured":"Levine S, et\u00a0al (2020) Offline reinforcement learning:Tutorial, review, and perspectives on open problems. In: arXiv:2005.01643"},{"key":"16906_CR15","doi-asserted-by":"crossref","unstructured":"Li L, et\u00a0al (2010) A contextual-bandit approach to personalized news article recommendation. In: Proceedings of the 19th international conference on World wide web, pp 661\u2013670","DOI":"10.1145\/1772690.1772758"},{"key":"16906_CR16","unstructured":"Linzecong. LPOJ usage and development Document. https:\/\/docs.lpoj.cn\/.2023.5.20"},{"key":"16906_CR17","doi-asserted-by":"crossref","unstructured":"Nandy A, et\u00a0al (2018) Reinforcement learning with keras, tensorflow, and chainerrl. In: Reinforcement learning: With open ai, tensorflow and keras using python, pp 129\u2013153","DOI":"10.1007\/978-1-4842-3285-9_5"},{"key":"16906_CR18","first-page":"27730","volume":"35","author":"L Ouyang","year":"2022","unstructured":"Ouyang L et al (2022) Training language models to follow instructions with human feedback. Adv Neural Inf Process Syst 35:27730\u201327744","journal-title":"Adv Neural Inf Process Syst"},{"issue":"3","key":"16906_CR19","first-page":"1","volume":"7","author":"O Pietquin","year":"2011","unstructured":"Pietquin O et al (2011) Sample-efficient batch reinforcement learning for dialogue management optimization. In: ACM Trans Audio Speech Lang Process (TSLP) 7(3):1\u201321s","journal-title":"In: ACM Trans Audio Speech Lang Process (TSLP)"},{"key":"16906_CR20","first-page":"24753","volume":"35","author":"RJ Qin","year":"2022","unstructured":"Qin RJ et al (2022) NeoRL: A near real-world benchmark for offline reinforcement learning. Adv Neural Inf Process Syst 35:24753\u201324765","journal-title":"Adv Neural Inf Process Syst"},{"issue":"1","key":"16906_CR21","first-page":"12348","volume":"22","author":"A Raffin","year":"2021","unstructured":"Raffin A et al (2021) Stable-baselines3: Reliable reinforcement learning implementations. In: J Mach Learn Res 22(1):12348\u201312355","journal-title":"In: J Mach Learn Res"},{"issue":"1","key":"16906_CR22","first-page":"14205","volume":"23","author":"T Seno","year":"2022","unstructured":"Seno T, Imai M (2022) d3rlpy: An offline deep reinforcement learning library. In: J Mach Learn Res 23(1):14205\u201314224","journal-title":"In: J Mach Learn Res"},{"key":"16906_CR23","volume-title":"Beginning MySQL","author":"R Sheldon","year":"2005","unstructured":"Sheldon R, Moes G (2005) Beginning MySQL. John Wiley & Sons"},{"issue":"7676","key":"16906_CR24","first-page":"354","volume":"550","author":"D Silver","year":"2017","unstructured":"Silver D et al (2017) Mastering the game of go without human knowledge. In: Nature 550(7676):354\u2013359","journal-title":"In: Nature"},{"key":"16906_CR25","unstructured":"Strehl A, et\u00a0al (2010) Learning from logged implicit exploration data. In: Adv Neural Inf Process Syst 23"},{"key":"16906_CR26","doi-asserted-by":"crossref","unstructured":"Thomas P, et\u00a0al (2017) Predictive off-policy policy evaluation for nonstationary decision problems, with applications to digital marketing. In: Proceedings of the AAAI Conference on Artificial Intelligence. Vol. 31(2), pp 4740\u20134745","DOI":"10.1609\/aaai.v31i2.19104"},{"issue":"7782","key":"16906_CR27","first-page":"350","volume":"575","author":"O Vinyals","year":"2019","unstructured":"Vinyals O et al (2019) Grandmaster level in Star-Craft II using multi-agent reinforcement learning. In: Nature 575(7782):350\u2013354","journal-title":"In: Nature"},{"key":"16906_CR28","unstructured":"Weng J, et\u00a0al (2021) Tianshou: A highly modularized deep reinforcement learning library. In: arXiv:2107.14171"},{"issue":"3","key":"16906_CR29","first-page":"729","volume":"12","author":"MA Wiering","year":"2012","unstructured":"Wiering MA, Van Otterlo M (2012) Reinforcement learning. In: Adapt Learn Optim 12(3):729","journal-title":"In: Adapt Learn Optim"},{"key":"16906_CR30","unstructured":"You E (2022) Vue.js Progressive JavaScript Framework. https:\/\/v2.cn.vuejs.org\/.2023.5.20"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-16906-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-023-16906-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-16906-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,4,2]],"date-time":"2024-04-02T13:17:45Z","timestamp":1712063865000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-023-16906-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,9,22]]},"references-count":30,"journal-issue":{"issue":"12","published-online":{"date-parts":[[2024,4]]}},"alternative-id":["16906"],"URL":"https:\/\/doi.org\/10.1007\/s11042-023-16906-5","relation":{},"ISSN":["1573-7721"],"issn-type":[{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,9,22]]},"assertion":[{"value":"13 July 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 May 2023","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 August 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 September 2023","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflicts of interest"}}]}}