{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T06:10:21Z","timestamp":1784268621205,"version":"3.55.0"},"reference-count":27,"publisher":"IEEE","license":[{"start":{"date-parts":[[2026,6,23]],"date-time":"2026-06-23T00:00:00Z","timestamp":1782172800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,6,23]],"date-time":"2026-06-23T00:00:00Z","timestamp":1782172800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100000780","name":"European Commission","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100000780","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026,6,23]]},"DOI":"10.1109\/med70602.2026.11598496","type":"proceedings-article","created":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T21:49:58Z","timestamp":1784238598000},"page":"239-244","source":"Crossref","is-referenced-by-count":0,"title":["A Multi-Stage RLHF Pipeline for LLM-based Smart Building Optimization"],"prefix":"10.1109","author":[{"given":"Ioannis","family":"Papaioannou","sequence":"first","affiliation":[{"name":"Informatics &#x0026; Telematics Institute (I.T.I.), Center for Research and Technology Hellas (CERTH),,Greece"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dimitrios G.","family":"Vamvakas","sequence":"additional","affiliation":[{"name":"Informatics &#x0026; Telematics Institute (I.T.I.), Center for Research and Technology Hellas (CERTH),,Greece"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Christos D.","family":"Tsaknakis","sequence":"additional","affiliation":[{"name":"Informatics &#x0026; Telematics Institute (I.T.I.), Center for Research and Technology Hellas (CERTH),,Greece"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Christos D.","family":"Korkas","sequence":"additional","affiliation":[{"name":"Informatics &#x0026; Telematics Institute (I.T.I.), Center for Research and Technology Hellas (CERTH),,Greece"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.3390\/s25175265"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1080\/00038628.2025.2488522"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1007\/s12273-025-1235-9"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1016\/j.rser.2025.115558"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.3390\/buildings15132303"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1016\/j.enbuild.2024.114278"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/TSG.2025.3589202"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/IEEECONF65522.2025.11137227"},{"key":"ref9","article-title":"Chatglm-rlhf: Practices of aligning large language models with human feedback","author":"Hou","year":"2024","journal-title":"arXiv preprint arXiv:2404.00934"},{"key":"ref10","article-title":"A comprehensive survey of 11 m alignment techniques: Rlhf, rlaif, ppo, dpo and more","author":"Wang","year":"2024","journal-title":"arXiv preprint arXiv:2407.16216"},{"key":"ref11","article-title":"Is dpo superior to ppo for 11 m alignment? a comprehensive study","author":"Xu","year":"2024","journal-title":"arXiv preprint arXiv:2404.10719"},{"key":"ref12","article-title":"Dpo meets ppo: Reinforced token optimization for rlhf","author":"Zhong","year":"2024","journal-title":"arXiv preprint arXiv:2404.18922"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1145\/3689031.3696075"},{"key":"ref14","article-title":"Rlaif vs. rlhf: Scaling reinforcement learning from human feedback with ai feedback","author":"Lee","year":"2023","journal-title":"arXiv preprint arXiv:2309.00267"},{"key":"ref15","article-title":"Explainable rewards in rlhf using llm-as-a-judge","author":"Shen","year":"2025"},{"key":"ref16","article-title":"Process reinforcement through implicit rewards","author":"Cui","year":"2025","journal-title":"arXiv preprint arXiv:2502.01456"},{"key":"ref17","article-title":"Safe rlhf: Safe reinforcement learning from human feedback","author":"Dai","year":"2023","journal-title":"arXiv preprint arXiv:2310.12773"},{"key":"ref18","article-title":"Una: unifying alignments of rlhf\/ppo, dpo and kto by a generalized implicit reward function","author":"Wang","year":"2024","journal-title":"arXiv preprint arXiv:2408.15339"},{"key":"ref19","article-title":"From system 1 to system 2: A survey of reasoning large language models","author":"Li","year":"2025","journal-title":"arXiv preprint arXiv:2502.17419"},{"key":"ref20","article-title":"Teaching large language models to reason with reinforcement learning","author":"Havrilla","year":"2024","journal-title":"arXiv preprint arXiv:2403.04642"},{"key":"ref21","article-title":"Learning to generate better than your 11 m","author":"Chang","year":"2023","journal-title":"arXiv preprint arXiv:2306.11816"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-acl.464"},{"key":"ref23","article-title":"Logic-rl: Unleashing llm reasoning with rule-based reinforcement learning","author":"Xie","year":"2025","journal-title":"arXiv preprint arXiv:2502.14768"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ICCUBEA.2018.8697857"},{"key":"ref25","article-title":"Reinforcement learning enhanced 11 ms: A survey","author":"Wang","year":"2024","journal-title":"arXiv preprint arXiv:2412.10400"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/tmc.2024.3522130"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1145\/3360322.3360998"}],"event":{"name":"2026 34th Mediterranean Conference on Control and Automation (MED)","location":"Ancona, Italy","start":{"date-parts":[[2026,6,23]]},"end":{"date-parts":[[2026,6,26]]}},"container-title":["2026 34th Mediterranean Conference on Control and Automation (MED)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11597968\/11597984\/11598496.pdf?arnumber=11598496","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T05:50:10Z","timestamp":1784267410000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11598496\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,23]]},"references-count":27,"URL":"https:\/\/doi.org\/10.1109\/med70602.2026.11598496","relation":{},"subject":[],"published":{"date-parts":[[2026,6,23]]}}}