{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,12]],"date-time":"2025-12-12T06:21:23Z","timestamp":1765520483746,"version":"3.48.0"},"reference-count":45,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,10,19]]},"DOI":"10.1109\/iros60139.2025.11247623","type":"proceedings-article","created":{"date-parts":[[2025,11,27]],"date-time":"2025-11-27T18:54:45Z","timestamp":1764269685000},"page":"4174-4181","source":"Crossref","is-referenced-by-count":0,"title":["Closing the intent-to-behavior gap via Fulfillment Priority Logic"],"prefix":"10.1109","author":[{"given":"B. El","family":"Mabsout","sequence":"first","affiliation":[{"name":"Boston University,Department of Computer Science,Boston,MA,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"A.","family":"Abdelgawad","sequence":"additional","affiliation":[{"name":"Boston University,Systems Engineering Division,Boston,MA,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"R.","family":"Mancuso","sequence":"additional","affiliation":[{"name":"Boston University,Department of Computer Science,Boston,MA,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2024.3504735"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i5.25733"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1016\/j.artint.2022.103829"},{"key":"ref4","first-page":"1988","article-title":"A brief guide to multi-objective reinforcement learning and planning","volume-title":"Proc. AAMAS","author":"Hayes"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-89378-3_37"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2017.8206234"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1126\/scirobotics.abc5986"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1007\/s10462-021-10108-x"},{"article-title":"Eureka: Human-level reward design via coding large language models","volume-title":"Proc. ICLR","author":"Ma","key":"ref9"},{"article-title":"Language to rewards for robotic skill synthesis","year":"2023","author":"Yu","key":"ref10"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1613\/jair.3987"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1007\/s10458-023-09604-x"},{"volume-title":"Reinforcement learning: An introduction","year":"2018","author":"Sutton","key":"ref13"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1016\/j.artint.2021.103535"},{"article-title":"The reward hypothesis is false","volume-title":"NeurIPS ML Safety Workshop","author":"Skalse","key":"ref15"},{"article-title":"Settling the reward hypothesis","volume-title":"Proc. ICML","author":"Bowling","key":"ref16"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN.2014.6889637"},{"article-title":"Challenges of real-world reinforcement learning","year":"2019","author":"Mankowitz","key":"ref18"},{"key":"ref19","article-title":"Learning to utilize shaping rewards: a new approach of reward shaping","author":"Hu","year":"2020","journal-title":"Adv. NeurIPS"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-021-04301-9"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1080\/0952813X.2017.1292319"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.2166\/hydro.2013.169"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1016\/j.rser.2019.03.019"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1145\/3466618"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1016\/S0165-0114(96)00352-1"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1016\/j.engappai.2020.103915"},{"article-title":"Prediction-guided multi-objective reinforcement learning for continuous robot control","volume-title":"Proc. ICML","author":"Xu","key":"ref27"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2024\/544"},{"key":"ref29","first-page":"2003","article-title":"Sample-efficient multi-objective learning via generalized policy improvement prioritization","volume-title":"Proc. AAMAS","author":"Alegre"},{"article-title":"A toolkit for reliable benchmarking and research in multi-objective reinforcement learning","volume-title":"Proceedings of the 37th (NeurIPS 2023)","author":"Felten","key":"ref30"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/TRO.2009.2030225"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/TRO.2011.2172150"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/CDC.2016.7799279"},{"key":"ref34","article-title":"A composable specification language for reinforcement learning tasks","volume":"32","author":"Jothimurugan","year":"2019","journal-title":"Adv. NeurIPS"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/LCSYS.2020.3047362"},{"key":"ref36","first-page":"113","article-title":"A modified average reward reinforcement learning based on fuzzy reward function","volume-title":"Proc. Int. MultiConf. Eng. Comput. Sci","volume":"1","author":"Zhai"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1016\/j.seta.2021.101665"},{"article-title":"Randomized ensembled double q-learning: Learning fast without a model","year":"2021","author":"Chen","key":"ref38"},{"key":"ref39","first-page":"5556","article-title":"Controlling overestimation bias with truncated mixture of continuous distributional quantile critics","volume-title":"Proc. ICML","author":"Kuznetsov"},{"article-title":"Crossq: Batch normalization in deep reinforcement learning for greater sample efficiency and simplicity","year":"2019","author":"Bhatt","key":"ref40"},{"key":"ref41","doi-asserted-by":"crossref","first-page":"175","DOI":"10.1007\/978-94-017-0399-4_3","article-title":"Chapter iii - the power means","volume-title":"Handbook of Means and Their Inequalities","author":"Bullen","year":"2003"},{"issue":"2","key":"ref42","first-page":"192","article-title":"Similarity measures for fuzzy sets","volume":"8","author":"Beg","year":"2009","journal-title":"Appl. Comput. Math."},{"article-title":"Gymnasium: A standard interface for reinforcement learning environments","year":"2024","author":"Towers","key":"ref43"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.32657\/10356\/90191"},{"key":"ref45","first-page":"1861","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","volume-title":"Proc. ICML","author":"Haarnoja"}],"event":{"name":"2025 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)","start":{"date-parts":[[2025,10,19]]},"location":"Hangzhou, China","end":{"date-parts":[[2025,10,25]]}},"container-title":["2025 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11245651\/11245652\/11247623.pdf?arnumber=11247623","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,12]],"date-time":"2025-12-12T06:17:44Z","timestamp":1765520264000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11247623\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,19]]},"references-count":45,"URL":"https:\/\/doi.org\/10.1109\/iros60139.2025.11247623","relation":{},"subject":[],"published":{"date-parts":[[2025,10,19]]}}}