{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T17:59:57Z","timestamp":1775066397101,"version":"3.50.1"},"reference-count":16,"publisher":"IEEE","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2014,12]]},"DOI":"10.1109\/cdc.2014.7040156","type":"proceedings-article","created":{"date-parts":[[2015,2,17]],"date-time":"2015-02-17T19:53:59Z","timestamp":1424202839000},"page":"4911-4916","source":"Crossref","is-referenced-by-count":46,"title":["Infinite time horizon maximum causal entropy inverse reinforcement learning"],"prefix":"10.1109","author":[{"given":"Michael","family":"Bloem","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Nicholas","family":"Bambos","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref10","article-title":"Convergence of Q-learning: A simple proof","author":"melo","year":"2007"},{"key":"ref11","article-title":"PyMDPToolbox: Markov decision process (MDP) toolbox for Python","author":"cordwell","year":"2013"},{"key":"ref12","article-title":"CVXPY: A Python package for modeling convex optimization problems","author":"rubira","year":"2013"},{"key":"ref13","article-title":"CVXOPT: Python software for convex optimization","author":"andersen","year":"2013"},{"key":"ref14","article-title":"Feedback control of the National Airspace System","author":"ny","year":"2010","journal-title":"AIAA Journal of Guidance Control and Dynamics"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2013.2260745"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.2514\/6.2014-2026"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TIT.2012.2234824"},{"key":"ref3","article-title":"Modeling interaction via the principle of maximum causal entropy","author":"ziebart","year":"2010","journal-title":"Proc of International Conference on Machine Learning"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1145\/1015330.1015430"},{"key":"ref5","volume":"1","author":"bertsekas","year":"2005","journal-title":"Dynamic Programming and Optimal Control"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1016\/j.jet.2004.12.006"},{"key":"ref7","article-title":"Directed information for channels with feedback","author":"kramer","year":"1998"},{"key":"ref2","article-title":"Modeling purposeful adaptive behavior with the principle of maximum causal entropy","author":"ziebart","year":"2010"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/CEC.2012.6256507"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511804441"}],"event":{"name":"2014 IEEE 53rd Annual Conference on Decision and Control (CDC)","location":"Los Angeles, CA, USA","start":{"date-parts":[[2014,12,15]]},"end":{"date-parts":[[2014,12,17]]}},"container-title":["53rd IEEE Conference on Decision and Control"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/7027307\/7039338\/07040156.pdf?arnumber=7040156","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2017,3,24]],"date-time":"2017-03-24T02:06:59Z","timestamp":1490321219000},"score":1,"resource":{"primary":{"URL":"http:\/\/ieeexplore.ieee.org\/document\/7040156\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2014,12]]},"references-count":16,"URL":"https:\/\/doi.org\/10.1109\/cdc.2014.7040156","relation":{},"subject":[],"published":{"date-parts":[[2014,12]]}}}