{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,31]],"date-time":"2026-03-31T08:36:50Z","timestamp":1774946210468,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":12,"publisher":"ACM","license":[{"start":{"date-parts":[[2007,6,20]],"date-time":"2007-06-20T00:00:00Z","timestamp":1182297600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2007,6,20]]},"DOI":"10.1145\/1273496.1273534","type":"proceedings-article","created":{"date-parts":[[2008,10,7]],"date-time":"2008-10-07T13:06:52Z","timestamp":1223384812000},"page":"297-304","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":25,"title":["Bayesian actor-critic algorithms"],"prefix":"10.1145","author":[{"given":"Mohammad","family":"Ghavamzadeh","sequence":"first","affiliation":[{"name":"University of Alberta, Edmonton, Alberta, Canada"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yaakov","family":"Engel","sequence":"additional","affiliation":[{"name":"University of Alberta, Edmonton, Alberta, Canada"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2007,6,20]]},"reference":[{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/TSMC.1983.6313077"},{"key":"e_1_3_2_1_3_1","article-title":"Infinite-horizon policy-gradient estimation","author":"Baxter J.","year":"2001","unstructured":"Baxter , J. , & Bartlett , P. ( 2001 ). Infinite-horizon policy-gradient estimation . Journal of Artificial Intelligence Research, 15. Baxter, J., & Bartlett, P. (2001). Infinite-horizon policy-gradient estimation. Journal of Artificial Intelligence Research, 15.","journal-title":"Journal of Artificial Intelligence Research, 15."},{"key":"e_1_3_2_1_4_1","volume-title":"Neuro-dynamic programming","author":"Bertsekas D.","year":"1996","unstructured":"Bertsekas , D. , & Tsitsiklis , J. ( 1996 ). Neuro-dynamic programming . Athena Scientific . Bertsekas, D., & Tsitsiklis, J. (1996). Neuro-dynamic programming. Athena Scientific."},{"key":"e_1_3_2_1_6_1","volume-title":"Proceedings of the 20th International Conference on Machine Learning.","author":"Engel Y.","year":"2003","unstructured":"Engel , Y. , Mannor , S. , & Meir , R. ( 2003 ). Bayes meets Bellman: The Gaussian process approach to temporal difference learning . Proceedings of the 20th International Conference on Machine Learning. Engel, Y., Mannor, S., & Meir, R. (2003). Bayes meets Bellman: The Gaussian process approach to temporal difference learning. Proceedings of the 20th International Conference on Machine Learning."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/1102351.1102377"},{"key":"e_1_3_2_1_8_1","volume-title":"Bayesian policy gradient algorithms. Advances in Neural Information Processing Systems 19","author":"Ghavamzadeh M.","year":"2007","unstructured":"Ghavamzadeh , M. , & Engel , Y. ( 2007 ). Bayesian policy gradient algorithms. Advances in Neural Information Processing Systems 19 . Cambridge, MA : MIT Press . Ghavamzadeh, M., & Engel, Y. (2007). Bayesian policy gradient algorithms. Advances in Neural Information Processing Systems 19. Cambridge, MA: MIT Press."},{"key":"e_1_3_2_1_9_1","volume-title":"Proceedings of Advances in Neural Information Processing Systems.","author":"Kakade S.","year":"2002","unstructured":"Kakade , S. ( 2002 ). A natural policy gradient . Proceedings of Advances in Neural Information Processing Systems. Kakade, S. (2002). A natural policy gradient. Proceedings of Advances in Neural Information Processing Systems."},{"key":"e_1_3_2_1_10_1","volume-title":"Actor-Critic algorithms. Advances in Neural Information Processing Systems 12","author":"Konda V.","year":"2000","unstructured":"Konda , V. , & Tsitsiklis , J. ( 2000 ). Actor-Critic algorithms. Advances in Neural Information Processing Systems 12 . Konda, V., & Tsitsiklis, J. (2000). Actor-Critic algorithms. Advances in Neural Information Processing Systems 12."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1016\/0378-3758(91)90002-V"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"crossref","DOI":"10.1017\/CBO9780511809682","volume-title":"Kernel methods for pattern analysis","author":"Shawe-Taylor J.","year":"2004","unstructured":"Shawe-Taylor , J. , & Cristianini , N. ( 2004 ). Kernel methods for pattern analysis . Cambridge University Press . Shawe-Taylor, J., & Cristianini, N. (2004). Kernel methods for pattern analysis. Cambridge University Press."},{"key":"e_1_3_2_1_14_1","volume-title":"An introduction to reinforcement learning","author":"Sutton R.","year":"1998","unstructured":"Sutton , R. , & Barto , A. ( 1998 ). An introduction to reinforcement learning . MIT Press . Sutton, R., & Barto, A. (1998). An introduction to reinforcement learning. MIT Press."},{"key":"e_1_3_2_1_15_1","volume-title":"Policy gradient methods for reinforcement learning with function approximation. Advances in Neural Information Processing Systems 12","author":"Sutton R.","year":"2000","unstructured":"Sutton , R. , McAllester , D. , Singh , S. , & Mansour , Y. ( 2000 ). Policy gradient methods for reinforcement learning with function approximation. Advances in Neural Information Processing Systems 12 . Sutton, R., McAllester, D., Singh, S., & Mansour, Y. (2000). Policy gradient methods for reinforcement learning with function approximation. Advances in Neural Information Processing Systems 12."}],"event":{"name":"ICML '07 & ILP '07: The 24th Annual International Conference on Machine Learning held in conjunction with the 2007 International Conference on Inductive Logic Programming","location":"Corvalis Oregon USA","acronym":"ICML '07 & ILP '07","sponsor":["Machine Learning Journal"]},"container-title":["Proceedings of the 24th international conference on Machine learning"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/1273496.1273534","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/1273496.1273534","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T14:58:01Z","timestamp":1750258681000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/1273496.1273534"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2007,6,20]]},"references-count":12,"alternative-id":["10.1145\/1273496.1273534","10.1145\/1273496"],"URL":"https:\/\/doi.org\/10.1145\/1273496.1273534","relation":{},"subject":[],"published":{"date-parts":[[2007,6,20]]},"assertion":[{"value":"2007-06-20","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}