{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2023,8,10]],"date-time":"2023-08-10T21:00:52Z","timestamp":1691701252129},"reference-count":28,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2014,10,15]],"date-time":"2014-10-15T00:00:00Z","timestamp":1413331200000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Speech Technol"],"published-print":{"date-parts":[[2014,12]]},"DOI":"10.1007\/s10772-014-9224-x","type":"journal-article","created":{"date-parts":[[2014,10,14]],"date-time":"2014-10-14T08:36:30Z","timestamp":1413275790000},"page":"325-340","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":6,"title":["Dialogue POMDP components (Part II): learning the reward function"],"prefix":"10.1007","volume":"17","author":[{"given":"H.","family":"Chinaei","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"B.","family":"Chaib-draa","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2014,10,15]]},"reference":[{"key":"9224_CR1","doi-asserted-by":"crossref","unstructured":"Abbeel, P., Ng, A. Y. (2004). Apprenticeship learning via inverse reinforcement learning. In Proceedings of the 21st International Conference on Machine learning (ICML\u201904). Banff, AB, Canada.","DOI":"10.1145\/1015330.1015430"},{"key":"9224_CR2","unstructured":"Boularias, A., Chinaei, H. R., & Chaib-draa, B., (2010). Learning the reward model of dialogue POMDPs from data. In NIPS 2010 Workshop on Machine Learning for Assistive Technologies. Vancouver, BC, Canada."},{"key":"9224_CR3","first-page":"182","volume":"15","author":"A Boularias","year":"2011","unstructured":"Boularias, A., Kober, J., & Peters, J. (2011). Relative entropy inverse reinforcement learning. Journal of Machine Learning Research\u2014Proceedings Track, 15, 182\u2013189.","journal-title":"Journal of Machine Learning Research\u2014Proceedings Track"},{"key":"9224_CR4","unstructured":"Chandramohan, S., Geist, M., Lef\u00e8vre, F., & Pietquin, O. (2012). Behavior specific user simulation in spoken dialogue systems. In Proceedings of the IEEE ITG Conference on Speech Communication. Braunschweig, Germany."},{"key":"9224_CR5","doi-asserted-by":"crossref","unstructured":"Chinaei, H. R., & Chaib-draa, B. (2011). Learning dialogue POMDP models from data. In Proceedings of the 24th Canadian Conference on Advances in Artificial Intelligence (Canadian AI\u201911). St. John\u2019s, NL, Canada.","DOI":"10.1007\/978-3-642-21043-3_11"},{"key":"9224_CR6","doi-asserted-by":"crossref","unstructured":"Chinaei, H. R., & Chaib-draa, B. (2014). Dialogue POMDP components (Part I): Learning states and observations. International Journal of Speech Technologyn (this issue).","DOI":"10.1007\/s10772-014-9224-x"},{"key":"9224_CR7","doi-asserted-by":"crossref","unstructured":"Chinaei, H. R., Chaib-draa, B., & Lamontagne, L. (2012). Learning observation models for dialogue POMDPs. In Proceedings of the 24th Canadian conference on advances in Artificial Intelligence (Canadian AI\u201912). Toronto, ON, Canada.","DOI":"10.1007\/978-3-642-30353-1_24"},{"key":"9224_CR8","first-page":"691","volume":"12","author":"J Choi","year":"2011","unstructured":"Choi, J., & Kim, K.-E. (2011). Inverse reinforcement learning in partially observable environments. Journal of Machine Learning Research, 12, 691\u2013730.","journal-title":"Journal of Machine Learning Research"},{"key":"9224_CR9","unstructured":"Ga\u0161i\u0107, M. (2011). Statistical Dialogue Modelling. PhD thesis, Department of Engineering, University of Cambridge."},{"key":"9224_CR10","unstructured":"Ji, S., Parr, R., Li, H., Liao, X., & Carin, L. (2007). Point-based policy iteration. In Proceedings of the 22nd National Conference on Artificial Intelligence (vol. 2) (AAAI\u201907). Vancouver, BC, Canada."},{"issue":"4","key":"9224_CR11","doi-asserted-by":"crossref","first-page":"1029","DOI":"10.1109\/TASL.2010.2076394","volume":"19","author":"D Kim","year":"2011","unstructured":"Kim, D., Kim, J., & Kim, K. (2011). Robust performance evaluation of POMDP-based dialogue systems. IEEE Transactions on Audio, Speech, and Language Processing, 19(4), 1029\u20131040.","journal-title":"IEEE Transactions on Audio, Speech, and Language Processing"},{"key":"9224_CR12","unstructured":"Neu, G., Szepesv\u00e1ri, C. (2007). Apprenticeship learning using inverse reinforcement learning and gradient methods. In Proceedings of the 23rd Conference on Uncertainty in Artificial Intelligence (UAI\u201907). Vancouver, BC, Canada."},{"key":"9224_CR13","unstructured":"Ng, A. Y., Russell, S. J. (2000). Algorithms for inverse reinforcement learning. In Proceedings of the 17th International Conference on Machine Learning (ICML\u201900). Stanford, CA, USA."},{"issue":"8","key":"9224_CR14","doi-asserted-by":"crossref","first-page":"716","DOI":"10.1016\/j.specom.2008.03.010","volume":"50","author":"T Paek","year":"2008","unstructured":"Paek, T., & Pieraccini, R. (2008). Automating spoken dialogue management design using machine learning: An industry perspective. Speech Communication, 50(8), 716\u2013729.","journal-title":"Speech Communication"},{"key":"9224_CR15","doi-asserted-by":"crossref","unstructured":"Pinault, F. and Lef\u00e8vre, F. (2011). Semantic graph clustering for pomdp-based spoken dialog systems. In Proceedings of the 12th Annual Conference of the International Speech Communication Association (INTERSPEECH\u201911). Florence, Italy.","DOI":"10.21437\/Interspeech.2011-439"},{"key":"9224_CR16","unstructured":"Pineau, J., Gordon, G., & Thrun, S. (2003). Point-based value iteration: An anytime algorithm for POMDPs. In International Joint Conference on Artificial Intelligence (IJCAI\u201903). Acapulco, Mexico."},{"issue":"2","key":"9224_CR17","first-page":"124","volume":"16","author":"J Pineau","year":"2011","unstructured":"Pineau, J., West, R., Atrash, A., Villemure, J., & Routhier, F. (2011). On the feasibility of using a standardized test for evaluating a speech-controlled smart wheelchair. International Journal of Intelligent Control and Systems, 16(2), 124\u2013131.","journal-title":"International Journal of Intelligent Control and Systems"},{"key":"9224_CR18","unstructured":"Ramachandran, D., & Amir, E. (2007). Bayesian inverse reinforcement learning. In Proceedings of the 20th International Joint Conference on Artificial Intelligence (IJCAI\u201907). Hyderabad, India."},{"key":"9224_CR19","doi-asserted-by":"crossref","unstructured":"Roy, N., Pineau, J., & Thrun, S. (2000). Spoken dialogue management using probabilistic reasoning. In Proceedings of the 38th Annual Meeting on Association for Computational Linguistics (ACL\u201900). Hong Kong.","DOI":"10.3115\/1075218.1075231"},{"key":"9224_CR20","doi-asserted-by":"crossref","unstructured":"Spaan, M., & Vlassis, N. (2005). Perseus: Randomized point-based value iteration for POMDPs. Journal of Artificial Intelligence Research, 24(1), 195\u2013220.","DOI":"10.1613\/jair.1659"},{"key":"9224_CR21","unstructured":"Syed, U. and Schapire, R. (2008). A game-theoretic approach to apprenticeship learning. In Proceedings of the Twenty-First Annual Conference on Neural Information Processing Systems. Vancouver, BC, Canada."},{"key":"9224_CR22","unstructured":"Thomson, B. (2009). Statistical Methods for Spoken Dialogue Management. PhD thesis, Department of Engineering, University of Cambridge."},{"key":"9224_CR23","unstructured":"Williams, J. D. (2006). Partially Observable Markov Decision Processes for Spoken Dialogue Management. PhD thesis, Department of Engineering, University of Cambridge."},{"key":"9224_CR24","unstructured":"Williams, J. D., & Young, S. (2005). The SACTI-1 corpus: Guide for research users. Technical Report. Department of Engineering, University of Cambridge."},{"key":"9224_CR25","doi-asserted-by":"crossref","first-page":"393","DOI":"10.1016\/j.csl.2006.06.008","volume":"21","author":"JD Williams","year":"2007","unstructured":"Williams, J. D., & Young, S. (2007). Partially observable Markov decision processes for spoken dialog systems. Computer Speech and Language, 21, 393\u2013422.","journal-title":"Computer Speech and Language"},{"key":"9224_CR26","doi-asserted-by":"crossref","unstructured":"Zhang, B., Cai, Q., Mao, J., Chang, E., & Guo, B. (2001a). Spoken dialogue management as planning and acting under uncertainty. In Proceedings of the 9th European Conference on Speech Communication and Technology (Eurospeech\u201901). Aalborg, Denmark.","DOI":"10.21437\/Eurospeech.2001-511"},{"key":"9224_CR27","unstructured":"Zhang, B., Cai, Q., Mao, J., & Guo, B. (2001b). Planning and acting under uncertainty: A new model for spoken dialogue system. In Proceedings of the 17th Conference in Uncertainty in Artificial Intelligence (UAI\u201901), Seattle, WA, USA."},{"key":"9224_CR28","unstructured":"Ziebart, B., Maas, A., Bagnell, J., & Dey, A. (2008). Maximum entropy inverse reinforcement learning. In Proceedings of the 23rd National Conference on Artificial Intelligence (AAAI\u201908). Chicago, IL, USA."}],"container-title":["International Journal of Speech Technology"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-014-9224-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10772-014-9224-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-014-9224-x","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,7,17]],"date-time":"2023-07-17T07:14:42Z","timestamp":1689578082000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10772-014-9224-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2014,10,15]]},"references-count":28,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2014,12]]}},"alternative-id":["9224"],"URL":"https:\/\/doi.org\/10.1007\/s10772-014-9224-x","relation":{},"ISSN":["1381-2416","1572-8110"],"issn-type":[{"value":"1381-2416","type":"print"},{"value":"1572-8110","type":"electronic"}],"subject":[],"published":{"date-parts":[[2014,10,15]]}}}