{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,15]],"date-time":"2025-11-15T10:16:26Z","timestamp":1763201786187},"publisher-location":"Berlin, Heidelberg","reference-count":25,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"type":"print","value":"9783642395925"},{"type":"electronic","value":"9783642395932"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2013]]},"DOI":"10.1007\/978-3-642-39593-2_8","type":"book-chapter","created":{"date-parts":[[2013,7,24]],"date-time":"2013-07-24T12:31:19Z","timestamp":1374669079000},"page":"93-101","source":"Crossref","is-referenced-by-count":6,"title":["Reward Shaping for Statistical Optimisation of Dialogue Management"],"prefix":"10.1007","author":[{"given":"Layla","family":"El Asri","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Romain","family":"Laroche","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Olivier","family":"Pietquin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"key":"8_CR1","unstructured":"Bos, J., Klein, E., Lemon, O., Oka, T.: DIPPER: Description and Formalisation of an Information-State Update Dialogue System Architecture. In: Proceedings of SIGdial Workshop on Discourse and Dialogue (2003)"},{"key":"8_CR2","unstructured":"Boularias, A., Chinaei, H.R., Chaib-draa, B.: Learning the reward model of dialogue pomdps from data. In: Proceedings of NIPS (2010)"},{"key":"8_CR3","first-page":"33","volume":"22","author":"S.J. Bradtke","year":"1996","unstructured":"Bradtke, S.J., Barto, A.G.: Linear least-squares algorithms for temporal difference learning. Machine Learning\u00a022, 33\u201357 (1996)","journal-title":"Machine Learning"},{"key":"8_CR4","doi-asserted-by":"crossref","unstructured":"Chandramohan, S., Geist, M., Lef\u00e8vre, F., Pietquin, O.: User simulation in dialogue systems using inverse reinforcement learning. In: Proceedings of Interspeech (2011)","DOI":"10.21437\/Interspeech.2011-302"},{"key":"8_CR5","unstructured":"El-Asri, L., Laroche, R., Pietquin, O.: Reward function learning for dialogue management. In: Proceedings of STAIRS (2012)"},{"key":"8_CR6","first-page":"1107","volume":"4","author":"M.G. Lagoudakis","year":"2003","unstructured":"Lagoudakis, M.G., Parr, R.: Least-squares policy iteration. Journal of Machine Learning Research\u00a04, 1107\u20131149 (2003)","journal-title":"Journal of Machine Learning Research"},{"key":"8_CR7","unstructured":"Larsen, L.B.: Issues in the evaluation of spoken dialogue systems using objective and subjective measures. In: Proceedings of IEEE ASRU, pp. 209\u2013214 (2003)"},{"key":"8_CR8","doi-asserted-by":"crossref","unstructured":"Lemon, O., Georgila, K., Henderson, J., Stuttle, M.: An ISU dialogue system exhibiting reinforcement learning of dialogue policies: Generic slot-filling in the talk in-car system. In: Proceedings of EACL (2006)","DOI":"10.3115\/1608974.1608986"},{"key":"8_CR9","doi-asserted-by":"crossref","unstructured":"Lemon, O., Georgila, K., Henderson, J., Stuttle, M.: An ISU dialogue system exhibiting reinforcement learning of dialogue policies: generic slot-filling in the talk in-car system. In: Proceedings of EACL (2006)","DOI":"10.3115\/1608974.1608986"},{"key":"8_CR10","doi-asserted-by":"crossref","unstructured":"Lemon, O., Pietquin, O.: Machine learning for spoken dialogue systems. In: Proceedings of Interspeech, pp. 2685\u20132688 (2007)","DOI":"10.21437\/Interspeech.2007-705"},{"key":"8_CR11","doi-asserted-by":"crossref","unstructured":"Li, L., Williams, J.D., Balakrishnan, S.: Reinforcement learning for dialog management using least-squares policy iteration and fast feature selection. In: Proceedings of Interspeech (2009)","DOI":"10.21437\/Interspeech.2009-659"},{"key":"8_CR12","doi-asserted-by":"crossref","unstructured":"Mataric, M.J.: Reward functions for accelerated learning. In: Proceedings of ICML, pp. 181\u2013189 (1994)","DOI":"10.1016\/B978-1-55860-335-6.50030-1"},{"key":"8_CR13","unstructured":"Meguro, T., Higashinaka, R., Minami, Y., Dohsaka, K.: Controlling listening-oriented dialogue using partially observable markov decision processes. In: Proceedings of Coling (2010)"},{"key":"8_CR14","unstructured":"Ng, A.Y., Harada, D., Russell, S.: Policy invariance under reward transformations: Theory and application to reward shaping. In: Proceedings of ICML, pp. 278\u2013287 (1999)"},{"key":"8_CR15","doi-asserted-by":"publisher","first-page":"716","DOI":"10.1016\/j.specom.2008.03.010","volume":"50","author":"T. Paek","year":"2008","unstructured":"Paek, T., Pieraccini, R.: Automating spoken dialogue management design using machine learning: An industry perspective. Speech Communication\u00a050, 716\u2013729 (2008)","journal-title":"Speech Communication"},{"issue":"3","key":"8_CR16","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/1966407.1966412","volume":"7","author":"O. Pietquin","year":"2011","unstructured":"Pietquin, O., Geist, M., Chandramohan, S., Frezza-Buet, H.: Sample-efficient batch reinforcement learning for dialogue management optimization. ACM Transaction on Speech and Language Processing\u00a07(3), 1\u201321 (2011)","journal-title":"ACM Transaction on Speech and Language Processing"},{"key":"8_CR17","unstructured":"Pietquin, O., Rossignol, S., Ianotto, M.: Training Bayesian networks for realistic man-machine spoken dialogue simulation. In: Proceedings of IWSDS 2009 (2009)"},{"key":"8_CR18","doi-asserted-by":"crossref","unstructured":"Rieser, V., Lemon, O.: Learning and evaluation of dialogue strategies for new applications: Empirical methods for optimization from small data sets. Computational Linguistics\u00a037 (2011)","DOI":"10.1162\/coli_a_00038"},{"key":"8_CR19","doi-asserted-by":"crossref","unstructured":"Russell, S.: Learning agents for uncertain environments (extended abstract). In: Proceedings of COLT (1998)","DOI":"10.1145\/279943.279964"},{"key":"8_CR20","doi-asserted-by":"publisher","first-page":"72","DOI":"10.2307\/1412159","volume":"15","author":"C. Spearman","year":"1904","unstructured":"Spearman, C.: The proof and measurement of association between two things. American Journal of Psychology\u00a015, 72\u2013101 (1904)","journal-title":"American Journal of Psychology"},{"key":"8_CR21","doi-asserted-by":"crossref","unstructured":"Sugiyama, H., Meguro, T., Minami, Y.: Preference-learning based Inverse Reinforcement Learning for Dialog Control. In: Proceedings of Interspeech (2012)","DOI":"10.21437\/Interspeech.2012-72"},{"key":"8_CR22","unstructured":"Sutton, R.S., Barto, A.G.: Reinforcement Learning. An introduction, pp. 56\u201357. MIT Press (1998)"},{"key":"8_CR23","doi-asserted-by":"crossref","unstructured":"Walker, M.A., Fromer, J.C., Narayanan, S.: Learning optimal dialogue strategies: A case study of a spoken dialogue agent for email. In: Proceedings of COLING\/ACL, pp. 1345\u20131352 (1998)","DOI":"10.3115\/980432.980788"},{"key":"8_CR24","doi-asserted-by":"crossref","unstructured":"Walker, M.A., Litman, D.J., Kamm, C.A., Abella, A.: PARADISE: a framework for evaluating spoken dialogue agents. In: Proceedings of EACL, pp. 271\u2013280 (1997)","DOI":"10.3115\/979617.979652"},{"key":"8_CR25","doi-asserted-by":"publisher","first-page":"231","DOI":"10.1016\/j.csl.2006.06.008","volume":"21","author":"J.D. Williams","year":"2007","unstructured":"Williams, J.D., Young, S.: Partially observable markov decision processes for spoken dialog systems. Computer Speech and Language\u00a021, 231\u2013422 (2007)","journal-title":"Computer Speech and Language"}],"container-title":["Lecture Notes in Computer Science","Statistical Language and Speech Processing"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-642-39593-2_8","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,3,1]],"date-time":"2022-03-01T19:57:53Z","timestamp":1646164673000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-642-39593-2_8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2013]]},"ISBN":["9783642395925","9783642395932"],"references-count":25,"URL":"https:\/\/doi.org\/10.1007\/978-3-642-39593-2_8","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2013]]}}}