{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,28]],"date-time":"2025-03-28T07:25:33Z","timestamp":1743146733919,"version":"3.40.3"},"publisher-location":"Cham","reference-count":32,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783030377199"},{"type":"electronic","value":"9783030377205"}],"license":[{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2020]]},"DOI":"10.1007\/978-3-030-37720-5_1","type":"book-chapter","created":{"date-parts":[[2020,1,3]],"date-time":"2020-01-03T05:26:59Z","timestamp":1578029219000},"page":"1-15","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["MQLV: Optimal Policy of Money Management in Retail Banking with Q-Learning"],"prefix":"10.1007","author":[{"given":"Jeremy","family":"Charlier","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Gaston","family":"Ormazabal","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Radu","family":"State","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jean","family":"Hilger","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2020,1,3]]},"reference":[{"key":"1_CR1","unstructured":"Mnih, V., et al.: Playing atari with deep reinforcement learning. arXiv preprint arXiv:1312.5602 (2013)"},{"key":"1_CR2","unstructured":"Sutton, R.S.: Temporal credit assignment in reinforcement learning (1984)"},{"key":"1_CR3","unstructured":"Watkins, C.J.C.H.: Learning from delayed rewards. Ph.D. thesis, King\u2019s College, Cambridge (1989)"},{"key":"1_CR4","unstructured":"Williams, R.: A class of gradient-estimation algorithms for reinforcement learning in neural networks. In: Proceedings of the International Conference on Neural Networks, pp. II-601 (1987)"},{"key":"1_CR5","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: Imagenet classification with deep convolutional neural networks. In: Advances in Neural Information Processing Systems, pp. 1097\u20131105 (2012)"},{"key":"1_CR6","doi-asserted-by":"crossref","unstructured":"Sermanet, P., Kavukcuoglu, K., Chintala, S., LeCun, Y.: Pedestrian detection with unsupervised multi-stage feature learning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3626\u20133633 (2013)","DOI":"10.1109\/CVPR.2013.465"},{"issue":"7540","key":"1_CR7","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih, V., et al.: Human-level control through deep reinforcement learning. Nature 518(7540), 529 (2015)","journal-title":"Nature"},{"issue":"3\u20134","key":"1_CR8","first-page":"279","volume":"8","author":"CJ Watkins","year":"1992","unstructured":"Watkins, C.J., Dayan, P.: Q-learning. Mach. Learn. 8(3\u20134), 279\u2013292 (1992)","journal-title":"Mach. Learn."},{"key":"1_CR9","doi-asserted-by":"crossref","unstructured":"Van Hasselt, H., Guez, A., Silver, D.: Deep reinforcement learning with double q-learning. In: AAAI, Phoenix, AZ, vol. 2, p. 5 (2016)","DOI":"10.1609\/aaai.v30i1.10295"},{"key":"1_CR10","unstructured":"Lillicrap, T.P., et al.: Continuous control with deep reinforcement learning. arXiv preprint arXiv:1509.02971 (2015)"},{"key":"1_CR11","doi-asserted-by":"crossref","unstructured":"Halperin, I.: Qlbs: Q-learner in the black-scholes (-merton) worlds. arXiv preprint arXiv:1712.04609 (2017)","DOI":"10.2139\/ssrn.3087076"},{"issue":"3","key":"1_CR12","doi-asserted-by":"publisher","first-page":"637","DOI":"10.1086\/260062","volume":"81","author":"F Black","year":"1973","unstructured":"Black, F., Scholes, M.: The pricing of options and corporate liabilities. J. Polit. Econ. 81(3), 637\u2013654 (1973)","journal-title":"J. Polit. Econ."},{"key":"1_CR13","doi-asserted-by":"crossref","unstructured":"Merton, R.C.: Theory of rational option pricing. Bell J. Econ. Manag. Sci. 4, 141\u2013183 (1973)","DOI":"10.2307\/3003143"},{"key":"1_CR14","unstructured":"Wilmott, P.: Paul Wilmott on Quantitative Finance. Wiley, Hoboken (2013)"},{"key":"1_CR15","unstructured":"Hull, J.C.: Options Futures and Other Derivatives. Pearson Education India, Bengaluru (2003)"},{"issue":"2","key":"1_CR16","doi-asserted-by":"publisher","first-page":"177","DOI":"10.1016\/0304-405X(77)90016-2","volume":"5","author":"O Vasicek","year":"1977","unstructured":"Vasicek, O.: An equilibrium characterization of the term structure. J. Financ. Econ. 5(2), 177\u2013188 (1977)","journal-title":"J. Financ. Econ."},{"key":"1_CR17","unstructured":"Sutton, R.S., Barto, A.G.: Reinforcement Learning: An Introduction. MIT Press, Cambridge (2018)"},{"key":"1_CR18","doi-asserted-by":"crossref","unstructured":"Robbins, H., Monro, S.: A stochastic approximation method. In: Herbert Robbins Selected Papers, pp. 102\u2013109. Springer, Heidelberg (1985)","DOI":"10.1007\/978-1-4612-5110-1_9"},{"key":"1_CR19","unstructured":"Hasselt, H.V.: Double q-learning. In: Advances in Neural Information Processing Systems, pp. 2613\u20132621 (2010)"},{"key":"1_CR20","unstructured":"Murphy, S.A.: A generalization error for Q-learning. J. Mach. Learn. Res. 6(Jul), 1073\u20131097 (2005)"},{"key":"1_CR21","unstructured":"Santander: Santander product recommendation (2016). https:\/\/www.kaggle.com\/c\/santander-product-recommendation\/data"},{"key":"1_CR22","unstructured":"Tsitsiklis, J.N., Van Roy, B.: Analysis of temporal-diffference learning with function approximation. In: Advances in Neural Information Processing Systems, pp. 1075\u20131081 (1997)"},{"key":"1_CR23","doi-asserted-by":"crossref","unstructured":"Baird, L.: Residual algorithms: reinforcement learning with function approximation. In: Machine Learning Proceedings 1995, pp. 30\u201337. Elsevier (1995)","DOI":"10.1016\/B978-1-55860-377-6.50013-X"},{"key":"1_CR24","unstructured":"Goodfellow, I., Bengio, Y., Courville, A., Bengio, Y.: Deep Learning, vol. 1. MIT Press, Cambridge (2016)"},{"key":"1_CR25","doi-asserted-by":"crossref","unstructured":"Lange, S., Riedmiller, M.: Deep auto-encoder neural networks in reinforcement learning. In: The 2010 International Joint Conference on Neural Networks (IJCNN), pp. 1\u20138. IEEE (2010)","DOI":"10.1109\/IJCNN.2010.5596468"},{"key":"1_CR26","doi-asserted-by":"publisher","first-page":"253","DOI":"10.1613\/jair.3912","volume":"47","author":"MG Bellemare","year":"2013","unstructured":"Bellemare, M.G., Naddaf, Y., Veness, J., Bowling, M.: The arcade learning environment: an evaluation platform for general agents. J. Artif. Intell. Res. 47, 253\u2013279 (2013)","journal-title":"J. Artif. Intell. Res."},{"key":"1_CR27","doi-asserted-by":"crossref","unstructured":"Dabney, W., Ostrovski, G., Silver, D., Munos, R.: Implicit quantile networks for distributional reinforcement learning. arXiv preprint arXiv:1806.06923 (2018)","DOI":"10.1609\/aaai.v32i1.11791"},{"key":"1_CR28","unstructured":"Bellemare, M.G., Dabney, W., Munos, R.: A distributional perspective on reinforcement learning. arXiv preprint arXiv:1707.06887 (2017)"},{"key":"1_CR29","doi-asserted-by":"publisher","first-page":"156","DOI":"10.1016\/j.neunet.2012.11.007","volume":"41","author":"P Wawrzy\u0144Ski","year":"2013","unstructured":"Wawrzy\u0144Ski, P., Tanwani, A.K.: Autonomous reinforcement learning with experience replay. Neural Netw. 41, 156\u2013167 (2013)","journal-title":"Neural Netw."},{"key":"1_CR30","unstructured":"Silver, D., Lever, G., Heess, N., Degris, T., Wierstra, D., Riedmiller, M.: Deterministic policy gradient algorithms. In: ICML (2014)"},{"key":"1_CR31","unstructured":"Balduzzi, D., Ghifary, M.: Compatible value gradients for reinforcement learning of continuous deep policies. arXiv preprint arXiv:1509.03005 (2015)"},{"key":"1_CR32","unstructured":"Heess, N., Wayne, G., Silver, D., Lillicrap, T., Erez, T., Tassa, Y.: Learning continuous control policies by stochastic value gradients. In: Advances in Neural Information Processing Systems, pp. 2944\u20132952 (2015)"}],"container-title":["Lecture Notes in Computer Science","Mining Data for Financial Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-37720-5_1","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,1,3]],"date-time":"2025-01-03T00:07:38Z","timestamp":1735862858000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-030-37720-5_1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020]]},"ISBN":["9783030377199","9783030377205"],"references-count":32,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-37720-5_1","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2020]]},"assertion":[{"value":"3 January 2020","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"MIDAS","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Workshop on Mining Data for Financial Applications","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"W\u00fcrzburg","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Germany","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2019","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"16 September 2019","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"16 September 2019","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"midas2019","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Single-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"EasyChair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"16","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"8","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"50% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"1","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}