{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,28]],"date-time":"2025-03-28T05:46:34Z","timestamp":1743140794428,"version":"3.40.3"},"publisher-location":"Singapore","reference-count":21,"publisher":"Springer Singapore","isbn-type":[{"type":"print","value":"9789811322020"},{"type":"electronic","value":"9789811322037"}],"license":[{"start":{"date-parts":[[2018,1,1]],"date-time":"2018-01-01T00:00:00Z","timestamp":1514764800000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2018]]},"DOI":"10.1007\/978-981-13-2203-7_17","type":"book-chapter","created":{"date-parts":[[2018,9,8]],"date-time":"2018-09-08T10:33:22Z","timestamp":1536402802000},"page":"225-240","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["A Novel Experience-Based Exploration Method for Q-Learning"],"prefix":"10.1007","author":[{"given":"Bohong","family":"Yang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hong","family":"Lu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Baogen","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zheng","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wenqiang","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2018,9,9]]},"reference":[{"key":"17_CR1","volume-title":"Reinforcement Learning: An Introduction","author":"R Sutton","year":"1998","unstructured":"Sutton, R., Barto, A.: Reinforcement Learning: An Introduction. MIT Press, Cambridge (1998)"},{"issue":"7540","key":"17_CR2","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih, V., et al.: Human-level control through deep reinforcement learning. Nature 518(7540), 529\u2013533 (2015)","journal-title":"Nature"},{"issue":"3\u20134","key":"17_CR3","first-page":"279","volume":"8","author":"CJCH Watkins","year":"1992","unstructured":"Watkins, C.J.C.H., Dayan, P.: Q-learning. Machine learning 8(3\u20134), 279\u2013292 (1992)","journal-title":"Machine learning"},{"key":"17_CR4","unstructured":"Hasselt, H.V., Guez, A., Silver, D.: Deep reinforcement learning with double Q-learning. In: Computer Science (2015)"},{"key":"17_CR5","first-page":"2613","volume-title":"Double Q-learning","author":"HV Hasselt","year":"2010","unstructured":"Hasselt, H.V.: Double Q-learning, pp. 2613\u20132621. Mit Press, Cambridge (2010)"},{"key":"17_CR6","unstructured":"Schaul, T., Quan, J., Antonoglou, I., Silver, D.: Prioritized Experience Replay. arXiv preprint arXiv:1511.05952 (2015)"},{"key":"17_CR7","unstructured":"Mnih, V., et al.: Asynchronous methods for deep reinforcement learning. arXiv preprint arXiv:1602.01783 (2016)"},{"key":"17_CR8","unstructured":"Parisotto, E., Ba, J., Salakhutdinov, R.: Actor-Mimic: Deep Multitask and Transfer Reinforcement Learning. arXiv preprint arXiv:1511.06342 (2015)"},{"key":"17_CR9","unstructured":"Wang, Z., Freitas, N., Lanctot, M.: Dueling Network Architectures for Deep Reinforcement Learning. arXiv preprint arXiv:1511.06581 (2015)"},{"issue":"7587","key":"17_CR10","doi-asserted-by":"publisher","first-page":"484","DOI":"10.1038\/nature16961","volume":"529","author":"D Silver","year":"2015","unstructured":"Silver, D., et al.: Mastering the game of go with deep neural networks and tree search. Nature 529(7587), 484\u2013489 (2015)","journal-title":"Nature"},{"key":"17_CR11","unstructured":"Silver, D., Lever, G., Heess, N., Degris, T., Wierstra, D., Riedmiller, M.: Deterministic policy gradient algorithms. In: International Conference on Machine Learning, pp. 387\u2013395 (2014)"},{"key":"17_CR12","unstructured":"Lillicrap, T.P., et al.: Continuous control with deep reinforcement learning. arXiv preprint arXiv:1509.02971 (2015)"},{"key":"17_CR13","doi-asserted-by":"crossref","unstructured":"MacAlpine, P., Depinet, M., Stone, P.: UT Austin villa 2014: RoboCup 3D simulation league champion via overlapping layered learning. In: Proceedings of the Twenty-Ninth AAAI Conference on Artificial Intelligence, pp. 2842\u20132848 (2015)","DOI":"10.1609\/aaai.v29i1.9540"},{"issue":"12","key":"17_CR14","doi-asserted-by":"publisher","first-page":"3083","DOI":"10.1109\/TNNLS.2015.2403394","volume":"26","author":"C Yu","year":"2015","unstructured":"Yu, C., Zhang, M., Ren, F., Tan, G.: Emotional multiagent reinforcement learning in spatial social dilemmas. IEEE Trans. Neural Netw. Learn. Syst. 26(12), 3083\u20133096 (2015)","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"issue":"2","key":"17_CR15","doi-asserted-by":"publisher","first-page":"346","DOI":"10.1109\/TNNLS.2014.2371046","volume":"26","author":"D Zhao","year":"2015","unstructured":"Zhao, D., Zhu, Y.: MEC\u2013A near-optimal online reinforcement learning algorithm for continuous deterministic systems. IEEE Trans. Neural Netw. Learn. Syst. 26(2), 346\u2013356 (2015)","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"issue":"9","key":"17_CR16","doi-asserted-by":"publisher","first-page":"2163","DOI":"10.1109\/TNNLS.2014.2376703","volume":"26","author":"M Kusy","year":"2015","unstructured":"Kusy, M., Zajdel, R.: Application of reinforcement learning algorithms for the adaptive computation of the smoothing parameter for probabilistic neural network. IEEE Trans. Neural Netw. Learn. Syst. 26(9), 2163\u20132175 (2015)","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"issue":"5","key":"17_CR17","doi-asserted-by":"publisher","first-page":"889","DOI":"10.1109\/TNNLS.2014.2327636","volume":"26","author":"TH Teng","year":"2015","unstructured":"Teng, T.H., Tan, A.H., Zurada, J.M.: Self-organizing neural networks integrating domain knowledge and reinforcement learning. IEEE Trans. Neural Netw. Learn. Syst. 26(5), 889\u2013902 (2015)","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"issue":"3","key":"17_CR18","doi-asserted-by":"publisher","first-page":"653","DOI":"10.1109\/TNNLS.2016.2522401","volume":"28","author":"Y Deng","year":"2017","unstructured":"Deng, Y., Bao, F., Kong, Y., Ren, Z., Dai, Q.: Deep direct reinforcement learning for financial signal representation and trading. IEEE Trans. Neural Netw. Learn. Syst. 28(3), 653\u2013664 (2017)","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"17_CR19","unstructured":"DeNero, J., Klein, D.: Pacman Project (2012). http:\/\/ai.berkeley.edu\/reinforcement.html"},{"key":"17_CR20","unstructured":"Ng, A.Y., Jordan, M.: PEGASUS: A policy search method for large MDPs and POMDPs. In: Proceedings of the Sixteenth Conference on Uncertainty in Artificial Intelligence. Morgan Kaufmann Publishers Inc., pp. 406\u2013415 (2014)"},{"key":"17_CR21","doi-asserted-by":"crossref","unstructured":"Jia, Y., et al.: Caffe: Convolutional Architecture for Fast Feature Embedding. arXiv preprint arXiv:1408.5093 (2014)","DOI":"10.1145\/2647868.2654889"}],"container-title":["Communications in Computer and Information Science","Data Science"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-13-2203-7_17","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,13]],"date-time":"2024-03-13T11:52:45Z","timestamp":1710330765000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-13-2203-7_17"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018]]},"ISBN":["9789811322020","9789811322037"],"references-count":21,"URL":"https:\/\/doi.org\/10.1007\/978-981-13-2203-7_17","relation":{},"ISSN":["1865-0929","1865-0937"],"issn-type":[{"type":"print","value":"1865-0929"},{"type":"electronic","value":"1865-0937"}],"subject":[],"published":{"date-parts":[[2018]]},"assertion":[{"value":"9 September 2018","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICPCSEE","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference of Pioneering Computer Scientists, Engineers and Educators","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Zhengzhou","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2018","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"21 September 2018","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 September 2018","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ICPCSEE 2018","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}