{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,19]],"date-time":"2025-09-19T08:36:16Z","timestamp":1758270976090,"version":"3.40.3"},"publisher-location":"Singapore","reference-count":16,"publisher":"Springer Singapore","isbn-type":[{"type":"print","value":"9789811534140"},{"type":"electronic","value":"9789811534157"}],"license":[{"start":{"date-parts":[[2020,1,1]],"date-time":"2020-01-01T00:00:00Z","timestamp":1577836800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2020]]},"DOI":"10.1007\/978-981-15-3415-7_59","type":"book-chapter","created":{"date-parts":[[2020,4,1]],"date-time":"2020-04-01T19:02:58Z","timestamp":1585767778000},"page":"693-706","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["An Adaptive Learning Rate Q-Learning Algorithm Based on Kalman Filter Inspired by Pigeon Pecking-Color Learning"],"prefix":"10.1007","author":[{"given":"Zhihui","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Li","family":"Shi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lifang","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhigang","family":"Shang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2020,4,2]]},"reference":[{"key":"59_CR1","doi-asserted-by":"publisher","first-page":"8","DOI":"10.1016\/j.arcontrol.2018.09.005","volume":"46","author":"L Busoniu","year":"2018","unstructured":"Busoniu, L., de Bruin, T., Toli\u0107, D., et al.: Reinforcement learning for control: performance, stability, and deep approximators. Ann. Rev. Control 46, 8\u201328 (2018)","journal-title":"Ann. Rev. Control"},{"issue":"6","key":"59_CR2","doi-asserted-by":"publisher","first-page":"2042","DOI":"10.1109\/TNNLS.2017.2773458","volume":"29","author":"B Kiumarsi","year":"2018","unstructured":"Kiumarsi, B., Vamvoudakis, K.G., Modares, H., et al.: Optimal and autonomous control using reinforcement learning: a survey. IEEE Trans. Neural Netw. Learn. Syst. 29(6), 2042\u20132062 (2018)","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"issue":"5","key":"59_CR3","doi-asserted-by":"publisher","first-page":"1308","DOI":"10.1109\/TNNLS.2018.2861945","volume":"30","author":"J Li","year":"2019","unstructured":"Li, J., Chai, T., Lewis, F.L., Ding, Z., Jiang, Y.: Off-policy interleaved Q-learning: optimal control for affine nonlinear discrete-time systems. IEEE Trans. Neural Netw. Learn. Syst. 30(5), 1308\u20131320 (2019)","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"issue":"1","key":"59_CR4","first-page":"589","volume":"5","author":"E Evendar","year":"2003","unstructured":"Evendar, E., Mansour, Y.: Learning rates for Q-learning. J. Mach. Learn. Res. 5(1), 589\u2013604 (2003)","journal-title":"J. Mach. Learn. Res."},{"doi-asserted-by":"crossref","unstructured":"Moriyama, K.: Learning-rate adjusting Q-learning for Prisoner\u2019s Dilemma games. In: International Conference on Web Intelligence Intelligent Agent Technology, pp. 322\u2013325. IEEE\/WIC\/ACM (2008)","key":"59_CR5","DOI":"10.1109\/WIIAT.2008.170"},{"key":"59_CR6","doi-asserted-by":"publisher","first-page":"1","DOI":"10.3389\/fpsyg.2014.00871","volume":"5","author":"Y Bai","year":"2014","unstructured":"Bai, Y., Katahira, K., Ohira, H.: Dual learning processes underlying human decision-making in reversal learning tasks: functional significance and evidence from the model fit to human behavior. Front. Psychol. 5, 1\u20138 (2014)","journal-title":"Front. Psychol."},{"key":"59_CR7","volume-title":"Reinforcement Learning: An Introduction","author":"R Sutton","year":"2018","unstructured":"Sutton, R., Barto, A.: Reinforcement Learning: An Introduction. MIT Press, Cambridge (2018)"},{"issue":"7","key":"59_CR8","doi-asserted-by":"publisher","first-page":"755","DOI":"10.1016\/S0893-6080(00)00051-4","volume":"13","author":"H Park","year":"2000","unstructured":"Park, H., Amari, S.I., Fukumizu, K.: Adaptive natural gradient learning algorithms for various stochastic models. Neural Netw. 13(7), 755\u2013764 (2000)","journal-title":"Neural Netw."},{"issue":"8","key":"59_CR9","doi-asserted-by":"publisher","first-page":"966","DOI":"10.1016\/j.mechatronics.2014.05.007","volume":"24","author":"JC Rooijen Van","year":"2014","unstructured":"Van Rooijen, J.C., Grondman, I., et al.: Learning rate free reinforcement learning for real-time motion control using a value-gradient based policy. Mechatronics 24(8), 966\u2013974 (2014)","journal-title":"Mechatronics"},{"doi-asserted-by":"crossref","unstructured":"Ruan, X., Cai, J.: Skinner-Pigeon experiment simulated based on probabilistic automata. In: IEEE Congress on Intelligent Systems, vol. 3, pp. 578\u2013581 (2009)","key":"59_CR10","DOI":"10.1109\/GCIS.2009.127"},{"issue":"1","key":"59_CR11","doi-asserted-by":"publisher","first-page":"125","DOI":"10.1016\/j.bbr.2008.10.038","volume":"198","author":"J Rose","year":"2009","unstructured":"Rose, J., Schmidt, R., et al.: Theory meets pigeons: the influence of reward-magnitude on discrimination-learninge. Behav. Brain Res. 198(1), 125\u2013129 (2009)","journal-title":"Behav. Brain Res."},{"issue":"5","key":"59_CR12","doi-asserted-by":"publisher","first-page":"128","DOI":"10.1109\/MSP.2012.2203621","volume":"29","author":"F Ramsey","year":"2012","unstructured":"Ramsey, F.: Understanding the basis of the Kalman filter via a simple and intuitive derivation [lecture notes]. IEEE Signal Process. Mag. 29(5), 128\u2013132 (2012)","journal-title":"IEEE Signal Process. Mag."},{"key":"59_CR13","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1186\/s41601-016-0032-y","volume":"2","author":"J Khodaparast","year":"2017","unstructured":"Khodaparast, J., Khederzadeh, M.: Least square and Kalman based methods for dynamic phasor estimation: a review. Prot. Control Modern Power Syst. 2, 1\u201318 (2017)","journal-title":"Prot. Control Modern Power Syst."},{"issue":"2","key":"59_CR14","doi-asserted-by":"publisher","first-page":"95","DOI":"10.1007\/s42113-019-00026-1","volume":"2","author":"C Velazquez","year":"2019","unstructured":"Velazquez, C., Villarreal, M., Bouzas, A.: Velocity estimation in reinforcement learning. Comput. Brain Behav. 2(2), 95\u2013108 (2019)","journal-title":"Comput. Brain Behav."},{"doi-asserted-by":"crossref","unstructured":"Ahumada, G.A., Nettle, C.J., Solis, M.A.: Accelerating Q-learning through Kalman filter estimations applied in a RoboCup SSL simulation. In: Robotics Symposium Competition, pp. 112\u2013117. IEEE (2014)","key":"59_CR15","DOI":"10.1109\/LARS.2013.66"},{"unstructured":"Shashua, D.C., Mannor, S.: Trust region value optimization using Kalman filtering. Mach. Learn. (2019)","key":"59_CR16"}],"container-title":["Communications in Computer and Information Science","Bio-inspired Computing: Theories and Applications"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-15-3415-7_59","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2020,4,2]],"date-time":"2020-04-02T01:29:18Z","timestamp":1585790958000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-981-15-3415-7_59"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020]]},"ISBN":["9789811534140","9789811534157"],"references-count":16,"URL":"https:\/\/doi.org\/10.1007\/978-981-15-3415-7_59","relation":{},"ISSN":["1865-0929","1865-0937"],"issn-type":[{"type":"print","value":"1865-0929"},{"type":"electronic","value":"1865-0937"}],"subject":[],"published":{"date-parts":[[2020]]},"assertion":[{"value":"2 April 2020","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"BIC-TA","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Bio-Inspired Computing: Theories and Applications","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Zhengzhou","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2019","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 November 2019","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"25 November 2019","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"14","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"bicta2019","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/2019.bicta.org","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Single-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"EasyChair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"197","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"121","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"61% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}