{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,7,20]],"date-time":"2025-07-20T03:48:10Z","timestamp":1752983290793,"version":"3.40.3"},"publisher-location":"Singapore","reference-count":30,"publisher":"Springer Nature Singapore","isbn-type":[{"type":"print","value":"9789811941085"},{"type":"electronic","value":"9789811941092"}],"license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022]]},"DOI":"10.1007\/978-981-19-4109-2_26","type":"book-chapter","created":{"date-parts":[[2022,8,1]],"date-time":"2022-08-01T14:38:26Z","timestamp":1659364706000},"page":"266-280","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["A Survey of\u00a0Linear Value Function Approximation in\u00a0Reinforcement Learning"],"prefix":"10.1007","author":[{"given":"Shicheng","family":"Guo","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xueyu","family":"Wei","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yun","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei","family":"Xue","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xuangou","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bo","family":"Wei","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,8,1]]},"reference":[{"key":"26_CR1","unstructured":"Li, Y.: Deep reinforcement learning. arXiv preprint arXiv:1810.06339 (2018)"},{"key":"26_CR2","volume-title":"Reinforcement Learning: An Introduction","author":"RS Sutton","year":"2018","unstructured":"Sutton, R.S., Barto, A.G.: Reinforcement Learning: An Introduction. MIT Press, Cambridge (2018)"},{"issue":"7587","key":"26_CR3","doi-asserted-by":"publisher","first-page":"484","DOI":"10.1038\/nature16961","volume":"529","author":"D Silver","year":"2016","unstructured":"Silver, D., et al.: Mastering the game of go with deep neural networks and tree search. Nature 529(7587), 484\u2013489 (2016)","journal-title":"Nature"},{"key":"26_CR4","unstructured":"OpenAI, et al.: Learning dexterous in-hand manipulation. arXiv preprint arXiv:1808.00177 (2018)"},{"issue":"6","key":"26_CR5","doi-asserted-by":"publisher","first-page":"1305","DOI":"10.1007\/s00607-019-00773-w","volume":"102","author":"Y Li","year":"2019","unstructured":"Li, Y., Ni, P., Chang, V.: Application of deep reinforcement learning in stock trading strategies and stock forecasting. Computing 102(6), 1305\u20131322 (2019). https:\/\/doi.org\/10.1007\/s00607-019-00773-w","journal-title":"Computing"},{"issue":"5","key":"26_CR6","doi-asserted-by":"publisher","first-page":"674","DOI":"10.1109\/9.580874","volume":"42","author":"JN Tsitsiklis","year":"1997","unstructured":"Tsitsiklis, J.N., Van Roy, B.: An analysis of temporal-difference learning with function approximation. IEEE Trans. Autom. Control 42(5), 674\u2013690 (1997)","journal-title":"IEEE Trans. Autom. Control"},{"issue":"1","key":"26_CR7","first-page":"33","volume":"22","author":"SJ Bradtke","year":"1996","unstructured":"Bradtke, S.J., Barto, A.G.: Linear least-squares algorithms for temporal difference learning. Mach. Learn. 22(1), 33\u201357 (1996)","journal-title":"Mach. Learn."},{"key":"26_CR8","doi-asserted-by":"crossref","unstructured":"Bolstad, W.M., Curran, J.M.: Introduction to Bayesian Statistics. Wiley, Hoboken (2016)","DOI":"10.1002\/9781118593165"},{"issue":"2","key":"26_CR9","doi-asserted-by":"publisher","first-page":"233","DOI":"10.1023\/A:1017936530646","volume":"49","author":"JA Boyan","year":"2002","unstructured":"Boyan, J.A.: Technical update: least-squares temporal difference learning. Mach. Learn. 49(2), 233\u2013246 (2002)","journal-title":"Mach. Learn."},{"key":"26_CR10","doi-asserted-by":"crossref","unstructured":"Xu, X., He, H.g., Hu, D.: Efficient reinforcement learning using recursive least-squares methods. J. Artif. Intell. Res. 16, 259\u2013292 (2002)","DOI":"10.1613\/jair.946"},{"key":"26_CR11","unstructured":"Wang, S., Jia, D., Weng, X.: Deep reinforcement learning for autonomous driving. arXiv preprint arXiv:1811.11329 (2018)"},{"key":"26_CR12","doi-asserted-by":"crossref","unstructured":"Liu, X.Y., et al.: FinRL: a deep reinforcement learning library for automated stock trading in quantitative finance. arXiv preprint arXiv:2011.09607 (2020)","DOI":"10.2139\/ssrn.3737859"},{"key":"26_CR13","unstructured":"Killian TW., Zhang H., Subramanian J., Fatemi M., Ghassemi M.: An empirical study of representation learning for reinforcement learning in healthcare. arXiv preprint arXiv:2011.11235 (2020)"},{"key":"26_CR14","unstructured":"Geramifard, A., Bowling, M., Sutton, R.S.: Incremental least-squares temporal difference learning. In: Proceedings of the National Conference on Artificial Intelligence, pp. 356\u2013361 (2006)"},{"key":"26_CR15","first-page":"1107","volume":"4","author":"MG Lagoudakis","year":"2003","unstructured":"Lagoudakis, M.G., Parr, R.: Least-squares policy iteration. J. Mach. Learn. Res. 4, 1107\u20131149 (2003)","journal-title":"J. Mach. Learn. Res."},{"issue":"2","key":"26_CR16","doi-asserted-by":"publisher","first-page":"357","DOI":"10.1109\/21.229449","volume":"23","author":"E Barnard","year":"1993","unstructured":"Barnard, E.: Temporal-difference methods and Markov models. IEEE Trans. Syst. Man Cybern. 23(2), 357\u2013365 (1993)","journal-title":"IEEE Trans. Syst. Man Cybern."},{"key":"26_CR17","unstructured":"Baird, III, L.C.: Reinforcement learning through gradient descent. Ph.D. thesis, Carnegie-Mellon University, May 1999"},{"key":"26_CR18","doi-asserted-by":"crossref","unstructured":"Sutton, R.S., Szepesv\u00e1ri, C., Maei, H.R.: A convergent o(n) algorithm for off-policy temporal-difference learning with linear function approximation. In: Proceedings of the Advances in Neural Information Processing Systems, pp. 1609\u20131616 (2008)","DOI":"10.1145\/1553374.1553501"},{"key":"26_CR19","first-page":"809","volume":"15","author":"C Dann","year":"2014","unstructured":"Dann, C., Neumann, G., Peters, J., et al.: Policy evaluation with temporal differences: a survey and comparison. J. Mach. Learn. Res. 15, 809\u2013883 (2014)","journal-title":"J. Mach. Learn. Res."},{"key":"26_CR20","unstructured":"Maei, H.R.: Gradient temporal-difference learning algorithms. Ph.D. thesis, University of Alberta, September 2011"},{"key":"26_CR21","doi-asserted-by":"crossref","unstructured":"Sutton, R.S., et al.: Fast gradient-descent methods for temporal-difference learning with linear function approximation. In: Proceedings of the 26th Annual International Conference on Machine Learning, pp. 993\u20131000 (2009)","DOI":"10.1145\/1553374.1553501"},{"issue":"357","key":"26_CR22","doi-asserted-by":"publisher","first-page":"77","DOI":"10.1080\/01621459.1977.10479910","volume":"72","author":"AP Dempster","year":"1977","unstructured":"Dempster, A.P., Schatzoff, M., Wermuth, N.: A simulation study of alternatives to ordinary least squares. J. Am. Stat. Assoc. 72(357), 77\u201391 (1977)","journal-title":"J. Am. Stat. Assoc."},{"issue":"1","key":"26_CR23","doi-asserted-by":"publisher","first-page":"79","DOI":"10.1023\/A:1022192903948","volume":"13","author":"A Nedi\u0106","year":"2003","unstructured":"Nedi\u0106, A., Bertsekas, D.P.: Least squares policy evaluation algorithms with linear function approximation. Discrete Event Dyn. S. 13(1), 79\u2013110 (2003)","journal-title":"Discrete Event Dyn. S."},{"issue":"9","key":"26_CR24","first-page":"54","volume":"11","author":"X Xu","year":"2005","unstructured":"Xu, X., Xie, T., Hu, D., Lu, X.: Kernel least-squares temporal difference learning. J. Inf. Technol. 11(9), 54\u201363 (2005)","journal-title":"J. Inf. Technol."},{"issue":"4","key":"26_CR25","doi-asserted-by":"publisher","first-page":"771","DOI":"10.1109\/TNNLS.2015.2424233","volume":"27","author":"T Song","year":"2015","unstructured":"Song, T., Li, D., Cao, L., Hirasawa, K.: Kernel-based least squares temporal difference with gradient correction. IEEE Trans. Neural Netw. 27(4), 771\u2013782 (2015)","journal-title":"IEEE Trans. Neural Netw."},{"key":"26_CR26","doi-asserted-by":"crossref","unstructured":"Xu, X.: A sparse kernel-based least-squares temporal difference algorithm for reinforcement learning. In: Proceedings of the International Conference on Natural Computation, pp. 47\u201356 (2006)","DOI":"10.1007\/11881070_8"},{"key":"26_CR27","unstructured":"Lagoudakis, M.G., Parr, R., et al.: Model-free least-squares policy iteration. In: Proceedings of the Conference and Workshop on Neural Information Processing Systems, pp. 345 (2001)"},{"key":"26_CR28","unstructured":"Maei, H.R., Szepesv\u00e1ri, C., Bhatnagar, S., Sutton, R.S.: Toward off-policy learning control with function approximation. In: Proceedings of the 27th International Conference on Machine Learning, pp. 719\u2013726 (2010)"},{"key":"26_CR29","doi-asserted-by":"crossref","unstructured":"Maei, H.R., Sutton, R.S.: GQ ($$\\lambda $$): a general gradient algorithm for temporal-difference prediction learning with eligibility traces. In: Proceedings of the Third Conference on Artificial General Intelligence, pp. 91\u201396 (2010)","DOI":"10.2991\/agi.2010.22"},{"key":"26_CR30","unstructured":"Brockman, G., et al.: OpenAI Gym. arXiv preprint arXiv:1606.01540 (2016)"}],"container-title":["Communications in Computer and Information Science","Exploration of Novel Intelligent Optimization Algorithms"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-19-4109-2_26","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,2,13]],"date-time":"2023-02-13T07:07:37Z","timestamp":1676272057000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-19-4109-2_26"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"ISBN":["9789811941085","9789811941092"],"references-count":30,"URL":"https:\/\/doi.org\/10.1007\/978-981-19-4109-2_26","relation":{},"ISSN":["1865-0929","1865-0937"],"issn-type":[{"type":"print","value":"1865-0929"},{"type":"electronic","value":"1865-0937"}],"subject":[],"published":{"date-parts":[[2022]]},"assertion":[{"value":"1 August 2022","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ISICA","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Symposium on Intelligence Computation and Applications","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Giangzhou","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2021","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20 November 2021","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"21 November 2021","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"12","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"isica2021","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/gdstinfo.scau.edu.cn\/isica2021\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Open","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"EasyChair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"99","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"48","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"48% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"No","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}