{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T18:55:57Z","timestamp":1775069757301,"version":"3.50.1"},"reference-count":53,"publisher":"Springer Science and Business Media LLC","issue":"12","license":[{"start":{"date-parts":[[2024,8,12]],"date-time":"2024-08-12T00:00:00Z","timestamp":1723420800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,8,12]],"date-time":"2024-08-12T00:00:00Z","timestamp":1723420800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Knowl Inf Syst"],"published-print":{"date-parts":[[2024,12]]},"DOI":"10.1007\/s10115-024-02190-8","type":"journal-article","created":{"date-parts":[[2024,8,12]],"date-time":"2024-08-12T05:02:07Z","timestamp":1723438927000},"page":"7389-7417","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["Adaptive moving average Q-learning"],"prefix":"10.1007","volume":"66","author":[{"given":"Tao","family":"Tan","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hong","family":"Xie","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yunni","family":"Xia","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaoyu","family":"Shi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mingsheng","family":"Shang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,8,12]]},"reference":[{"issue":"8","key":"2190_CR1","doi-asserted-by":"publisher","first-page":"2059","DOI":"10.1007\/s10115-022-01696-3","volume":"64","author":"K Ali","year":"2022","unstructured":"Ali K, Wang C-Y, Chen Y-S (2022) Leveraging transfer learning in reinforcement learning to tackle competitive influence maximization. Knowl Inf Syst 64(8):2059\u20132090","journal-title":"Knowl Inf Syst"},{"key":"2190_CR2","doi-asserted-by":"publisher","first-page":"911","DOI":"10.1007\/s10115-016-0992-2","volume":"51","author":"J Garc\u00eda","year":"2017","unstructured":"Garc\u00eda J, Iglesias R, Rodr\u00edguez MA, Regueiro CV (2017) Incremental reinforcement learning for multi-objective robotic tasks. Knowl Inf Syst 51:911\u2013940","journal-title":"Knowl Inf Syst"},{"key":"2190_CR3","doi-asserted-by":"publisher","first-page":"2479","DOI":"10.1007\/s10115-021-01590-4","volume":"63","author":"C Li","year":"2021","unstructured":"Li C, Zhang Y, Luo Y (2021) Deep reinforcement learning-based resource allocation and seamless handover in multi-access edge computing based on sdn. Knowl Inf Syst 63:2479\u20132511","journal-title":"Knowl Inf Syst"},{"issue":"8","key":"2190_CR4","doi-asserted-by":"publisher","first-page":"2239","DOI":"10.1007\/s10115-022-01711-7","volume":"64","author":"Z Liu","year":"2022","unstructured":"Liu Z, Ma Y, Hildebrandt M, Ouyang Y, Xiong Z (2022) Cdarl: a contrastive discriminator-augmented reinforcement learning framework for sequential recommendations. Knowl Inf Syst 64(8):2239\u20132265","journal-title":"Knowl Inf Syst"},{"key":"2190_CR5","doi-asserted-by":"publisher","first-page":"603","DOI":"10.1007\/s10115-018-1175-0","volume":"57","author":"HC Neto","year":"2018","unstructured":"Neto HC, Julia RMS (2018) Ace-rl-checkers: decision-making adaptability through integration of automatic case elicitation, reinforcement learning, and sequential pattern mining. Knowl Inf Syst 57:603\u2013634","journal-title":"Knowl Inf Syst"},{"issue":"1","key":"2190_CR6","doi-asserted-by":"publisher","first-page":"409","DOI":"10.1007\/s10115-022-01746-w","volume":"65","author":"G Saranya","year":"2023","unstructured":"Saranya G, Sasikala E (2023) An efficient computational offloading framework using HAA optimization-based deep reinforcement learning in edge-based cloud computing architecture. Knowl Inf Syst 65(1):409\u2013433","journal-title":"Knowl Inf Syst"},{"issue":"4","key":"2190_CR7","doi-asserted-by":"publisher","first-page":"1611","DOI":"10.1007\/s10115-022-01804-3","volume":"65","author":"Z Xiao","year":"2023","unstructured":"Xiao Z, Zhang D (2023) A deep reinforcement learning agent for geometry online tutoring. Knowl Inf Syst 65(4):1611\u20131625","journal-title":"Knowl Inf Syst"},{"issue":"9","key":"2190_CR8","doi-asserted-by":"publisher","first-page":"2515","DOI":"10.1007\/s10115-022-01713-5","volume":"64","author":"SG Rizzo","year":"2022","unstructured":"Rizzo SG, Chen Y, Pang L, Lucas J, Kaoudi Z, Quiane J, Chawla S (2022) Uncertainty-bounded reinforcement learning for revenue optimization in air cargo: a prescriptive learning approach. Knowl Inf Syst 64(9):2515\u20132541","journal-title":"Knowl Inf Syst"},{"key":"2190_CR9","doi-asserted-by":"publisher","first-page":"557","DOI":"10.1146\/annurev-statistics-040220-090158","volume":"9","author":"GL Jones","year":"2022","unstructured":"Jones GL, Qin Q (2022) Markov chain Monte Carlo in practice. Annu Rev Stat Appl 9:557\u2013578","journal-title":"Annu Rev Stat Appl"},{"issue":"1","key":"2190_CR10","first-page":"6918","volume":"23","author":"Y Jia","year":"2022","unstructured":"Jia Y, Zhou XY (2022) Policy evaluation and temporal-difference learning in continuous time and space: a martingale approach. J Mach Learn Res 23(1):6918\u20136972","journal-title":"J Mach Learn Res"},{"key":"2190_CR11","doi-asserted-by":"crossref","unstructured":"Zhang L, Zhang Q, Shen L, Yuan B, Wang X, Tao D (2023) Evaluating model-free reinforcement learning toward safety-critical tasks. In: Proceedings of the AAAI conference on artificial intelligence, vol 37, pp 15313\u201315321","DOI":"10.1609\/aaai.v37i12.26786"},{"key":"2190_CR12","unstructured":"Watkins, CJCH (1989) Learning from delayed rewards. King\u2019s College, Cambridge, United Kingdom"},{"key":"2190_CR13","unstructured":"Kearns M, Singh S (1999) Finite-sample convergence rates for Qlearning and indirect algorithms. Adv Neural Inf Process Syst 11"},{"issue":"1","key":"2190_CR14","first-page":"8674512","volume":"2020","author":"Y Yang","year":"2020","unstructured":"Yang Y, Wang X, Xu Y, Huang Q (2020) Multiagent reinforcement learning-based taxi predispatching model to balance taxi supply and demand. J Adv Transp 2020(1):8674512","journal-title":"J Adv Transp"},{"issue":"1","key":"2190_CR15","first-page":"36","volume":"15","author":"JW Mock","year":"2023","unstructured":"Mock JW, Muknahallipatna SS (2023) A comparison of ppo, td3 and sac reinforcement algorithms for quadruped walking gait generation. J Intell Learn Syst Appl 15(1):36\u201356","journal-title":"J Intell Learn Syst Appl"},{"issue":"1","key":"2190_CR16","doi-asserted-by":"publisher","first-page":"976","DOI":"10.1109\/TASE.2023.3234961","volume":"21","author":"B Wang","year":"2023","unstructured":"Wang B, Li X, Chen Y, Wu J, Zeng B, Chen J (2023) Continuous control with swarm intelligence based value function approximation. IEEE Trans Autom Sci Eng 21(1):976\u2013988","journal-title":"IEEE Trans Autom Sci Eng"},{"key":"2190_CR17","unstructured":"Upadhyay I (2021) Analysis of Q-learning based game playing agents for abstract board games with increasing state-space complexity. PhD thesis, Miami University"},{"key":"2190_CR18","unstructured":"Thrun S, Schwartz A (1993) Issues in using function approximation for reinforcement learning. In: Proceedings of the fourth connectionist models summer school, Hillsdale, NJ, pp 255\u2013263"},{"key":"2190_CR19","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2022.109998","volume":"258","author":"B Wang","year":"2022","unstructured":"Wang B, Wu J, Li X, Shen J, Zhong Y (2022) Uncertainty quantification for operators in online reinforcement learning. Knowl Based Syst 258:109998","journal-title":"Knowl Based Syst"},{"issue":"2","key":"2190_CR20","doi-asserted-by":"publisher","first-page":"308","DOI":"10.1287\/mnsc.1060.0614","volume":"53","author":"S Mannor","year":"2007","unstructured":"Mannor S, Simester D, Sun P, Tsitsiklis JN (2007) Bias and variance approximation in value function estimates. Manag Sci 53(2):308\u2013322","journal-title":"Manag Sci"},{"key":"2190_CR21","first-page":"2613","volume":"23","author":"H Hasselt","year":"2010","unstructured":"Hasselt H (2010) Double Q-learning. Adv Neural Inf Process Syst 23:2613\u20132621","journal-title":"Adv Neural Inf Process Syst"},{"key":"2190_CR22","unstructured":"Anschel O, Baram N, Shimkin N (2017) Averaged-dqn: variance reduction and stabilization for deep reinforcement learning. In: International conference on machine learning. PMLR, pp 176\u2013185"},{"key":"2190_CR23","doi-asserted-by":"crossref","unstructured":"Zhang Z, Pan Z, Kochenderfer MJ (2017) Weighted double Q-learning. In: IJCAI, pp 3455\u20133461","DOI":"10.24963\/ijcai.2017\/483"},{"key":"2190_CR24","unstructured":"Song Z, Parr R, Carin L (2019) Revisiting the softmax bellman operator: new benefits and new perspective. In: International conference on machine learning. PMLR, pp 5916\u20135925"},{"key":"2190_CR25","unstructured":"Lan Q, Pan Y, Fyshe A, White M (2020) Maxmin Q-learning: controlling the estimation bias of Q-learning. arXiv preprint arXiv:2002.06487"},{"key":"2190_CR26","doi-asserted-by":"crossref","unstructured":"Zhu R, Rigotti M (2021) Self-correcting Q-learning. In: Proceedings of the AAAI conference on artificial intelligence, vol 35, pp 11185\u201311192","DOI":"10.1609\/aaai.v35i12.17334"},{"key":"2190_CR27","doi-asserted-by":"crossref","unstructured":"Cetin E, Celiktutan O (2023) Learning pessimism for reinforcement learning. In: Proceedings of the AAAI conference on artificial intelligence, vol 37, pp 6971\u20136979","DOI":"10.1609\/aaai.v37i6.25852"},{"key":"2190_CR28","first-page":"10246","volume":"34","author":"Z Ren","year":"2021","unstructured":"Ren Z, Zhu G, Hu H, Han B, Chen J, Zhang C (2021) On the estimation bias in double Q-learning. Adv Neural Inf Process Syst 34:10246\u201310259","journal-title":"Adv Neural Inf Process Syst"},{"key":"2190_CR29","first-page":"7242","volume":"34","author":"L Zhao","year":"2021","unstructured":"Zhao L, Xiong H, Liang Y (2021) Faster non-asymptotic convergence for double Q-learning. Adv Neural Inf Process Syst 34:7242\u20137253","journal-title":"Adv Neural Inf Process Syst"},{"key":"2190_CR30","doi-asserted-by":"crossref","unstructured":"Lee D, Defourny B, Powell WB (2013) Bias-corrected Q-learning to control max-operator bias in Q-learning. In: 2013 IEEE symposium on adaptive dynamic programming and reinforcement learning (ADPRL). IEEE, pp 93\u201399","DOI":"10.1109\/ADPRL.2013.6614994"},{"key":"2190_CR31","unstructured":"D\u00a0Eramo C, Restelli M, Nuara A (2016) Estimating maximum expected value through gaussian approximation. In: International conference on machine learning. PMLR, pp 1032\u20131040"},{"key":"2190_CR32","doi-asserted-by":"crossref","unstructured":"Li J, Kuang K, Wang B, Liu F, Chen L, Fan C, Wu F, Xiao J (2022) Deconfounded value decomposition for multi-agent reinforcement learning. In: International conference on machine learning. PMLR, pp 12843\u201312856","DOI":"10.1145\/3447548.3467420"},{"key":"2190_CR33","unstructured":"Mao W, Yang L, Zhang K, Basar T (2022) On improving model-free algorithms for decentralized multi-agent reinforcement learning. In: International conference on machine learning. PMLR, pp 15007\u201315049"},{"key":"2190_CR34","first-page":"1365","volume":"34","author":"L Pan","year":"2021","unstructured":"Pan L, Rashid T, Peng B, Huang L, Whiteson S (2021) Regularized softmax deep multi-agent Q-learning. Adv Neural Inf Process Syst 34:1365\u20131377","journal-title":"Adv Neural Inf Process Syst"},{"key":"2190_CR35","first-page":"3680","volume":"34","author":"N Hansen","year":"2021","unstructured":"Hansen N, Su H, Wang X (2021) Stabilizing deep Q-learning with convnets and vision transformers under data augmentation. Adv Neural Inf Process Syst 34:3680\u20133693","journal-title":"Adv Neural Inf Process Syst"},{"key":"2190_CR36","first-page":"24778","volume":"34","author":"H Wang","year":"2021","unstructured":"Wang H, Lin S, Zhang J (2021) Adaptive ensemble Q-learning: minimizing estimation bias via error feedback. Adv Neural Inf Process Syst 34:24778\u201324790","journal-title":"Adv Neural Inf Process Syst"},{"key":"2190_CR37","unstructured":"Chen L, Jain R, Luo H (2022) Learning infinite-horizon average-reward Markov decision process with constraints. In: International conference on machine learning. PMLR, pp 3246\u20133270"},{"key":"2190_CR38","doi-asserted-by":"publisher","first-page":"57369","DOI":"10.1109\/ACCESS.2022.3178194","volume":"10","author":"H-T Joo","year":"2022","unstructured":"Joo H-T, Baek I-C, Kim K-J (2022) A swapping target Q-value technique for data augmentation in offline reinforcement learning. IEEE Access 10:57369\u201357382","journal-title":"IEEE Access"},{"key":"2190_CR39","unstructured":"Littman ML, Szepesv\u00e1ri C (1996) A generalized reinforcement-learning model: convergence and applications. In: ICML, vol 96. Citeseer, pp 310\u2013318"},{"key":"2190_CR40","unstructured":"Dai B, Shaw A, Li L, Xiao L, He N, Liu Z, Chen J, Song L (2018) Sbeed: convergent reinforcement learning with nonlinear function approximation. In: International conference on machine learning. PMLR, pp 1125\u20131134"},{"key":"2190_CR41","doi-asserted-by":"crossref","unstructured":"Bertsekas DP, Tsitsiklis JN (1995) Neuro-dynamic programming: an overview. In: Proceedings of 1995 34th IEEE conference on decision and control, vol 1. IEEE, pp 560\u2013564","DOI":"10.1109\/CDC.1995.478953"},{"issue":"1","key":"2190_CR42","first-page":"89","volume":"16","author":"DB Ishwaei","year":"1985","unstructured":"Ishwaei DB, Shabma D, Krishnamoorthy K (1985) Non-existence of unbiased estimators of ordered parameters. Stat J Theor Appl Stat 16(1):89\u201395","journal-title":"Stat J Theor Appl Stat"},{"issue":"2","key":"2190_CR43","doi-asserted-by":"publisher","first-page":"177","DOI":"10.1017\/S0004972712001098","volume":"87","author":"SS Dragomir","year":"2013","unstructured":"Dragomir SS (2013) Some reverses of the Jensen inequality with applications. Bull Aust Math Soc 87(2):177\u2013194","journal-title":"Bull Aust Math Soc"},{"key":"2190_CR44","unstructured":"Van\u00a0Hasselt H (2013) Estimating the maximum expected value: an analysis of (nested) cross validation and the maximum sample average. arXiv preprint arXiv:1302.7175"},{"issue":"1","key":"2190_CR45","doi-asserted-by":"publisher","first-page":"145","DOI":"10.1016\/S0893-6080(98)00116-6","volume":"12","author":"N Qian","year":"1999","unstructured":"Qian N (1999) On the momentum term in gradient descent learning algorithms. Neural Netw 12(1):145\u2013151","journal-title":"Neural Netw"},{"key":"2190_CR46","doi-asserted-by":"crossref","unstructured":"Dabney W, Barreto A, Rowland M, Dadashi R, Quan J, Bellemare MG, Silver D (2021) The value-improvement path: towards better representations for reinforcement learning. In: Proceedings of the AAAI conference on artificial intelligence, vol 35, pp 7160\u20137168","DOI":"10.1609\/aaai.v35i8.16880"},{"key":"2190_CR47","unstructured":"Buckman J, Hafner D, Tucker G, Brevdo E, Lee H (2018) Sample-efficient reinforcement learning with stochastic ensemble value expansion. Adv Neural Inf Process Syst 31"},{"issue":"7540","key":"2190_CR48","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih V, Kavukcuoglu K, Silver D, Rusu AA, Veness J, Bellemare MG, Graves A, Riedmiller M, Fidjeland AK, Ostrovski G et al (2015) Human-level control through deep reinforcement learning. Nature 518(7540):529\u2013533","journal-title":"Nature"},{"key":"2190_CR49","doi-asserted-by":"crossref","unstructured":"Van\u00a0Hasselt H, Guez A, Silver D (2016) Deep reinforcement learning with double Q-learning. In: Proceedings of the AAAI conference on artificial intelligence, vol 30","DOI":"10.1609\/aaai.v30i1.10295"},{"issue":"10","key":"2190_CR50","doi-asserted-by":"publisher","first-page":"4011","DOI":"10.1109\/TAC.2019.2912443","volume":"64","author":"D Lee","year":"2019","unstructured":"Lee D, Powell WB (2019) Bias-corrected Q-learning with multistate extension. IEEE Trans Autom Control 64(10):4011\u20134023","journal-title":"IEEE Trans Autom Control"},{"key":"2190_CR51","unstructured":"Sutton RS, Barto AG (2018) Reinforcement learning: an introduction. MIT press"},{"key":"2190_CR52","doi-asserted-by":"crossref","unstructured":"Urtans E, Nikitenko A (2018) Survey of deep q-network variants in pygame learning environment. In: Proceedings of the 2018 2nd international conference on deep learning technologies, pp 27\u201336","DOI":"10.1145\/3234804.3234816"},{"key":"2190_CR53","unstructured":"Young K, Tian T (2019) Minatar: an atari-inspired testbed for more efficient reinforcement learning experiments (2019). arXiv preprint arXiv:1903.03176"}],"container-title":["Knowledge and Information Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10115-024-02190-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10115-024-02190-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10115-024-02190-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,24]],"date-time":"2024-10-24T12:06:21Z","timestamp":1729771581000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10115-024-02190-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,8,12]]},"references-count":53,"journal-issue":{"issue":"12","published-print":{"date-parts":[[2024,12]]}},"alternative-id":["2190"],"URL":"https:\/\/doi.org\/10.1007\/s10115-024-02190-8","relation":{},"ISSN":["0219-1377","0219-3116"],"issn-type":[{"value":"0219-1377","type":"print"},{"value":"0219-3116","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,8,12]]},"assertion":[{"value":"20 September 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 July 2024","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 July 2024","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 August 2024","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"This paper has a conflict of interest with editors or reviewers from Chongqing University with the email domain cqu.edu.cn. This paper has no conflict of interest with the current editorial board of Knowledge and Information Systems.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}