{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,28]],"date-time":"2026-02-28T22:39:20Z","timestamp":1772318360836,"version":"3.50.1"},"reference-count":35,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2024,11,27]],"date-time":"2024-11-27T00:00:00Z","timestamp":1732665600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,27]],"date-time":"2024-11-27T00:00:00Z","timestamp":1732665600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2025,1]]},"DOI":"10.1007\/s10489-024-05867-3","type":"journal-article","created":{"date-parts":[[2024,11,27]],"date-time":"2024-11-27T06:17:28Z","timestamp":1732688248000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Windows deep transformer Q-networks: an extended variance reduction architecture for partially observable reinforcement learning"],"prefix":"10.1007","volume":"55","author":[{"given":"Zijian","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0092-3951","authenticated-orcid":false,"given":"Bin","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hongbo","family":"Dou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhongyuan","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,11,27]]},"reference":[{"key":"5867_CR1","unstructured":"Esslinger K,\u00a0Platt R,\u00a0Amato C (2022) Deep transformer q-networks for partially observable reinforcement learning. arXiv:2206.01078"},{"key":"5867_CR2","doi-asserted-by":"publisher","unstructured":"Lee Y,\u00a0Cai P,\u00a0Hsu D (2021) MAGIC: learning macro-actions for online POMDP planning, in robotics: science and systems XVII, virtual event, July 12-16, 2021, ed. by Shell DA,\u00a0Toussaint M, Hsieh MA. https:\/\/doi.org\/10.15607\/RSS.2021.XVII.041","DOI":"10.15607\/RSS.2021.XVII.041"},{"key":"5867_CR3","doi-asserted-by":"publisher","unstructured":"Ogunfowora O, Najjaran H (2023) Reinforcement and deep reinforcement learning-based solutions for machine maintenance planning, scheduling policies, and optimization. J Manufac Syst 70:244\u2013263.https:\/\/doi.org\/10.1016\/j.jmsy.2023.07.014, https:\/\/www.sciencedirect.com\/science\/article\/pii\/S0278612523001462","DOI":"10.1016\/j.jmsy.2023.07.014"},{"key":"5867_CR4","doi-asserted-by":"publisher","unstructured":"Chen L, Jiang Z, Cheng L, Knoll AC, Zhou M (2022) Deep reinforcement learning based trajectory planning under uncertain constraints. Front Neurorobot 16:883562. https:\/\/doi.org\/10.3389\/FNBOT.2022.883562","DOI":"10.3389\/FNBOT.2022.883562"},{"key":"5867_CR5","doi-asserted-by":"publisher","unstructured":"Mnih V, Kavukcuoglu K, Silver D, Rusu AA, Veness J, Bellemare MG, Graves A, Riedmiller MA, Fidjeland A, Ostrovski G, Petersen S, Beattie C, Sadik A, Antonoglou I, King H, Kumaran D, Wierstra D, Legg S, Hassabis D (2015) Human-level control through deep reinforcement learning. Nat 518(7540):529\u2013533. https:\/\/doi.org\/10.1038\/NATURE14236","DOI":"10.1038\/NATURE14236"},{"key":"5867_CR6","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s10846-021-01348-8","volume":"101","author":"SA Serrano","year":"2021","unstructured":"Serrano SA, Santiago E, Martinez-Carranza J, Morales EF, Sucar LE (2021) Knowledge-based hierarchical pomdps for task planning. J Intell Robot Syst 101:1\u201330","journal-title":"J Intell Robot Syst"},{"key":"5867_CR7","first-page":"25502","volume":"34","author":"D Ghosh","year":"2021","unstructured":"Ghosh D, Rahme J, Kumar A, Zhang A, Adams RP, Levine S (2021) Why generalization in rl is difficult: epistemic pomdps and implicit partial observability. Adv Neural Inf Process Syst 34:25502\u201325515","journal-title":"Adv Neural Inf Process Syst"},{"key":"5867_CR8","unstructured":"Hausknecht M,\u00a0Stone P (2015) Deep recurrent q-learning for partially observable mdps. In: 2015 AAAI fall symposium series"},{"key":"5867_CR9","unstructured":"Sorokin I,\u00a0Seleznev A,\u00a0Pavlov M,\u00a0Fedorov A,\u00a0Ignateva A (2015) Deep attention recurrent q-network. arXiv:1512.01693"},{"key":"5867_CR10","doi-asserted-by":"crossref","unstructured":"Zhu P,\u00a0Li X,\u00a0Poupart P,\u00a0Miao G (2017) On improving deep reinforcement learning for pomdps. arXiv:1704.07978","DOI":"10.1007\/978-1-4899-7687-1_929"},{"key":"5867_CR11","unstructured":"Chen L,\u00a0Lu K,\u00a0Rajeswaran A,\u00a0Lee K,\u00a0Grover A,\u00a0Laskin M,\u00a0Abbeel P,\u00a0Srinivas A,\u00a0Mordatch I (2021) Decision transformer: reinforcement learning via sequence modeling, in advances in neural information processing systems 34: annual conference on neural information processing systems 2021, NeurIPS 2021, December 6-14, 2021, virtual, ed. by\u00a0Ranzato M,\u00a0Beygelzimer A, Dauphin YN,\u00a0Liang P, Vaughan JW, pp 15084\u201315097. https:\/\/proceedings.neurips.cc\/paper\/2021\/hash\/7f489f642a0ddb10272b5c31057f0663-Abstract.html"},{"key":"5867_CR12","unstructured":"Meng L,\u00a0Goodwin M,\u00a0Yazidi A,\u00a0Engelstad P (2022) Deep reinforcement learning with swin transformers. arXiv:2206.15269"},{"key":"5867_CR13","unstructured":"Vaswani A,\u00a0Shazeer N,\u00a0Parmar N,\u00a0Uszkoreit J,\u00a0Jones L, Gomez AN,\u00a0Kaiser \u0141,\u00a0Polosukhin I (2017) Attention is all you need. Adv Neural Inf Process Syst 30"},{"key":"5867_CR14","unstructured":"Lin L,\u00a0Bai Y,\u00a0Mei S (2023) Transformers as decision makers: provable in-context reinforcement learning via supervised pretraining. arXiv:2310.08566"},{"key":"5867_CR15","unstructured":"Ajay A,\u00a0Du Y,\u00a0Gupta A,\u00a0Tenenbaum J,\u00a0Jaakkola T,\u00a0Agrawal P (2022) Is conditional generative modeling all you need for decision-making? arXiv:2211.15657"},{"key":"5867_CR16","unstructured":"Chebotar Y,\u00a0Vuong Q,\u00a0Hausman K,\u00a0Xia F,\u00a0Lu Y,\u00a0Irpan A,\u00a0Kumar A,\u00a0Yu T,\u00a0Herzog A,\u00a0Pertsch K, et\u00a0al (2023) Q-transformer: scalable offline reinforcement learning via autoregressive q-functions. In: Conference on robot learning (PMLR), pp 3909\u20133928"},{"key":"5867_CR17","doi-asserted-by":"publisher","unstructured":"van Hasselt H,\u00a0Guez A,\u00a0Silver D (2016) Deep reinforcement learning with double Q-learning, in proceedings of the thirtieth AAAI conference on artificial intelligence, February 12-17, 2016, Phoenix, Arizona, USA, ed. by\u00a0Schuurmans D, Wellman MP (AAAI Press), pp 2094\u20132100. https:\/\/doi.org\/10.1609\/AAAI.V30I1.10295","DOI":"10.1609\/AAAI.V30I1.10295"},{"key":"5867_CR18","doi-asserted-by":"publisher","unstructured":"Lin T, Wang Y, Liu X, Qiu X (2022) A Surv Trans AI Open 3:111\u2013132. https:\/\/doi.org\/10.1016\/J.AIOPEN.2022.10.001","DOI":"10.1016\/J.AIOPEN.2022.10.001"},{"key":"5867_CR19","unstructured":"Anschel O,\u00a0Baram N,\u00a0Shimkin N (2017) Averaged-DQN: variance reduction and stabilization for deep reinforcement learning. In: Proceedings of the 34th international conference on machine learning, ICML 2017, Sydney, NSW, Australia, 6-11 August 2017, Proceedings of Machine Learning Research, vol.\u00a070, ed. by\u00a0Precup D, Teh YW (PMLR), pp 176\u2013185. http:\/\/proceedings.mlr.press\/v70\/anschel17a.html"},{"key":"5867_CR20","doi-asserted-by":"publisher","unstructured":"Ly A, Dazeley R, Vamplew P, Cruz F, Aryal S (2024) Elastic step DQN: a novel multi-step algorithm to alleviate overestimation in deep q-networks. Neurocomputing 576:127170. https:\/\/doi.org\/10.1016\/J.NEUCOM.2023.127170","DOI":"10.1016\/J.NEUCOM.2023.127170"},{"key":"5867_CR21","unstructured":"Wang Z,\u00a0Schaul T,\u00a0Hessel M,\u00a0Hasselt H,\u00a0Lanctot M,\u00a0Freitas N (2016) Dueling network architectures for deep reinforcement learning. In: International conference on machine learning (PMLR), pp 1995\u20132003"},{"key":"5867_CR22","unstructured":"Fujimoto S,\u00a0van Hoof H,\u00a0Meger D (2018) Addressing function approximation error in actor-critic methods. In: Proceedings of the 35th international conference on machine learning, ICML 2018, Stockholmsm\u00e4ssan, Stockholm, Sweden, July 10-15, 2018, Proceedings of Machine Learning Research, vol.\u00a080, ed. by Dy JG,\u00a0Krause A (PMLR), pp 1582\u20131591. http:\/\/proceedings.mlr.press\/v80\/fujimoto18a.html"},{"key":"5867_CR23","doi-asserted-by":"publisher","unstructured":"Hessel M,\u00a0Modayil J,\u00a0van Hasselt H,\u00a0Schaul T,\u00a0Ostrovski G,\u00a0Dabney W,\u00a0Horgan D,\u00a0Piot B, Azar MG,\u00a0Silver D (2018) Rainbow: combining improvements in deep reinforcement learning. In: Proceedings of the thirty-second AAAI conference on artificial intelligence, (AAAI-18), the 30th Innovative Applications of Artificial Intelligence (IAAI-18), and the 8th AAAI Symposium on Educational Advances in Artificial Intelligence (EAAI-18), New Orleans, Louisiana, USA, February 2-7, 2018, ed. by McIlraith SA, Weinberger KQ (AAAI Press), pp 3215\u20133222. https:\/\/doi.org\/10.1609\/AAAI.V32I1.11796","DOI":"10.1609\/AAAI.V32I1.11796"},{"key":"5867_CR24","unstructured":"Liang L,\u00a0Xu Y,\u00a0McAleer S,\u00a0Hu D,\u00a0Ihler A,\u00a0Abbeel P,\u00a0Fox R (2022) Reducing variance in temporal-difference value estimation via ensemble of deep networks. In: International conference on machine learning, ICML 2022, 17-23 July 2022, Baltimore, Maryland, USA, Proceedings of Machine Learning Research, vol. 162, ed. by\u00a0Chaudhuri K,\u00a0Jegelka S,\u00a0Song L,\u00a0Szepesv\u00e1ri C,\u00a0Niu G,\u00a0Sabato S (PMLR), pp 13285\u201313301. https:\/\/proceedings.mlr.press\/v162\/liang22c.html"},{"key":"5867_CR25","doi-asserted-by":"publisher","unstructured":"Kara AD,\u00a0Y\u00fcksel S (2021) Convergence and near optimality of Q-learning with finite memory for partially observed models. In: 2021 60th IEEE Conference on Decision and Control (CDC), pp 1603\u20131608. https:\/\/doi.org\/10.1109\/CDC45484.2021.9682777","DOI":"10.1109\/CDC45484.2021.9682777"},{"key":"5867_CR26","doi-asserted-by":"publisher","unstructured":"Tavanaei A, Ghodrati M, Kheradpisheh SR, Masquelier T, Maida A (2019) Deep learning in spiking neural networks. Neural Netw 111:47\u201363. https:\/\/doi.org\/10.1016\/J.NEUNET.2018.12.002","DOI":"10.1016\/J.NEUNET.2018.12.002"},{"issue":"11","key":"5867_CR27","doi-asserted-by":"publisher","first-page":"7187","DOI":"10.1109\/TCYB.2022.3198259","volume":"53","author":"G Liu","year":"2023","unstructured":"Liu G, Deng W, Xie X, Huang L, Tang H (2023) Human-level control through directly trained deep spiking q-networks. IEEE Trans Cybernet 53(11):7187\u20137198. https:\/\/doi.org\/10.1109\/TCYB.2022.3198259","journal-title":"IEEE Trans Cybernet"},{"key":"5867_CR28","doi-asserted-by":"publisher","unstructured":"Sun Y,\u00a0Zeng Y,\u00a0Li Y (2022) Solving the spike feature information vanishing problem in spiking deep Q network with potential based normalization. https:\/\/doi.org\/10.48550\/ARXIV.2206.03654, arXiv:2206.03654","DOI":"10.48550\/ARXIV.2206.03654"},{"key":"5867_CR29","unstructured":"Zheng Q,\u00a0Zhang A,\u00a0Grover A (2022) Online decision transformer. In: International conference on machine learning, ICML 2022, 17-23 July 2022, Baltimore, Maryland, USA, Proceedings of Machine Learning Research, vol. 162, ed. by\u00a0Chaudhuri K,\u00a0Jegelka S,\u00a0Song L,\u00a0Szepesv\u00e1ri C,\u00a0Niu G,\u00a0Sabato S (PMLR), pp 27042\u201327059. https:\/\/proceedings.mlr.press\/v162\/zheng22c.html"},{"key":"5867_CR30","unstructured":"Janner M,\u00a0Li Q,\u00a0Levine S (2021) Offline reinforcement learning as one big sequence modeling problem. In: Advances in neural information processing systems 34: Annual Conference on Neural Information Processing Systems 2021, NeurIPS 2021, December 6-14, 2021, virtual, ed. by\u00a0Ranzato M,\u00a0Beygelzimer A, Dauphin YN,\u00a0Liang P, Vaughan JW, pp 1273\u20131286. https:\/\/proceedings.neurips.cc\/paper\/2021\/hash\/099fe6b0b444c23836c4a5d07346082b-Abstract.html"},{"key":"5867_CR31","unstructured":"Baisero A,\u00a0Katt S (2021) gym-gridverse: Gridworld domains for fully and partially observable reinforcement learning. https:\/\/github.com\/abaisero\/gym-gridverse"},{"key":"5867_CR32","unstructured":"Fortunato M, Azar MG,\u00a0Piot B,\u00a0Menick J,\u00a0Osband I,\u00a0Graves A,\u00a0Mnih V,\u00a0Munos R,\u00a0Hassabis D,\u00a0Pietquin O,\u00a0Blundell C,\u00a0Legg S (2017) Noisy networks for exploration. arXiv:1706.10295. https:\/\/api.semanticscholar.org\/CorpusID:5176587"},{"issue":"1","key":"5867_CR33","doi-asserted-by":"publisher","first-page":"253","DOI":"10.1146\/annurev-control-042920-092451","volume":"5","author":"H Kurniawati","year":"2022","unstructured":"Kurniawati H (2022) Partially observable markov decision processes and robotics. Ann Rev Control, Robot, Autonom Syst 5(1):253\u2013277","journal-title":"Ann Rev Control, Robot, Autonom Syst"},{"key":"5867_CR34","unstructured":"Kumar A,\u00a0Zhou A,\u00a0Tucker G,\u00a0Levine S (2020) Conservative Q-learning for offline reinforcement learning, in advances in neural information processing systems 33: annual conference on neural information processing systems 2020, NeurIPS 2020, December 6-12, 2020, virtual, ed. by\u00a0Larochelle H,\u00a0Ranzato M,\u00a0Hadsell R,\u00a0Balcan M,\u00a0Lin H. https:\/\/proceedings.neurips.cc\/paper\/2020\/hash\/0d2b2061826a5df3221116a5085a6052-Abstract.html"},{"key":"5867_CR35","unstructured":"Fujimoto S, Gu SS (2021) A minimalist approach to offline reinforcement learning. In: Advances in neural information processing systems 34: annual conference on neural information processing systems 2021, NeurIPS 2021, December 6-14, 2021, virtual, ed. by\u00a0Ranzato M,\u00a0Beygelzimer A, Dauphin YN,\u00a0Liang P, Vaughan JW, pp 20132\u201320145. https:\/\/proceedings.neurips.cc\/paper\/2021\/hash\/a8166da05c5a094f7dc03724b41886e5-Abstract.html"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-024-05867-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10489-024-05867-3\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-024-05867-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,1,2]],"date-time":"2025-01-02T15:10:24Z","timestamp":1735830624000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10489-024-05867-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,27]]},"references-count":35,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2025,1]]}},"alternative-id":["5867"],"URL":"https:\/\/doi.org\/10.1007\/s10489-024-05867-3","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"value":"0924-669X","type":"print"},{"value":"1573-7497","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,27]]},"assertion":[{"value":"6 November 2024","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 November 2024","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that there is no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interest"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval"}},{"value":"Not applicable.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to participate"}},{"value":"Not applicable.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}}],"article-number":"35"}}