{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T10:55:03Z","timestamp":1776941703236,"version":"3.51.4"},"reference-count":40,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2024,3,26]],"date-time":"2024-03-26T00:00:00Z","timestamp":1711411200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,3,26]],"date-time":"2024-03-26T00:00:00Z","timestamp":1711411200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"US Department of Education","award":["ED#P116S210005"],"award-info":[{"award-number":["ED#P116S210005"]}]},{"name":"US Department of Education","award":["ED#P116S210005"],"award-info":[{"award-number":["ED#P116S210005"]}]},{"name":"National Science Foundation, United States","award":["#2226936"],"award-info":[{"award-number":["#2226936"]}]},{"name":"National Science Foundation, United States","award":["#2226936"],"award-info":[{"award-number":["#2226936"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Auton Agent Multi-Agent Syst"],"published-print":{"date-parts":[[2024,6]]},"DOI":"10.1007\/s10458-024-09641-0","type":"journal-article","created":{"date-parts":[[2024,3,26]],"date-time":"2024-03-26T04:04:21Z","timestamp":1711425861000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":10,"title":["Model-free reinforcement learning for motion planning of autonomous agents with complex tasks in partially observable environments"],"prefix":"10.1007","volume":"38","author":[{"given":"Junchao","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mingyu","family":"Cai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhen","family":"Kan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shaoping","family":"Xiao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,3,26]]},"reference":[{"key":"9641_CR1","doi-asserted-by":"publisher","DOI":"10.1146\/annurev-control-042920-092451","author":"H Kurniawati","year":"2022","unstructured":"Kurniawati, H. (2022). Partially observable Markov decision processes and robotics. Annual Review of Control, Robotics, and Autonomous Systems. https:\/\/doi.org\/10.1146\/annurev-control-042920-092451","journal-title":"Annual Review of Control, Robotics, and Autonomous Systems"},{"key":"9641_CR2","doi-asserted-by":"publisher","DOI":"10.2307\/2312519","author":"H Kaufman","year":"1961","unstructured":"Kaufman, H., & Howard, R. A. (1961). Dynamic programming and Markov processes. The American Mathematical Monthly. https:\/\/doi.org\/10.2307\/2312519","journal-title":"The American Mathematical Monthly"},{"key":"9641_CR3","doi-asserted-by":"publisher","unstructured":"Cai, M., Xiao, S., Li, B., Li, Z., & Kan, Z. (2021). Reinforcement learning based temporal logic control with maximum probabilistic satisfaction (Vol. 2021). https:\/\/doi.org\/10.1109\/ICRA48506.2021.9561903","DOI":"10.1109\/ICRA48506.2021.9561903"},{"key":"9641_CR4","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.2021.3138704","author":"M Cai","year":"2021","unstructured":"Cai, M., Xiao, S., Li, Z., & Kan, Z. (2021). Optimal probabilistic motion planning with potential infeasible ltl constraints. IEEE Transactions on Automatic Control. https:\/\/doi.org\/10.1109\/TAC.2021.3138704","journal-title":"IEEE Transactions on Automatic Control"},{"key":"9641_CR5","doi-asserted-by":"crossref","unstructured":"Perez, A., Platt, R., Konidaris, G., Kaelbling, L., & Lozano-Perez, T. (2012). Lqr-rrt*: Optimal sampling-based motion planning with automatically derived extension heuristics. In 2012 IEEE International Conference on Robotics and Automation (pp. 2537\u20132542). IEEE.","DOI":"10.1109\/ICRA.2012.6225177"},{"key":"9641_CR6","volume-title":"Reinforcement learning: An introduction","author":"RS Sutton","year":"2018","unstructured":"Sutton, R. S., & Barto, A. G. (2018). Reinforcement learning: An introduction. MIT Press."},{"key":"9641_CR7","doi-asserted-by":"publisher","DOI":"10.1007\/s10458-012-9200-2","author":"G Shani","year":"2013","unstructured":"Shani, G., Pineau, J., & Kaplow, R. (2013). A survey of point-based pomdp solvers. Autonomous Agents and Multi-Agent Systems. https:\/\/doi.org\/10.1007\/s10458-012-9200-2","journal-title":"Autonomous Agents and Multi-Agent Systems"},{"key":"9641_CR8","unstructured":"Pineau, J., Gordon, G., & Thrun, S. (2003). Point-based value iteration: An anytime algorithm for pomdps."},{"key":"9641_CR9","doi-asserted-by":"publisher","unstructured":"Sarsop: Efficient point-based pomdp planning by approximating optimally reachable belief spaces (Vol. 4) (2009). https:\/\/doi.org\/10.15607\/rss.2008.iv.009","DOI":"10.15607\/rss.2008.iv.009"},{"issue":"14","key":"9641_CR10","doi-asserted-by":"publisher","first-page":"871","DOI":"10.1080\/01691864.2023.2226191","volume":"37","author":"J Li","year":"2023","unstructured":"Li, J., Cai, M., Wang, Z., & Xiao, S. (2023). Model-based motion planning in pomdps with temporal logic specifications. Advanced Robotics, 37(14), 871\u2013886.","journal-title":"Advanced Robotics"},{"key":"9641_CR11","unstructured":"Mnih, V., Silver, D., & Riedmiller, M. (2013). Playing atari with deep q learning. Nips."},{"key":"9641_CR12","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236","author":"V Mnih","year":"2015","unstructured":"Mnih, V., Kavukcuoglu, K., Silver, D., Rusu, A. A., Veness, J., Bellemare, M. G., Graves, A., Riedmiller, M., Fidjeland, A. K., Ostrovski, G., Petersen, S., Beattie, C., Sadik, A., Antonoglou, I., King, H., Kumaran, D., Wierstra, D., Legg, S., & Hassabis, D. (2015). Human-level control through deep reinforcement learning. Nature. https:\/\/doi.org\/10.1038\/nature14236","journal-title":"Nature"},{"key":"9641_CR13","doi-asserted-by":"publisher","DOI":"10.1145\/3065386","author":"A Krizhevsky","year":"2017","unstructured":"Krizhevsky, A., Sutskever, I., & Hinton, G. E. (2017). Imagenet classification with deep convolutional neural networks. Communications of the ACM. https:\/\/doi.org\/10.1145\/3065386","journal-title":"Communications of the ACM"},{"key":"9641_CR14","unstructured":"Hausknecht, M., & Stone, P. (2015). Deep recurrent q-learning for partially observable mdps (Vol. FS-15-06)."},{"key":"9641_CR15","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter, S., & Schmidhuber, J. (1997). Long short-term memory. Neural Computation. https:\/\/doi.org\/10.1162\/neco.1997.9.8.1735","journal-title":"Neural Computation"},{"key":"9641_CR16","unstructured":"Foerster, J. N.,\u00a0Assael, Y. M., de Freitas, N., & Whiteson, S. (2016). Learning to communicate to solve riddles with deep distributed recurrent q-networks."},{"key":"9641_CR17","unstructured":"Zhu, P., Li, X., Poupart, P., & Miao, G. (2017). On improving deep reinforcement learning for pomdps."},{"key":"9641_CR18","unstructured":"Heess, N., Hunt, J. J., Lillicrap, T. P., & Silver, D. (2015). Memory-based control with recurrent neural networks."},{"key":"9641_CR19","doi-asserted-by":"publisher","unstructured":"Meng, L., Gorbet, R., & Kulic, D. (2021). Memory-based deep reinforcement learning for pomdps. https:\/\/doi.org\/10.1109\/IROS51168.2021.9636140","DOI":"10.1109\/IROS51168.2021.9636140"},{"key":"9641_CR20","unstructured":"0 Baier, C., & Katoen, J.-P. (2008). Principles of model checking (Vol. 950)."},{"key":"9641_CR21","doi-asserted-by":"publisher","unstructured":"K\u0159et\u00ednsk\u00fd, J., Meggendorfer, T., & Sickert, S. (2018). Owl: A library for $$\\omega $$-words, automata, and ltl, vol. 11138 LNCS. https:\/\/doi.org\/10.1007\/978-3-030-01090-4_34","DOI":"10.1007\/978-3-030-01090-4_34"},{"key":"9641_CR22","doi-asserted-by":"publisher","unstructured":"Chatterjee, K., Chmel\u00edk, M., Gupta, R., & Kanodia, A. (2015). Qualitative analysis of pomdps with temporal logic specifications for robotics applications, vol. 2015-June. https:\/\/doi.org\/10.1109\/ICRA.2015.7139019","DOI":"10.1109\/ICRA.2015.7139019"},{"key":"9641_CR23","unstructured":"Icarte, R. T., Waldie, E., Klassen, T. Q., Valenzano, R., Castro, M. P., & McIlraith, S. A. (2019). Learning reward machines for partially observable reinforcement learning, vol. 32."},{"key":"9641_CR24","doi-asserted-by":"publisher","first-page":"173","DOI":"10.1613\/jair.1.12440","volume":"73","author":"RT Icarte","year":"2022","unstructured":"Icarte, R. T., Klassen, T. Q., Valenzano, R., & McIlraith, S. A. (2022). Reward machines: Exploiting reward function structure in reinforcement learning. Journal of Artificial Intelligence Research, 73, 173\u2013208.","journal-title":"Journal of Artificial Intelligence Research"},{"key":"9641_CR25","doi-asserted-by":"publisher","unstructured":"Sharan, R., & Burdick, J. (2014). Finite state control of pomdps with ltl specifications. https:\/\/doi.org\/10.1109\/ACC.2014.6858909","DOI":"10.1109\/ACC.2014.6858909"},{"key":"9641_CR26","doi-asserted-by":"publisher","unstructured":"Bouton, M., & Kochenderfer, M. J. (2020). Point-based methods for model checking in partially observable Markov decision processes. https:\/\/doi.org\/10.1609\/aaai.v34i06.6563","DOI":"10.1609\/aaai.v34i06.6563"},{"key":"9641_CR27","unstructured":"Ahmadi, M., Sharan, R., & Burdick, J. W. (2020). Stochastic finite state control of pomdps with ltl specifications."},{"key":"9641_CR28","doi-asserted-by":"publisher","unstructured":"Carr, S., Jansen, N., Wimmer, R., Serban, A., Becker, B., & Topcu, U. (2019). Counterexample-guided strategy improvement for pomdps using recurrent neural networks (vol. 2019). https:\/\/doi.org\/10.24963\/ijcai.2019\/768.","DOI":"10.24963\/ijcai.2019\/768"},{"key":"9641_CR29","doi-asserted-by":"publisher","unstructured":"Carr, S., Jansen, N., & Topcu, U. (2020) Verifiable rnn-based policies for pomdps under temporal logic constraints (Vol. 2021). https:\/\/doi.org\/10.24963\/ijcai.2020\/570.","DOI":"10.24963\/ijcai.2020\/570"},{"key":"9641_CR30","doi-asserted-by":"publisher","unstructured":"Hahn, E. M., Perez, M., Schewe, S., Somenzi, F., Trivedi, A., & Wojtczak, D. (2019). Omega-regular objectives in model-free reinforcement learning (Vol. 11427). LNCS. https:\/\/doi.org\/10.1007\/978-3-030-17462-0_27","DOI":"10.1007\/978-3-030-17462-0_27"},{"key":"9641_CR31","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2021.3101544","author":"M Cai","year":"2021","unstructured":"Cai, M., Hasanbeig, M., Xiao, S., Abate, A., & Kan, Z. (2021). Modular deep reinforcement learning for continuous motion planning with temporal logic. IEEE Robotics and Automation Letters. https:\/\/doi.org\/10.1109\/LRA.2021.3101544","journal-title":"IEEE Robotics and Automation Letters"},{"key":"9641_CR32","doi-asserted-by":"publisher","unstructured":"Hasanbeig, M., Kantaros, Y., Abate, A., Kroening, D., Pappas, G. J., & Lee, I. (2019). Reinforcement learning for temporal logic control synthesis with probabilistic satisfaction guarantees (Vol. 2019). https:\/\/doi.org\/10.1109\/CDC40024.2019.9028919","DOI":"10.1109\/CDC40024.2019.9028919"},{"key":"9641_CR33","doi-asserted-by":"publisher","DOI":"10.1016\/j.artint.2023.103949","volume":"322","author":"H Hasanbeig","year":"2023","unstructured":"Hasanbeig, H., Kroening, D., & Abate, A. (2023). Certified reinforcement learning with logic guidance. Artificial Intelligence, 322, 103949.","journal-title":"Artificial Intelligence"},{"key":"9641_CR34","doi-asserted-by":"publisher","DOI":"10.1109\/LCSYS.2020.2980552","author":"R Oura","year":"2020","unstructured":"Oura, R., Sakakibara, A., & Ushio, T. (2020). Reinforcement learning of control policy for linear temporal logic specifications using limit-deterministic generalized b\u00fcchi automata. IEEE Control Systems Letters. https:\/\/doi.org\/10.1109\/LCSYS.2020.2980552","journal-title":"IEEE Control Systems Letters"},{"key":"9641_CR35","doi-asserted-by":"publisher","DOI":"10.1007\/bf00992699","author":"L-J Lin","year":"1992","unstructured":"Lin, L.-J. (1992). Self-improving reactive agents based on reinforcement learning, planning and teaching. Machine Learning. https:\/\/doi.org\/10.1007\/bf00992699","journal-title":"Machine Learning"},{"key":"9641_CR36","doi-asserted-by":"publisher","unstructured":"Bozkurt, A. K., Wang, Y., Zavlanos, M. M., & Pajic, M. (2020). Control synthesis from linear temporal logic specifications using model-free reinforcement learning. https:\/\/doi.org\/10.1109\/ICRA40945.2020.9196796","DOI":"10.1109\/ICRA40945.2020.9196796"},{"key":"9641_CR37","doi-asserted-by":"publisher","unstructured":"Sickert, S., Esparza, J., Jaax, S., & K\u0159et\u00ednsk\u00fd, J. (2016). Limit-deterministic b\u00fcchi automata for linear temporal logic (Vol. 9780). https:\/\/doi.org\/10.1007\/978-3-319-41540-6_17.","DOI":"10.1007\/978-3-319-41540-6_17"},{"key":"9641_CR38","unstructured":"Coumans, E., & Bai, Y. PyBullet, a Python module for physics simulation for games, robotics and machine learning. http:\/\/pybullet.org (2016\u20132021)"},{"key":"9641_CR39","doi-asserted-by":"publisher","DOI":"10.1007\/s10489-022-04105-y","author":"A Oroojlooy","year":"2022","unstructured":"Oroojlooy, A., & Hajinezhad, D. (2022). A review of cooperative multi-agent deep reinforcement learning. Applied Intelligence. https:\/\/doi.org\/10.1007\/s10489-022-04105-y","journal-title":"Applied Intelligence"},{"issue":"11","key":"9641_CR40","doi-asserted-by":"publisher","first-page":"339","DOI":"10.3390\/drones6110339","volume":"6","author":"W Zhou","year":"2022","unstructured":"Zhou, W., Li, J., & Zhang, Q. (2022). Joint communication and action learning in multi-target tracking of uav swarms with deep reinforcement learning. Drones, 6(11), 339.","journal-title":"Drones"}],"container-title":["Autonomous Agents and Multi-Agent Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10458-024-09641-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10458-024-09641-0\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10458-024-09641-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,7,1]],"date-time":"2024-07-01T19:05:18Z","timestamp":1719860718000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10458-024-09641-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,3,26]]},"references-count":40,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2024,6]]}},"alternative-id":["9641"],"URL":"https:\/\/doi.org\/10.1007\/s10458-024-09641-0","relation":{"has-preprint":[{"id-type":"doi","id":"10.21203\/rs.3.rs-2856026\/v1","asserted-by":"object"}]},"ISSN":["1387-2532","1573-7454"],"issn-type":[{"value":"1387-2532","type":"print"},{"value":"1573-7454","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,3,26]]},"assertion":[{"value":"29 February 2024","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 March 2024","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"14"}}