{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,16]],"date-time":"2026-02-16T09:15:59Z","timestamp":1771233359094,"version":"3.50.1"},"reference-count":56,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2025,12,4]],"date-time":"2025-12-04T00:00:00Z","timestamp":1764806400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"},{"start":{"date-parts":[[2026,1,6]],"date-time":"2026-01-06T00:00:00Z","timestamp":1767657600000},"content-version":"vor","delay-in-days":33,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Complex Intell. Syst."],"published-print":{"date-parts":[[2026,2]]},"DOI":"10.1007\/s40747-025-02193-0","type":"journal-article","created":{"date-parts":[[2025,12,4]],"date-time":"2025-12-04T20:46:06Z","timestamp":1764881166000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["SRL: A segmented reinforcement learning framework for long sequence layout decisions"],"prefix":"10.1007","volume":"12","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7933-8220","authenticated-orcid":false,"given":"Jie","family":"Yang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jian","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jinjin","family":"Hai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kai","family":"Qiao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Haoran","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bin","family":"Yan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,12,4]]},"reference":[{"issue":"4","key":"2193_CR1","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3464959","volume":"17","author":"UJ Botero","year":"2021","unstructured":"Botero UJ, Wilson R, Lu H, Rahman MT, Mallaiyan MA, Ganji F, Asadizanjani N, Tehranipoor MM, Woodard DL, Forte D (2021) Hardware trust and assurance through reverse engineering: a tutorial and outlook from image analysis and machine learning perspectives 17(4):1\u201353. https:\/\/doi.org\/10.1145\/3464959","journal-title":"Hardware trust and assurance through reverse engineering: a tutorial and outlook from image analysis and machine learning perspectives"},{"key":"2193_CR2","doi-asserted-by":"publisher","unstructured":"AEM-PCB reverser (2024) Circuit schematic generation in PCB reverse engineering using reinforcement learning based on aesthetic evaluation metric 43, 1608\u20131612. https:\/\/doi.org\/10.1109\/TCAD.2023.3340869","DOI":"10.1109\/TCAD.2023.3340869"},{"issue":"6419","key":"2193_CR3","doi-asserted-by":"publisher","first-page":"6404","DOI":"10.1126\/science.aar6404","volume":"362","author":"D Silver","year":"2018","unstructured":"Silver D, Hubert T, Schrittwieser J, Antonoglou I, Lai M, Guez A, Lanctot M, Sifre L, Kumaran D, Graepel T, Lillicrap T, Simonyan K, Hassabis D (2018) A general reinforcement learning algorithm that masters chess, shogi, and go through self-play. Science 362(6419):6404. https:\/\/doi.org\/10.1126\/science.aar6404","journal-title":"Science"},{"key":"2193_CR4","doi-asserted-by":"publisher","unstructured":"Goldie A, Mirhoseini A, Placement optimization with deep reinforcement learning. In: Proceedings of the 2020 International Symposium on Physical Design, pp. 3\u20137. ACM. https:\/\/doi.org\/10.1145\/3372780.3378174","DOI":"10.1145\/3372780.3378174"},{"issue":"7873","key":"2193_CR5","doi-asserted-by":"publisher","first-page":"583","DOI":"10.1038\/s41586-021-03819-2","volume":"596","author":"J Jumper","year":"2021","unstructured":"Jumper J, Evans R, Pritzel A, Green T, Figurnov M, Ronneberger O, Tunyasuvunakool K, Bates R, \u017d\u00eddek A, Potapenko A, Bridgland A, Meyer C, Kohl SAA, Ballard AJ, Cowie A, Romera-Paredes B, Nikolov S, Jain R, Adler J, Back T, Petersen S, Reiman D, Clancy E, Zielinski M, Steinegger M, Pacholska M, Berghammer T, Bodenstein S, Silver D, Vinyals O, Senior AW, Kavukcuoglu K, Kohli P, Hassabis D (2021) Highly accurate protein structure prediction with AlphaFold. Nature 596(7873):583\u2013589 (https:\/\/doi.org\/10\/gk7nfp)","journal-title":"Nature"},{"key":"2193_CR6","doi-asserted-by":"publisher","unstructured":"Hsu H-Y, Lin MP-H, Automatic analog schematic diagram generation based on building block classification and reinforcement learning. In: Proceedings of the 2022 ACM\/IEEE Workshop on Machine Learning For CAD, pp. 43\u201348. ACM. https:\/\/doi.org\/10.1145\/3551901.3556486","DOI":"10.1145\/3551901.3556486"},{"key":"2193_CR7","doi-asserted-by":"publisher","unstructured":"Ivanova M, Rozeva A, Ninov A, Stosovic MA, Reinforcement learning at design of electronic circuits: Review and analysis. In: Proceedings of the 2022 5th Artificial Intelligence and Cloud Computing Conference, pp. 275\u2013284. ACM. https:\/\/doi.org\/10.1145\/3582099.3582140","DOI":"10.1145\/3582099.3582140"},{"key":"2193_CR8","doi-asserted-by":"publisher","unstructured":"Meng, D, Zheng Y-L (2022) Circuit partitioning for PCB netlist based on net attributes. In: International Conference on Machine Learning and Cybernetics (ICMLC), 31\u201336. IEEE. https:\/\/doi.org\/10.1109\/ICMLC56445.2022.9941328","DOI":"10.1109\/ICMLC56445.2022.9941328"},{"key":"2193_CR9","first-page":"1928","volume":"48","author":"V Mnih","year":"2016","unstructured":"Mnih V, Badia AP, Mirza M, Graves A, Lillicrap TP, Harley T, Silver D, Kavukcuoglu K (2016) Asynchronous methods for deep reinforcement learning. In: International Conference on Machine Learning 48:1928\u20131937","journal-title":"In: International Conference on Machine Learning"},{"key":"2193_CR10","unstructured":"Bellemare MG, Dabney W, Munos R (2017) A distributional perspective on reinforcement learning. In: Proceedings of the 34th International Conference on Machine Learning - Volume 70. ICML\u201917, pp. 449\u2013458"},{"issue":"10","key":"2193_CR11","doi-asserted-by":"publisher","first-page":"6320","DOI":"10.1109\/TSMC.2024.3428482","volume":"54","author":"D Wang","year":"2024","unstructured":"Wang D, Wang J, Hu L, Zhang L (2024) Novel parallel formulation for iterative reinforcement learning control 54(10):6320\u20136331. https:\/\/doi.org\/10.1109\/TSMC.2024.3428482","journal-title":"Novel parallel formulation for iterative reinforcement learning control"},{"key":"2193_CR12","doi-asserted-by":"publisher","unstructured":"Hou J, Chen G, Zhang R, Li Z, Gu S, Jiang C (2025) Spreeze: High-throughput parallel reinforcement learning framework 36(2), 282\u2013292 https:\/\/doi.org\/10.1109\/TPDS.2024.3497986","DOI":"10.1109\/TPDS.2024.3497986"},{"key":"2193_CR13","unstructured":"Horgan D, Quan J, Budden D, Barth-Maron G, Hessel M, Hasselt Hv, Silver D (2018) Distributed prioritized experience replay. In: International Conference on Learning Representations"},{"issue":"9","key":"2193_CR14","doi-asserted-by":"publisher","first-page":"9326","DOI":"10.1109\/TCYB.2021.3053414","volume":"52","author":"Q Wei","year":"2022","unstructured":"Wei Q, Ma H, Chen C, Dong D (2022) Deep reinforcement learning with quantum-inspired experience replay 52(9):9326\u20139338. https:\/\/doi.org\/10.1109\/TCYB.2021.3053414","journal-title":"Deep reinforcement learning with quantum-inspired experience replay"},{"key":"2193_CR15","doi-asserted-by":"publisher","unstructured":"Li H, Qian X, Song W (2024) Prioritized experience replay based on dynamics priority 14(1):6014. https:\/\/doi.org\/10.1038\/s41598-024-56673-3","DOI":"10.1038\/s41598-024-56673-3"},{"key":"2193_CR16","doi-asserted-by":"publisher","unstructured":"Yu J, Li J, L S, Han S (2024) Mixed experience sampling for off-policy reinforcement learning 251, 124017. https:\/\/doi.org\/10.1016\/j.eswa.2024.124017","DOI":"10.1016\/j.eswa.2024.124017"},{"key":"2193_CR17","doi-asserted-by":"publisher","unstructured":"Ahn H, Choi H, Han J, Moon T (2025) Option-aware Temporally Abstracted Value for Offline Goal-Conditioned Reinforcement Learning. arXiv. https:\/\/doi.org\/10.48550\/arXiv.2505.12737","DOI":"10.48550\/arXiv.2505.12737"},{"key":"2193_CR18","doi-asserted-by":"publisher","first-page":"108","DOI":"10.1016\/j.patrec.2024.02.002","volume":"179","author":"Z Tan","year":"2024","unstructured":"Tan Z, Mu Y (2024) Hierarchical reinforcement learning for chip-macro placement in integrated circuit. Pattern Recognit. Lett. 179:108\u2013114. https:\/\/doi.org\/10.1016\/j.patrec.2024.02.002","journal-title":"Pattern Recognit. Lett."},{"issue":"3","key":"2193_CR19","doi-asserted-by":"publisher","first-page":"3852","DOI":"10.1109\/TASE.2023.3288037","volume":"21","author":"X Liu","year":"2024","unstructured":"Liu X, Wang G, Liu Z, Liu Y, Liu Z, Huang P (2024) Hierarchical reinforcement learning integrating with human knowledge for practical robot skill learning in complex multi-stage manipulation. IEEE Trans Autom Sci Eng 21(3):3852\u20133862. https:\/\/doi.org\/10.1109\/TASE.2023.3288037","journal-title":"IEEE Trans Autom Sci Eng"},{"key":"2193_CR20","doi-asserted-by":"publisher","unstructured":"Yu Y, Zhai Z, Li W, Ma J (2024) Target-oriented multi-agent coordination with hierarchical reinforcement learning 14(16):7084. https:\/\/doi.org\/10.3390\/app14167084","DOI":"10.3390\/app14167084"},{"key":"2193_CR21","doi-asserted-by":"publisher","unstructured":"Yang G (2025) State filtered disturbance rejection control 113(7):6739\u20136755. https:\/\/doi.org\/10.1007\/s11071-024-10449-6","DOI":"10.1007\/s11071-024-10449-6"},{"key":"2193_CR22","doi-asserted-by":"publisher","unstructured":"Yang G, Yao J (2024) Multilayer neurocontrol of high-order uncertain nonlinear systems with active disturbance rejection 34(4):2972\u20132987. https:\/\/doi.org\/10.1002\/rnc.7118","DOI":"10.1002\/rnc.7118"},{"key":"2193_CR23","doi-asserted-by":"publisher","unstructured":"Beg A, Beg A (2016) Auto-generating publication-quality circuit schematics using open technologies. In: Proceedings of the 21st Western Canadian Conference on Computing Education, pp. 1\u20136. https:\/\/doi.org\/10.1145\/2910925.2910927","DOI":"10.1145\/2910925.2910927"},{"key":"2193_CR24","doi-asserted-by":"crossref","unstructured":"Ferreira A, Lourenco N, Martins R, Horta N (2016) Automated analog IC design constraints generation for a layout-aware sizing approach. In: 2016 13th International Conference on Synthesis, Modeling, Analysis and Simulation Methods and Applications to Circuit Design (SMACD), pp. 1\u20134. 10\/grr3qm","DOI":"10.1109\/SMACD.2016.7520740"},{"key":"2193_CR25","doi-asserted-by":"publisher","unstructured":"Kunal K, Madhusudan M, Sharma AK, Xu W, Burns SM, Harjani R, Hu J, Kirkpatrick DA, Sapatnekar SS (2019) ALIGN: Open-source analog layout automation from the ground up. In: Proceedings of the 56th Annual Design Automation Conference, pp. 1\u20134. ACM. https:\/\/doi.org\/10.1145\/3316781.3323471","DOI":"10.1145\/3316781.3323471"},{"key":"2193_CR26","doi-asserted-by":"publisher","unstructured":"Fu R, Zhang Z-M, Tang G-M, Huang J, Ye X-C, Fan D-R, Sun N-H. Design automation methodology from RTL to gate-level netlist and schematic for RSFQ logic circuits. In: Proceedings of the 2020 on Great Lakes Symposium on VLSI, pp. 145\u2013150. ACM. https:\/\/doi.org\/10.1145\/3386263.3406898","DOI":"10.1145\/3386263.3406898"},{"key":"2193_CR27","unstructured":"Patyal A, Chen H-M, Lin MP-H, Fang G-Q, Chen SY-H (2022) Pole-aware Analog Layout Synthesis Considering Monotonic Current Flows and Wire-Crossings, 1\u20131. 10\/gq8x27"},{"issue":"1","key":"2193_CR28","doi-asserted-by":"publisher","first-page":"765","DOI":"10.1109\/TPWRS.2024.3404116","volume":"40","author":"J Wang","year":"2025","unstructured":"Wang J, Zhang J, Hou Q, Zhang N (2025) Synchronous condenser placement for multiple hvdc power systems considering short-circuit ratio requirements. IEEE Trans Power Syst 40(1):765\u2013779. https:\/\/doi.org\/10.1109\/TPWRS.2024.3404116","journal-title":"IEEE Trans Power Syst"},{"key":"2193_CR29","doi-asserted-by":"publisher","unstructured":"Wang D, Xie W, Cai Y, Liu X (2022) Adaptive data augmentation network for human pose estimation 129, 103681. https:\/\/doi.org\/10.1016\/j.dsp.2022.103681","DOI":"10.1016\/j.dsp.2022.103681"},{"key":"2193_CR30","doi-asserted-by":"crossref","unstructured":"Mirhoseini A, Goldie A, Yazgan M, Jiang JW, Songhori E, Wang S, Lee Y-J, Johnson E, Pathak O, Nazi A, Pak J, Tong A, Srinivasa K, Hang W, Tuncer E, Le QV, Laudon J, Ho R, Carpenter R, Dean J (2022) A graph placement methodology for fast chip design 604(7906):24\u201324. https:\/\/doi.org\/10\/gqtz7c","DOI":"10.1038\/s41586-022-04657-6"},{"key":"2193_CR31","doi-asserted-by":"publisher","unstructured":"Lai Y, Liu J, Tang Z, Wang B, Hao J, Luo P (2023) ChiPFormer: Transferable chip placement via offline decision transformer. In: Proceedings of the 40th International Conference on Machine Learning. ICML\u201923. JMLR.org, Honolulu, Hawaii, USA. https:\/\/doi.org\/10.5555\/3618408.3619165","DOI":"10.5555\/3618408.3619165"},{"key":"2193_CR32","doi-asserted-by":"publisher","unstructured":"Wang J, Zhang X, Tan Z, Zhu M (2024) RTplace: A reinforcement learning-based macro placement method with ResNet and transformer. In: 2024 9th International Conference on Integrated Circuits and Microsystems (ICICM), pp. 677\u2013681. IEEE, Wuhan, China. https:\/\/doi.org\/10.1109\/ICICM63644.2024.10814149","DOI":"10.1109\/ICICM63644.2024.10814149"},{"key":"2193_CR33","doi-asserted-by":"publisher","unstructured":"You F, Li J, Zhu L, Chen Z, Huang Z (2021) Domain adaptive semantic segmentation without source data. In: MM \u201921: ACM Multimedia Conference, Virtual Event, China, October 20 - 24, pp. 3293\u20133302. ACM. https:\/\/doi.org\/10.1145\/3474085.3475482","DOI":"10.1145\/3474085.3475482"},{"key":"2193_CR34","doi-asserted-by":"publisher","unstructured":"Wang Z, Luo Y, Chen Z, Wang S, Huang Z (2023) Cal-SFDA: Source-free domain-adaptive semantic segmentation with differentiable expected calibration error. In: Proceedings of the 31st ACM International Conference on Multimedia, MM 2023, Ottawa, On, Canada, 29 October 2023- 3 November pp. 1167\u20131178. ACM. https:\/\/doi.org\/10.1145\/3581783.3611808","DOI":"10.1145\/3581783.3611808"},{"key":"2193_CR35","doi-asserted-by":"publisher","unstructured":"Yu Z, Liao Z, Li J, Chen Z, Zhu L (2025) Dynamic target distribution estimation for source-free open-set domain adaptation. In: Walsh, T., Shah, J., Kolter, Z. (eds.) AAAI-25, Sponsored by the Association for the Advancement of Artificial Intelligence, February 25 - March 4, Philadelphia, PA, USA, pp. 22254\u201322262. AAAI Press. https:\/\/doi.org\/10.1609\/AAAI.V39I21.34380","DOI":"10.1609\/AAAI.V39I21.34380"},{"key":"2193_CR36","doi-asserted-by":"publisher","unstructured":"Su H, Li J, Chen Z, Zhu L, Lu K (2022) Distinguishing unseen from seen for generalized zero-shot learning. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2022, New Orleans, LA, USA, June 18-24, pp. 7875\u20137884. IEEE. https:\/\/doi.org\/10.1109\/CVPR52688.2022.00773","DOI":"10.1109\/CVPR52688.2022.00773"},{"key":"2193_CR37","doi-asserted-by":"publisher","unstructured":"Chen Z, Luo Y, Qiu R, Wang S, Huang Z, Li J, Zhang Z (2021) Semantics disentangling for generalized zero-shot learning. In: 2021 IEEE\/CVF International Conference on Computer Vision, ICCV 2021, Montreal, QC, Canada, pp. 8692\u20138700. IEEE. https:\/\/doi.org\/10.1109\/ICCV48922.2021.00859","DOI":"10.1109\/ICCV48922.2021.00859"},{"key":"2193_CR38","doi-asserted-by":"publisher","unstructured":"Chen Z, Zhang P-F, Li J, Wang S, Huang Z. Zero-shot learning by harnessing adversarial samples. In: Proceedings of the 31st ACM International Conference on Multimedia, MM 2023, Ottawa, On, Canada, 29 October 2023- 3 November 2023, pp. 4138\u20134146. ACM. https:\/\/doi.org\/10.1145\/3581783.3611823","DOI":"10.1145\/3581783.3611823"},{"key":"2193_CR39","doi-asserted-by":"publisher","unstructured":"Chen Z, Zhao Z, Guo J, Li J, Huang Z (2025) SVIP: Semantically contextualized visual patches for zero-shot learning. https:\/\/doi.org\/10.48550\/ARXIV.2503.10252. arxiv:2503.10252","DOI":"10.48550\/ARXIV.2503.10252"},{"issue":"2","key":"2193_CR40","doi-asserted-by":"publisher","first-page":"733","DOI":"10.13328\/j.cnki.jos.006706","volume":"34","author":"ZG Huang","year":"2023","unstructured":"Huang ZG, Liu Q, Zhang LH, Cao JQ, Zhu F (2023) Research and development on deep hierarchical reinforcement learning. Journal of Software 34(2):733\u2013760. https:\/\/doi.org\/10.13328\/j.cnki.jos.006706","journal-title":"Journal of Software"},{"issue":"1\u20132","key":"2193_CR41","doi-asserted-by":"publisher","first-page":"181","DOI":"10.1016\/S0004-3702(99)00052-1","volume":"112","author":"RS Sutton","year":"1999","unstructured":"Sutton RS, Precup D, Singh S (1999) Between MDPs and semi-MDPs: A framework for temporal abstraction in reinforcement learning. Artif Intell 112(1\u20132):181\u2013211. https:\/\/doi.org\/10.1016\/S0004-3702(99)00052-1","journal-title":"Artif Intell"},{"key":"2193_CR42","doi-asserted-by":"publisher","unstructured":"Bacon P-L, Harb J, Precup D (2017) The option-critic architecture. In: Proceedings of the Thirty-first AAAI Conference on Artificial Intelligence. AAAI\u201917, pp. 1726\u20131734. AAAI Press, San Francisco, California, USA. https:\/\/doi.org\/10.5555\/3298483.3298491","DOI":"10.5555\/3298483.3298491"},{"key":"2193_CR43","doi-asserted-by":"publisher","unstructured":"Kamat A, Precup D (2020) Diversity-Enriched Option-Critic. arXiv. https:\/\/doi.org\/10.48550\/arXiv.2011.02565","DOI":"10.48550\/arXiv.2011.02565"},{"key":"2193_CR44","doi-asserted-by":"publisher","unstructured":"Klissarov M, Machado MC (2023) Deep laplacian-based options for temporally-extended exploration. In: Proceedings of the 40th International Conference on Machine Learning. ICML\u201923. JMLR.org, Honolulu, Hawaii, USA. https:\/\/doi.org\/10.5555\/3618408.3619115","DOI":"10.5555\/3618408.3619115"},{"key":"2193_CR45","doi-asserted-by":"publisher","unstructured":"Vezhnevets AS, Osindero S, Schaul T, Heess N, Jaderberg M, Silver D, Kavukcuoglu K (2017) FeUdal networks for hierarchical reinforcement learning. In: Proceedings of the 34th International Conference on Machine Learning - Volume 70. ICML\u201917, pp. 3540\u20133549. JMLR.org, Sydney, NSW, Australia. https:\/\/doi.org\/10.5555\/3305890.3306047","DOI":"10.5555\/3305890.3306047"},{"key":"2193_CR46","doi-asserted-by":"publisher","unstructured":"Nachum O, Gu S, Lee H, Levine S (2018) Data-efficient hierarchical reinforcement learning. In: Proceedings of the 32nd International Conference on Neural Information Processing Systems. NIPS\u201918, pp. 3307\u20133317. Curran Associates Inc., Montr\u00e9al, Canada and Red Hook, NY, USA. https:\/\/doi.org\/10.5555\/3327144.3327250","DOI":"10.5555\/3327144.3327250"},{"key":"2193_CR47","doi-asserted-by":"publisher","unstructured":"Wang R, Yu R, An B, Rabinovich Z (2021) I2HRL: Interactive influence-based hierarchical reinforcement learning. In: Proceedings of the Twenty-ninth International Joint Conference on Artificial Intelligence. IJCAI\u201920, Yokohama, Yokohama, Japan. https:\/\/doi.org\/10.5555\/3491440.3491873","DOI":"10.5555\/3491440.3491873"},{"key":"2193_CR48","doi-asserted-by":"publisher","unstructured":"Dietterich TG (1998) The MAXQ method for hierarchical reinforcement learning. In: Proceedings of the Fifteenth International Conference on Machine Learning. Icml \u201998, pp. 118\u2013126. Morgan Kaufmann Publishers Inc., San Francisco, CA, USA . https:\/\/doi.org\/10.5555\/645527.657449","DOI":"10.5555\/645527.657449"},{"key":"2193_CR49","doi-asserted-by":"publisher","unstructured":"Esteban D, Rozo L, Caldwell DG (2019) Hierarchical reinforcement learning for concurrent discovery of compound and composable policies. In: 2019 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS), pp. 1818\u20131825. IEEE, Macau, China. https:\/\/doi.org\/10.1109\/IROS40897.2019.8968149","DOI":"10.1109\/IROS40897.2019.8968149"},{"key":"2193_CR50","doi-asserted-by":"publisher","first-page":"533","DOI":"10.1609\/icaps.v31i1.16001","volume":"31","author":"H Kokel","year":"2021","unstructured":"Kokel H, Manoharan A, Natarajan S, Ravindran B, Tadepalli P (2021) RePReL: Integrating relational planning and reinforcement learning for effective abstraction. Proc. Int. Conf. Autom. Plan. Sched. 31:533\u2013541. https:\/\/doi.org\/10.1609\/icaps.v31i1.16001","journal-title":"Proc. Int. Conf. Autom. Plan. Sched."},{"key":"2193_CR51","doi-asserted-by":"publisher","unstructured":"Sutton RS, Barto AG (2018) Reinforcement Learning: An Introduction. A Bradford Book, Cambridge, MA, USA. https:\/\/doi.org\/10.5555\/3312046","DOI":"10.5555\/3312046"},{"key":"2193_CR52","unstructured":"Haarnoja T, Zhou A, Abbeel P, Levine S (2018) Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor. In: Dy, J., Krause, A. (eds.) Proceedings of the 35th International Conference on Machine Learning. Proceedings of Machine Learning Research, vol. 80, pp. 1861\u20131870"},{"key":"2193_CR53","unstructured":"Schulman J, Wolski F, Dhariwal P, Radford A, Klimov O (2017) Proximal policy optimization algorithms. CoRR arxiv:1707.06347"},{"key":"2193_CR54","doi-asserted-by":"publisher","unstructured":"Sutton RS, McAllester D, Singh S, Mansour Y (1999) Policy gradient methods for reinforcement learning with function approximation. In: Proceedings of the 13th International Conference on Neural Information Processing Systems. NIPS\u201999, pp. 1057\u20131063, Cambridge, MA, USA. https:\/\/doi.org\/10.5555\/3009657.3009806","DOI":"10.5555\/3009657.3009806"},{"key":"2193_CR55","unstructured":"Schulman J, Moritz P, Levine S, Jordan MI, Abbeel P (2016) High-dimensional continuous control using generalized advantage estimation. In: Bengio, Y., LeCun, Y. (eds.) 4th International Conference on Learning Representations, ICLR 2016, San Juan, Puerto Rico, May 2-4, Conference Track Proceedings. arxiv:1506.02438"},{"key":"2193_CR56","doi-asserted-by":"publisher","unstructured":"Hu Z, Kaneko T (2021) Hierarchical advantage for reinforcement learning in parameterized action space. In: 2021 IEEE Conference on Games (CoG), pp. 1\u20138. https:\/\/doi.org\/10.1109\/CoG52621.2021.9619068","DOI":"10.1109\/CoG52621.2021.9619068"}],"container-title":["Complex &amp; Intelligent Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s40747-025-02193-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s40747-025-02193-0","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s40747-025-02193-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,16]],"date-time":"2026-02-16T08:24:09Z","timestamp":1771230249000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s40747-025-02193-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,4]]},"references-count":56,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2026,2]]}},"alternative-id":["2193"],"URL":"https:\/\/doi.org\/10.1007\/s40747-025-02193-0","relation":{},"ISSN":["2199-4536","2198-6053"],"issn-type":[{"value":"2199-4536","type":"print"},{"value":"2198-6053","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,12,4]]},"assertion":[{"value":"9 September 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 November 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 December 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"We declare that there are no financial or non-financial conflicts of interest associated with this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflicts of interest"}},{"value":"We declare that all authors have given their consent to publish the paper in your journal and that the paper does not involve human, animal, etc. research.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent to publish"}}],"article-number":"61"}}