{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,6]],"date-time":"2026-08-06T00:48:20Z","timestamp":1785977300354,"version":"3.56.0"},"reference-count":35,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61941113"],"award-info":[{"award-number":["61941113"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition"],"published-print":{"date-parts":[[2026,12]]},"DOI":"10.1016\/j.patcog.2026.114404","type":"journal-article","created":{"date-parts":[[2026,7,8]],"date-time":"2026-07-08T16:38:34Z","timestamp":1783528714000},"page":"114404","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PD","title":["Reward shaping using graph MAMBA"],"prefix":"10.1016","volume":"180","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7537-6844","authenticated-orcid":false,"given":"Jianghui","family":"Sang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Guangdi","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yongli","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hua","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2022-5747","authenticated-orcid":false,"given":"Jun","family":"Huang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.patcog.2026.114404_b1","doi-asserted-by":"crossref","unstructured":"D. Zhang, J. Liang, K. Guo, S. Lu, Q. Wang, R. Xiong, Z. Miao, Y. Wang, Car-planner: Consistent auto-regressive trajectory planning for large-scale reinforcement learning in autonomous driving, in: Proceedings of the Computer Vision and Pattern Recognition Conference, 2025, pp. 17239\u201317248.","DOI":"10.1109\/CVPR52734.2025.01607"},{"issue":"325","key":"10.1016\/j.patcog.2026.114404_b2","first-page":"1","article-title":"Mentored learning: Improving generalization and convergence of student learner","volume":"25","author":"Cao","year":"2024","journal-title":"J. Mach. Learn. Res."},{"key":"10.1016\/j.patcog.2026.114404_b3","doi-asserted-by":"crossref","unstructured":"C. Tang, B. Abbatematteo, J. Hu, R. Chandra, R. Mart\u00edn-Mart\u00edn, P. Stone, Deep reinforcement learning for robotics: A survey of real-world successes, in: Proceedings of the AAAI Conference on Artificial Intelligence, Vol. 39, (27) 2025, pp. 28694\u201328698.","DOI":"10.1609\/aaai.v39i27.35095"},{"key":"10.1016\/j.patcog.2026.114404_b4","doi-asserted-by":"crossref","unstructured":"Z. Hu, F. Zhang, L. Chen, K. Kuang, J. Li, K. Gao, J. Xiao, X. Wang, W. Zhu, Towards better alignment: Training diffusion models with reinforcement learning against sparse rewards, in: Proceedings of the Computer Vision and Pattern Recognition Conference, 2025, pp. 23604\u201323614.","DOI":"10.1109\/CVPR52734.2025.02198"},{"key":"10.1016\/j.patcog.2026.114404_b5","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110884","article-title":"A multi-agent curiosity reward model for task-oriented dialogue systems","volume":"157","author":"Sun","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.114404_b6","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2021.108352","article-title":"Unified curiosity-driven learning with smoothed intrinsic reward estimation","volume":"123","author":"Huang","year":"2022","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.114404_b7","unstructured":"A.Y. Ng, D. Harada, S. Russell, Policy invariance under reward transformations: Theory and application to reward shaping, in: International Conference on Machine Learning, Vol. 99, 1999, pp. 278\u2013287."},{"key":"10.1016\/j.patcog.2026.114404_b8","first-page":"12895","article-title":"Reward propagation using graph convolutional networks","volume":"33","author":"Klissarov","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.114404_b9","doi-asserted-by":"crossref","unstructured":"S.M. Devlin, D. Kudenko, Dynamic potential-based reward shaping, in: Proceedings of the 11th International Conference on Autonomous Agents and Multiagent Systems, 2012, pp. 433\u2013440.","DOI":"10.65109\/JJTT8551"},{"key":"10.1016\/j.patcog.2026.114404_b10","doi-asserted-by":"crossref","unstructured":"A. Harutyunyan, S. Devlin, P. Vrancx, A. Now\u00e9, Expressing arbitrary reward functions as potential-based advice, in: Proceedings of the AAAI Conference on Artificial Intelligence, Vol. 29, (1) 2015.","DOI":"10.1609\/aaai.v29i1.9628"},{"issue":"6","key":"10.1016\/j.patcog.2026.114404_b11","doi-asserted-by":"crossref","first-page":"424","DOI":"10.1038\/s44159-024-00304-1","article-title":"Understanding the development of reward learning through the lens of meta-learning","volume":"3","author":"Nussenbaum","year":"2024","journal-title":"Nat. Rev. Psychol."},{"key":"10.1016\/j.patcog.2026.114404_b12","doi-asserted-by":"crossref","first-page":"38786","DOI":"10.52202\/075280-1683","article-title":"Efficient potential-based exploration in reinforcement learning using inverse dynamic bisimulation metric","volume":"36","author":"Wang","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.114404_b13","doi-asserted-by":"crossref","first-page":"57765","DOI":"10.52202\/079017-1842","article-title":"Rethinking exploration in reinforcement learning with effective metric-based exploration bonus","volume":"37","author":"Wang","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.114404_b14","article-title":"Leveraging privileged information for partially observable reinforcement learning","author":"Li","year":"2025","journal-title":"IEEE Trans. Games"},{"key":"10.1016\/j.patcog.2026.114404_b15","doi-asserted-by":"crossref","first-page":"184","DOI":"10.1007\/s11063-024-11632-x","article-title":"Hierarchical reinforcement learning from demonstration via reachability-based reward shaping","author":"Gao","year":"2024","journal-title":"Neural Process. Lett."},{"key":"10.1016\/j.patcog.2026.114404_b16","article-title":"A reward-shaping dueling distributed multi-agent deep reinforcement learning framework for dynamic flexible job shop scheduling with random job arrivals","author":"Zhang","year":"2025","journal-title":"Expert Syst. Appl."},{"key":"10.1016\/j.patcog.2026.114404_b17","series-title":"Mamba: Linear-time sequence modeling with selective state spaces","author":"Gu","year":"2023"},{"key":"10.1016\/j.patcog.2026.114404_b18","doi-asserted-by":"crossref","DOI":"10.1016\/j.engappai.2025.111279","article-title":"Mamba-360: Survey of state space models as transformer alternative for long sequence modelling: Methods, applications, and challenges","volume":"159","author":"Patro","year":"2025","journal-title":"Eng. Appl. Artif. Intell."},{"key":"10.1016\/j.patcog.2026.114404_b19","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2024.125961","article-title":"Mamba-GIE: A visual state space models-based generalized image extrapolation method via dual-level adaptive feature fusion","volume":"264","author":"Zhang","year":"2025","journal-title":"Expert Syst. Appl."},{"key":"10.1016\/j.patcog.2026.114404_b20","doi-asserted-by":"crossref","unstructured":"A. Behrouz, F. Hashemi, Graph mamba: Towards learning on graphs with state space models, in: Proceedings of the 30th ACM SIGKDD Conference on Knowledge Discovery and Data Mining, 2024, pp. 119\u2013130.","DOI":"10.1145\/3637528.3672044"},{"key":"10.1016\/j.patcog.2026.114404_b21","doi-asserted-by":"crossref","unstructured":"M. Toussaint, A. Storkey, Probabilistic inference for solving discrete and continuous state Markov Decision Processes, in: International Conference on Machine Learning, 2006, pp. 945\u2013952.","DOI":"10.1145\/1143844.1143963"},{"key":"10.1016\/j.patcog.2026.114404_b22","article-title":"Reward shaping based on optimal-policy-free","author":"Sang","year":"2024","journal-title":"IEEE Trans. Big Data"},{"key":"10.1016\/j.patcog.2026.114404_b23","series-title":"Reinforcement learning and control as probabilistic inference: Tutorial and review","author":"Levine","year":"2018"},{"key":"10.1016\/j.patcog.2026.114404_b24","series-title":"Aaai","first-page":"1433","article-title":"Maximum entropy inverse reinforcement learning","volume":"Vol. 8","author":"Ziebart","year":"2008"},{"issue":"1","key":"10.1016\/j.patcog.2026.114404_b25","doi-asserted-by":"crossref","first-page":"4","DOI":"10.1109\/MASSP.1986.1165342","article-title":"An introduction to hidden Markov models","volume":"3","author":"Rabiner","year":"1986","journal-title":"IEEE Assp Mag."},{"issue":"3","key":"10.1016\/j.patcog.2026.114404_b26","doi-asserted-by":"crossref","first-page":"379","DOI":"10.1002\/j.1538-7305.1948.tb01338.x","article-title":"A mathematical theory of communication","volume":"27","author":"Shannon","year":"1948","journal-title":"Bell Syst. Tech. J."},{"key":"10.1016\/j.patcog.2026.114404_b27","series-title":"Rethinking attention with performers","author":"Choromanski","year":"2020"},{"issue":"6","key":"10.1016\/j.patcog.2026.114404_b28","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3773075","article-title":"Analytical survey of learning with low-resource data: From analysis to investigation","volume":"58","author":"Cao","year":"2025","journal-title":"ACM Comput. Surv."},{"key":"10.1016\/j.patcog.2026.114404_b29","series-title":"Information Theoretic Learning: Renyi\u2019s Entropy and Kernel Perspectives","author":"Principe","year":"2010"},{"key":"10.1016\/j.patcog.2026.114404_b30","doi-asserted-by":"crossref","first-page":"16455","DOI":"10.52202\/068431-1197","article-title":"Discovered policy optimisation","volume":"35","author":"Lu","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.114404_b31","series-title":"Relu to the rescue: Improve your on-policy actor-critic with positive advantages","author":"Jesson","year":"2023"},{"key":"10.1016\/j.patcog.2026.114404_b32","series-title":"Random latent exploration for deep reinforcement learning","author":"Mahankali","year":"2024"},{"key":"10.1016\/j.patcog.2026.114404_b33","doi-asserted-by":"crossref","first-page":"15320","DOI":"10.52202\/079017-0490","article-title":"Improving deep reinforcement learning by reducing the chain effect of value and policy churn","volume":"37","author":"Tang","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.114404_b34","unstructured":"H. Ma, F. Li, J.Y. Lim, Z. Luo, T.V. Vo, T.-Y. Leong, Catching two birds with one stone: Reward shaping with dual random networks for balancing exploration and exploitation, in: Forty-Second International Conference on Machine Learning, 2025."},{"key":"10.1016\/j.patcog.2026.114404_b35","doi-asserted-by":"crossref","DOI":"10.1016\/j.engappai.2025.110676","article-title":"Continuous reinforcement learning via advantage value difference reward shaping: A proximal policy optimization perspective","volume":"151","author":"Lin","year":"2025","journal-title":"Eng. Appl. Artif. Intell."}],"container-title":["Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326013695?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326013695?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,8,6]],"date-time":"2026-08-06T00:26:11Z","timestamp":1785975971000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0031320326013695"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,12]]},"references-count":35,"alternative-id":["S0031320326013695"],"URL":"https:\/\/doi.org\/10.1016\/j.patcog.2026.114404","relation":{},"ISSN":["0031-3203"],"issn-type":[{"value":"0031-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,12]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Reward shaping using graph MAMBA","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patcog.2026.114404","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Published by Elsevier Ltd.","name":"copyright","label":"Copyright"}],"article-number":"114404"}}