{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T16:15:53Z","timestamp":1783181753599,"version":"3.54.6"},"reference-count":90,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2027,4,28]],"date-time":"2027-04-28T00:00:00Z","timestamp":1808870400000},"content-version":"am","delay-in-days":301,"URL":"http:\/\/www.elsevier.com\/open-access\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["CMMI 2027425"],"award-info":[{"award-number":["CMMI 2027425"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Journal of Network and Computer Applications"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1016\/j.jnca.2026.104508","type":"journal-article","created":{"date-parts":[[2026,4,24]],"date-time":"2026-04-24T15:50:43Z","timestamp":1777045843000},"page":"104508","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["A structured state\u2013reward compatibility framework for reinforcement learning in resource-constrained cyber\u2013physical systems"],"prefix":"10.1016","volume":"251","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-3964-2042","authenticated-orcid":false,"given":"Tahsin Afroz Hoque","family":"Nishat","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5056-1154","authenticated-orcid":false,"given":"Hongki","family":"Jo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"issue":"10","key":"10.1016\/j.jnca.2026.104508_b1","doi-asserted-by":"crossref","first-page":"6795","DOI":"10.1109\/TPAMI.2021.3103132","article-title":"Continuous action reinforcement learning from a mixture of interpretable experts","volume":"44","author":"Akrour","year":"2022","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.jnca.2026.104508_b2","series-title":"Concrete problems in AI safety","author":"Amodei","year":"2016"},{"key":"10.1016\/j.jnca.2026.104508_b3","series-title":"Proceedings of the 33rd International Conference on Neural Information Processing Systems","first-page":"13566","article-title":"RUDDER: return decomposition for delayed rewards","author":"Arjona-Medina","year":"2019"},{"key":"10.1016\/j.jnca.2026.104508_b4","doi-asserted-by":"crossref","DOI":"10.1016\/j.artint.2021.103500","article-title":"A survey of inverse reinforcement learning: Challenges, methods and progress","volume":"297","author":"Arora","year":"2021","journal-title":"Artificial Intelligence"},{"key":"10.1016\/j.jnca.2026.104508_b5","doi-asserted-by":"crossref","first-page":"469","DOI":"10.1016\/j.future.2024.03.057","article-title":"Potential-based reward shaping using state\u2013space segmentation for efficiency in reinforcement learning","volume":"157","author":"Bal","year":"2024","journal-title":"Future Gener. Comput. Syst."},{"issue":"1","key":"10.1016\/j.jnca.2026.104508_b6","doi-asserted-by":"crossref","first-page":"355","DOI":"10.1007\/s10994-023-06479-7","article-title":"Explainable reinforcement learning (XRL): a systematic literature review and taxonomy","volume":"113","author":"Bekkemoen","year":"2024","journal-title":"Mach. Learn."},{"issue":"10","key":"10.1016\/j.jnca.2026.104508_b7","doi-asserted-by":"crossref","first-page":"2143","DOI":"10.1587\/transinf.2019EDP7170","article-title":"Towards interpretable reinforcement learning with state abstraction driven by external knowledge","volume":"E103.D","author":"Bougie","year":"2020","journal-title":"IEICE Trans. Inf. Syst."},{"issue":"3","key":"10.1016\/j.jnca.2026.104508_b8","doi-asserted-by":"crossref","first-page":"1659","DOI":"10.1109\/COMST.2021.3073036","article-title":"Deep reinforcement learning for internet of things: A comprehensive survey","volume":"23","author":"Chen","year":"2021","journal-title":"IEEE Commun. Surv. Tutor."},{"issue":"3","key":"10.1016\/j.jnca.2026.104508_b9","doi-asserted-by":"crossref","first-page":"1394","DOI":"10.1109\/LRA.2018.2800101","article-title":"Integrating state representation learning into deep reinforcement learning","volume":"3","author":"de Bruin","year":"2018","journal-title":"IEEE Robot. Autom. Lett."},{"issue":"3","key":"10.1016\/j.jnca.2026.104508_b10","doi-asserted-by":"crossref","first-page":"653","DOI":"10.1109\/TNNLS.2016.2522401","article-title":"Deep direct reinforcement learning for financial signal representation and trading","volume":"28","author":"Deng","year":"2016","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"issue":"9","key":"10.1016\/j.jnca.2026.104508_b11","doi-asserted-by":"crossref","first-page":"2419","DOI":"10.1007\/s10994-021-05961-4","article-title":"Challenges of real-world reinforcement learning: Definitions, benchmarks and analysis","volume":"110","author":"Dulac-Arnold","year":"2021","journal-title":"Mach. Learn."},{"key":"10.1016\/j.jnca.2026.104508_b12","doi-asserted-by":"crossref","first-page":"73","DOI":"10.1016\/j.comcom.2023.02.011","article-title":"Reinforcement learning based flow and energy management in resource-constrained wireless networks","volume":"202","author":"Dutta","year":"2023","journal-title":"Comput. Commun."},{"issue":"13","key":"10.1016\/j.jnca.2026.104508_b13","doi-asserted-by":"crossref","DOI":"10.3390\/math12132102","article-title":"Cooperative multi-agent reinforcement learning for data gathering in energy-harvesting wireless sensor networks","volume":"12","author":"Dvir","year":"2024","journal-title":"Mathematics"},{"key":"10.1016\/j.jnca.2026.104508_b14","series-title":"International Conference on Learning Representations","article-title":"Implementation matters in deep RL: A case study on PPO and TRPO","author":"Engstrom","year":"2020"},{"key":"10.1016\/j.jnca.2026.104508_b15","doi-asserted-by":"crossref","DOI":"10.1016\/j.comnet.2023.109934","article-title":"A survey on how network simulators serve reinforcement learning in wireless networks","volume":"234","author":"Ergun","year":"2023","journal-title":"Comput. Netw."},{"issue":"17","key":"10.1016\/j.jnca.2026.104508_b16","doi-asserted-by":"crossref","DOI":"10.3390\/s24175791","article-title":"A novel medium access policy based on reinforcement learning in energy-harvesting underwater sensor networks","volume":"24","author":"Eri\u015f","year":"2024","journal-title":"Sensors"},{"key":"10.1016\/j.jnca.2026.104508_b17","series-title":"ICML","first-page":"2056","article-title":"Learning robust rewards with adversarial inverse reinforcement learning","author":"Fu","year":"2018"},{"key":"10.1016\/j.jnca.2026.104508_b18","doi-asserted-by":"crossref","DOI":"10.1016\/j.coche.2024.101012","article-title":"Deep reinforcement learning for process design: Review and perspective","volume":"44","author":"Gao","year":"2024","journal-title":"Curr. Opin. Chem. Eng."},{"issue":"1","key":"10.1016\/j.jnca.2026.104508_b19","doi-asserted-by":"crossref","first-page":"6918","DOI":"10.1016\/j.ifacol.2017.08.1217","article-title":"Reinforcement learning for electric power system decision and control: Past considerations and perspectives","volume":"50","author":"Glavic","year":"2017","journal-title":"IFAC-PapersOnLine"},{"key":"10.1016\/j.jnca.2026.104508_b20","series-title":"Proceedings of the 35th International Conference on Machine Learning","first-page":"1792","article-title":"Visualizing and understanding atari agents","volume":"vol. 80","author":"Greydanus","year":"2018"},{"issue":"3","key":"10.1016\/j.jnca.2026.104508_b21","first-page":"186","article-title":"Adaptive reward shaping for reinforcement learning agents","volume":"29","author":"Grze\u015b","year":"2017","journal-title":"Connect. Sci."},{"key":"10.1016\/j.jnca.2026.104508_b22","doi-asserted-by":"crossref","DOI":"10.1016\/j.iot.2023.100980","article-title":"Deep reinforcement learning based efficient access scheduling algorithm with an adaptive number of devices for federated learning IoT systems","volume":"24","author":"Guan","year":"2023","journal-title":"Internet Things"},{"key":"10.1016\/j.jnca.2026.104508_b23","doi-asserted-by":"crossref","DOI":"10.3389\/frcmn.2022.933047","article-title":"Status update control based on reinforcement learning in internet of things networks","volume":"3","author":"Han","year":"2022","journal-title":"Front. Commun. Netw."},{"key":"10.1016\/j.jnca.2026.104508_b24","first-page":"3207","article-title":"Deep reinforcement learning that matters","volume":"vol. 32","author":"Henderson","year":"2018"},{"key":"10.1016\/j.jnca.2026.104508_b25","series-title":"2019 IEEE International Conference on Communications Workshops (ICC Workshops)","first-page":"1","article-title":"Using deep Q-learning to prolong the lifetime of correlated internet of things devices","author":"Hribar","year":"2019"},{"issue":"5_6","key":"10.1016\/j.jnca.2026.104508_b26","doi-asserted-by":"crossref","first-page":"439","DOI":"10.12989\/sss.2010.6.5_6.439","article-title":"Structural health monitoring of a cable-stayed bridge using smart sensor technology: deployment and evaluation","volume":"6","author":"Jang","year":"2010","journal-title":"Smart Struct. Syst."},{"key":"10.1016\/j.jnca.2026.104508_b27","series-title":"Active management of battery degradation in wireless sensor network using deep reinforcement learning for group battery replacement","author":"Jeong","year":"2025"},{"key":"10.1016\/j.jnca.2026.104508_b28","article-title":"Hybrid wireless smart sensor network for full-scale structural health monitoring of a cable-stayed bridge","volume":"vol. 7981","author":"Jo","year":"2011"},{"issue":"1\u20132","key":"10.1016\/j.jnca.2026.104508_b29","doi-asserted-by":"crossref","first-page":"99","DOI":"10.1016\/S0004-3702(98)00023-X","article-title":"Planning and acting in partially observable stochastic domains","volume":"101","author":"Kaelbling","year":"1998","journal-title":"Artificial Intelligence"},{"issue":"11","key":"10.1016\/j.jnca.2026.104508_b30","doi-asserted-by":"crossref","first-page":"1238","DOI":"10.1177\/0278364913495721","article-title":"Reinforcement learning in robotics: A survey","volume":"32","author":"Kober","year":"2013","journal-title":"Int. J. Robot. Res."},{"issue":"10","key":"10.1016\/j.jnca.2026.104508_b31","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1371\/journal.pcbi.1008317","article-title":"Reward-predictive representations generalize across tasks in reinforcement learning","volume":"16","author":"Lehnert","year":"2020","journal-title":"PLoS Comput. Biol."},{"issue":"3","key":"10.1016\/j.jnca.2026.104508_b32","doi-asserted-by":"crossref","first-page":"1722","DOI":"10.1109\/COMST.2020.2988367","article-title":"Deep reinforcement learning for autonomous internet of things: Model, applications and challenges","volume":"22","author":"Lei","year":"2020","journal-title":"IEEE Commun. Surv. Tutor."},{"key":"10.1016\/j.jnca.2026.104508_b33","doi-asserted-by":"crossref","first-page":"379","DOI":"10.1016\/j.neunet.2018.07.006","article-title":"State representation learning for control: An overview","volume":"108","author":"Lesort","year":"2018","journal-title":"Neural Netw."},{"key":"10.1016\/j.jnca.2026.104508_b34","series-title":"2015 IEEE International Conference on Communication Workshop","first-page":"2625","article-title":"Smart duty cycle control with reinforcement learning for machine to machine communications","author":"Li","year":"2015"},{"key":"10.1016\/j.jnca.2026.104508_b35","doi-asserted-by":"crossref","first-page":"13728","DOI":"10.1109\/TASE.2025.3554861","article-title":"Auxiliary reward generation with transition distance representation learning","volume":"22","author":"Li","year":"2025","journal-title":"IEEE Trans. Autom. Sci. Eng."},{"issue":"1","key":"10.1016\/j.jnca.2026.104508_b36","doi-asserted-by":"crossref","DOI":"10.1016\/j.heliyon.2023.e23014","article-title":"Progress and summary of reinforcement learning on energy management of multi-power sources electric vehicles","volume":"10","author":"Lin","year":"2024","journal-title":"Heliyon"},{"issue":"3\u20134","key":"10.1016\/j.jnca.2026.104508_b37","doi-asserted-by":"crossref","first-page":"117","DOI":"10.1504\/IJSNET.2006.012027","article-title":"RL-MAC: a reinforcement learning based MAC protocol for wireless sensor networks","volume":"1","author":"Liu","year":"2006","journal-title":"Int. J. Sens. Netw."},{"key":"10.1016\/j.jnca.2026.104508_b38","article-title":"Reinforcement learning-based sensor activation scheduling in structural health monitoring","volume":"245","author":"Liu","year":"2021","journal-title":"Eng. Struct."},{"issue":"7","key":"10.1016\/j.jnca.2026.104508_b39","doi-asserted-by":"crossref","first-page":"4401","DOI":"10.1109\/TWC.2022.3225085","article-title":"Throughput maximization of wireless-powered communication network with mobile access points","volume":"22","author":"Liu","year":"2023","journal-title":"IEEE Trans. Wirel. Commun."},{"key":"10.1016\/j.jnca.2026.104508_b40","first-page":"22145","article-title":"Learning Pareto-optimal policies in multi-objective Markov decision processes","volume":"34","author":"Liu","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"issue":"6","key":"10.1016\/j.jnca.2026.104508_b41","doi-asserted-by":"crossref","first-page":"534","DOI":"10.1111\/mice.12522","article-title":"Collaborative duty cycling strategies in energy harvesting sensor networks","volume":"35","author":"Long","year":"2020","journal-title":"Computer-Aided Civ. Infrastruct. Eng."},{"issue":"2","key":"10.1016\/j.jnca.2026.104508_b42","first-page":"349","article-title":"A qos-aware energy-efficient scheduling algorithm for wireless sensor networks using reinforcement learning","volume":"19","author":"Ma","year":"2019","journal-title":"Sensors"},{"issue":"1","key":"10.1016\/j.jnca.2026.104508_b43","article-title":"Belief reward shaping in reinforcement learning","volume":"32","author":"Marom","year":"2018","journal-title":"Proc. the AAAI Conf. Artif. Intell."},{"issue":"7","key":"10.1016\/j.jnca.2026.104508_b44","doi-asserted-by":"crossref","DOI":"10.1145\/3616864","article-title":"Explainable reinforcement learning: A survey and comparative review","volume":"56","author":"Milani","year":"2024","journal-title":"ACM Comput. Surv."},{"key":"10.1016\/j.jnca.2026.104508_b45","doi-asserted-by":"crossref","DOI":"10.1016\/j.est.2021.103743","article-title":"SimSES: A holistic simulation framework for modeling and analyzing stationary energy storage systems","volume":"49","author":"M\u00f6ller","year":"2022","journal-title":"J. Energy Storage"},{"issue":"18","key":"10.1016\/j.jnca.2026.104508_b46","first-page":"7819","article-title":"Reinforcement-learning-based routing and resource allocation in IoT: A survey","volume":"23","author":"Musaddiq","year":"2023","journal-title":"Sensors"},{"key":"10.1016\/j.jnca.2026.104508_b47","unstructured":"Ng, A.Y., Harada, D., Russell, S.J., 1999. Policy invariance under reward transformations: Theory and application to reward shaping. In: ICML. pp. 278\u2013287."},{"key":"10.1016\/j.jnca.2026.104508_b48","doi-asserted-by":"crossref","DOI":"10.1016\/j.apenergy.2025.125731","article-title":"Reinforcement learning for adaptive battery management of structural health monitoring IoT sensor network","volume":"390","author":"Nishat","year":"2025","journal-title":"Appl. Energy"},{"key":"10.1016\/j.jnca.2026.104508_b49","first-page":"124860Z","article-title":"Actively managed battery degradation of wireless sensors for structural health monitoring","volume":"vol. 12486","author":"Nishat","year":"2023"},{"key":"10.1016\/j.jnca.2026.104508_b50","series-title":"System advisor model\u2122 version 2023.12.17 (sam\u2122 2023.12.17)","author":"NREL","year":"2023"},{"issue":"6","key":"10.1016\/j.jnca.2026.104508_b51","doi-asserted-by":"crossref","DOI":"10.1145\/3459991","article-title":"A survey of reinforcement learning algorithms for dynamically varying environments","volume":"54","author":"Padakandla","year":"2021","journal-title":"ACM Comput. Surv."},{"key":"10.1016\/j.jnca.2026.104508_b52","doi-asserted-by":"crossref","first-page":"139727","DOI":"10.1109\/ACCESS.2020.3010575","article-title":"Multi-agent reinforcement-learning-based time-slotted channel hopping medium access control scheduling scheme","volume":"8","author":"Park","year":"2020","journal-title":"IEEE Access"},{"key":"10.1016\/j.jnca.2026.104508_b53","doi-asserted-by":"crossref","first-page":"543","DOI":"10.1016\/j.proeng.2012.09.551","article-title":"Modal assurance criterion","volume":"48","author":"Pastor","year":"2012","journal-title":"Procedia Eng."},{"key":"10.1016\/j.jnca.2026.104508_b54","doi-asserted-by":"crossref","unstructured":"Pathak, D., Agrawal, P., Efros, A.A., Darrell, T., 2017. Curiosity-driven exploration by self-supervised prediction. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition Workshops. pp. 16\u201317.","DOI":"10.1109\/CVPRW.2017.70"},{"key":"10.1016\/j.jnca.2026.104508_b55","series-title":"Connectivity of soft random geometric graphs","author":"Penrose","year":"2016"},{"issue":"5_6","key":"10.1016\/j.jnca.2026.104508_b56","doi-asserted-by":"crossref","first-page":"423","DOI":"10.12989\/sss.2010.6.5_6.423","article-title":"Flexible smart sensor framework for autonomous structural health monitoring","volume":"6","author":"Rice","year":"2010","journal-title":"Smart Struct. Syst."},{"key":"10.1016\/j.jnca.2026.104508_b57","doi-asserted-by":"crossref","first-page":"13","DOI":"10.1016\/j.neunet.2022.05.013","article-title":"A survey for deep reinforcement learning in markovian cyber\u2013physical systems: Common problems and solutions","volume":"153","author":"Rupprecht","year":"2022","journal-title":"Neural Netw."},{"issue":"6","key":"10.1016\/j.jnca.2026.104508_b58","doi-asserted-by":"crossref","first-page":"3542","DOI":"10.1007\/s12083-024-01764-1","article-title":"A delay aware routing approach for FANET based on emperor penguins colony algorithm","volume":"17","author":"Sadrishojaei","year":"2024","journal-title":"Peer-To-Peer Netw. Appl."},{"issue":"7","key":"10.1016\/j.jnca.2026.104508_b59","doi-asserted-by":"crossref","first-page":"2251","DOI":"10.1080\/03772063.2025.2487936","article-title":"Energy-efficient routing for internet of things using combination of meta-heuristic algorithms in viral pandemics","volume":"71","author":"Sadrishojaei","year":"2025","journal-title":"IETE J. Res."},{"issue":"2","key":"10.1016\/j.jnca.2026.104508_b60","doi-asserted-by":"crossref","first-page":"47","DOI":"10.1007\/s43069-024-00331-x","article-title":"Clustered routing scheme in IoT during COVID-19 pandemic using hybrid black widow optimization and harmony search algorithm","volume":"5","author":"Sadrishojaei","year":"2024","journal-title":"Oper. Res. Forum"},{"issue":"6","key":"10.1016\/j.jnca.2026.104508_b61","first-page":"518","article-title":"New routing method in flying ad hoc networks based on squirrel search algorithm","volume":"47","author":"Sadrishojaei","year":"2025","journal-title":"Int. J. Comput. Appl."},{"issue":"19","key":"10.1016\/j.jnca.2026.104508_b62","doi-asserted-by":"crossref","first-page":"70","DOI":"10.2352\/ISSN.2470-1173.2017.19.AVM-023","article-title":"Deep reinforcement learning framework for autonomous driving","volume":"2017","author":"Sallab","year":"2017","journal-title":"Electron. Imaging"},{"key":"10.1016\/j.jnca.2026.104508_b63","series-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017"},{"key":"10.1016\/j.jnca.2026.104508_b64","doi-asserted-by":"crossref","first-page":"51","DOI":"10.1016\/j.rser.2018.03.003","article-title":"The national solar radiation data base (NSRDB)","volume":"89","author":"Sengupta","year":"2018","journal-title":"Renew. Sustain. Energy Rev."},{"key":"10.1016\/j.jnca.2026.104508_b65","series-title":"Logger \u2014 Stable Baselines3 documentation","author":"Stable Baselines3 Contributors","year":"2026"},{"key":"10.1016\/j.jnca.2026.104508_b66","series-title":"Reinforcement Learning: An Introduction","author":"Sutton","year":"2018"},{"issue":"21","key":"10.1016\/j.jnca.2026.104508_b67","doi-asserted-by":"crossref","first-page":"7836","DOI":"10.3390\/app10217836","article-title":"Accurate real time on-line estimation of state-of-health and remaining useful life of li ion batteries","volume":"10","author":"Tan","year":"2020","journal-title":"Appl. Sci."},{"key":"10.1016\/j.jnca.2026.104508_b68","series-title":"Using TensorBoard (ML-agents)","author":"Unity Technologies","year":"2018"},{"key":"10.1016\/j.jnca.2026.104508_b69","series-title":"Multi-objective reinforcement learning: A comprehensive overview","author":"Van Moffaert","year":"2014"},{"key":"10.1016\/j.jnca.2026.104508_b70","doi-asserted-by":"crossref","DOI":"10.1016\/j.adhoc.2024.103751","article-title":"Computational offloading and resource allocation for IoT applications using decision tree based reinforcement learning","volume":"170","author":"Walia","year":"2025","journal-title":"Ad Hoc Networks"},{"key":"10.1016\/j.jnca.2026.104508_b71","doi-asserted-by":"crossref","DOI":"10.1016\/j.jnca.2025.104250","article-title":"ReinFog: A deep reinforcement learning empowered framework for resource management in edge and cloud computing environments","volume":"242","author":"Wang","year":"2025","journal-title":"J. Netw. Comput. Appl."},{"issue":"9","key":"10.1016\/j.jnca.2026.104508_b72","doi-asserted-by":"crossref","first-page":"1617","DOI":"10.1109\/49.12889","article-title":"Routing of multipoint connections","volume":"6","author":"Waxman","year":"1988","journal-title":"IEEE J. Sel. Areas Commun."},{"key":"10.1016\/j.jnca.2026.104508_b73","doi-asserted-by":"crossref","unstructured":"Xia, S., Nishat, T.A.H., Jo, H., Liu, J., 2025. Regularized Tensor Completion for Structural Health Monitoring Data Imputation. In: 2025 11th International Conference on Computing and Artificial Intelligence (ICCAI). pp. 648\u2013653. http:\/\/dx.doi.org\/10.1109\/ICCAI66501.2025.00104.","DOI":"10.1109\/ICCAI66501.2025.00104"},{"key":"10.1016\/j.jnca.2026.104508_b74","series-title":"Proceedings of the 2023 4th International Conference on Computing, Networks and Internet of Things","first-page":"803","article-title":"Distributed reinforcement learning for optimizing age of information and energy consumption in wireless powered IoT systems","author":"Xu","year":"2023"},{"key":"10.1016\/j.jnca.2026.104508_b75","article-title":"An autonomic learning-based energy-efficient data gathering scheme in wireless sensor networks","volume":"176","author":"Yang","year":"2020","journal-title":"Comput. Netw."},{"issue":"3","key":"10.1016\/j.jnca.2026.104508_b76","doi-asserted-by":"crossref","first-page":"113","DOI":"10.3390\/wevj12030113","article-title":"A review of lithium-ion battery state of health estimation and prediction methods","volume":"12","author":"Yao","year":"2021","journal-title":"World Electr. Veh. J."},{"issue":"11","key":"10.1016\/j.jnca.2026.104508_b77","doi-asserted-by":"crossref","first-page":"1045","DOI":"10.1007\/s00607-014-0438-1","article-title":"Application of reinforcement learning to wireless sensor networks: models and algorithms","volume":"97","author":"Yau","year":"2015","journal-title":"Computing"},{"key":"10.1016\/j.jnca.2026.104508_b78","first-page":"1567","article-title":"An energy-efficient MAC protocol for wireless sensor networks","volume":"vol. 3","author":"Ye","year":"2002"},{"issue":"3","key":"10.1016\/j.jnca.2026.104508_b79","doi-asserted-by":"crossref","first-page":"493","DOI":"10.1109\/TNET.2004.828953","article-title":"Medium access control with coordinated adaptive sleeping for wireless sensor networks","volume":"12","author":"Ye","year":"2004","journal-title":"IEEE\/ACM Trans. Netw."},{"issue":"1","key":"10.1016\/j.jnca.2026.104508_b80","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3477600","article-title":"Reinforcement learning in healthcare: A survey","volume":"55","author":"Yu","year":"2021","journal-title":"ACM Comput. Surv."},{"issue":"5","key":"10.1016\/j.jnca.2026.104508_b81","doi-asserted-by":"crossref","DOI":"10.3390\/s24051632","article-title":"Deep reinforcement learning-based energy consumption optimization for peer-to-peer (P2P) communication in wireless sensor networks","volume":"24","author":"Yuan","year":"2024","journal-title":"Sensors"},{"key":"10.1016\/j.jnca.2026.104508_b82","series-title":"International Conference on Machine Learning","first-page":"11968","article-title":"Discovering interpretable reinforcement learning policies using saliency maps","author":"Zahavy","year":"2021"},{"key":"10.1016\/j.jnca.2026.104508_b83","first-page":"130988","article-title":"Designing reward functions for reinforcement learning","volume":"7","author":"Zhang","year":"2019","journal-title":"IEEE Access"},{"key":"10.1016\/j.jnca.2026.104508_b84","first-page":"319","article-title":"Learning state representations for reinforcement learning: A survey","volume":"74","author":"Zhang","year":"2022","journal-title":"J. Artificial Intelligence Res."},{"key":"10.1016\/j.jnca.2026.104508_b85","first-page":"321","article-title":"Multi-agent reinforcement learning: A selective overview of theories and algorithms","author":"Zhang","year":"2019","journal-title":"Handb. Reinf. Learn. Control."},{"issue":"2","key":"10.1016\/j.jnca.2026.104508_b86","first-page":"331","article-title":"Survey of reinforcement-learning-based MAC protocols for wireless networks","volume":"25","author":"Zheng","year":"2023","journal-title":"Entropy"},{"issue":"1","key":"10.1016\/j.jnca.2026.104508_b87","article-title":"Interpretable saliency map for deep reinforcement learning","volume":"1757","author":"Zheng","year":"2021","journal-title":"J. Phys.: Conf. Ser."},{"issue":"9","key":"10.1016\/j.jnca.2026.104508_b88","doi-asserted-by":"crossref","first-page":"16235","DOI":"10.1109\/JSEN.2025.3551916","article-title":"Relay selection and deployment for NOMA-enabled multi-AAV-assisted WSN","volume":"25","author":"Zheng","year":"2025","journal-title":"IEEE Sensors J."},{"issue":"3","key":"10.1016\/j.jnca.2026.104508_b89","doi-asserted-by":"crossref","first-page":"1755","DOI":"10.1109\/TCOMM.2023.3237854","article-title":"DRL-based offloading for computation delay minimization in wireless-powered multi-access edge computing","volume":"71","author":"Zheng","year":"2023","journal-title":"IEEE Trans. Commun."},{"issue":"17","key":"10.1016\/j.jnca.2026.104508_b90","doi-asserted-by":"crossref","first-page":"29102","DOI":"10.1109\/JIOT.2024.3406044","article-title":"Distributed DDPG-based resource allocation for age of information minimization in mobile wireless-powered internet of things","volume":"11","author":"Zheng","year":"2024","journal-title":"IEEE Internet Things J."}],"container-title":["Journal of Network and Computer Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1084804526000834?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1084804526000834?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T15:20:58Z","timestamp":1783178458000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S1084804526000834"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":90,"alternative-id":["S1084804526000834"],"URL":"https:\/\/doi.org\/10.1016\/j.jnca.2026.104508","relation":{},"ISSN":["1084-8045"],"issn-type":[{"value":"1084-8045","type":"print"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"A structured state\u2013reward compatibility framework for reinforcement learning in resource-constrained cyber\u2013physical systems","name":"articletitle","label":"Article Title"},{"value":"Journal of Network and Computer Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.jnca.2026.104508","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"104508"}}