{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,11]],"date-time":"2026-06-11T16:09:18Z","timestamp":1781194158016,"version":"3.54.1"},"reference-count":51,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1016\/j.eswa.2026.132126","type":"journal-article","created":{"date-parts":[[2026,3,19]],"date-time":"2026-03-19T16:47:34Z","timestamp":1773938854000},"page":"132126","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Improving generalization in visual reinforcement learning through data distribution"],"prefix":"10.1016","volume":"320","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-1417-3084","authenticated-orcid":false,"given":"Hao","family":"Lei","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-4426-9151","authenticated-orcid":false,"given":"Yv","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-9414-2148","authenticated-orcid":false,"given":"Yi","family":"Xin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-2689-139X","authenticated-orcid":false,"given":"Liang","family":"Hongjian","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2920-0853","authenticated-orcid":false,"given":"Ke","family":"Liangjun","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.132126_bib0001","unstructured":"Agarwal, R., Machado, M. C., Castro, P. S., & Bellemare, M. G. (2021). Contrastive behavioral similarity embeddings for generalization in reinforcement learning. arXiv: 2101.05265."},{"key":"10.1016\/j.eswa.2026.132126_bib0002","doi-asserted-by":"crossref","DOI":"10.1016\/j.robot.2023.104399","article-title":"Robotic assembly strategy via reinforcement learning based on force and visual information","volume":"164","author":"Ahn","year":"2023","journal-title":"Robotics and Autonomous Systems"},{"key":"10.1016\/j.eswa.2026.132126_bib0003","first-page":"130","article-title":"A recipe for unbounded data augmentation in visual reinforcement learning","volume":"1","author":"Almuzairee","year":"2024","journal-title":"Reinforcement Learning Journal"},{"key":"10.1016\/j.eswa.2026.132126_bib0004","doi-asserted-by":"crossref","unstructured":"Balestriero, R., Misra, I., & LeCun, Y. (2022). A data-augmentation is worth a thousand samples: Exact quantification from analytical augmented sample moments. arXiv: 2202.08325.","DOI":"10.52202\/068431-1427"},{"key":"10.1016\/j.eswa.2026.132126_bib0005","unstructured":"Batra, S., & Sukhatme, G. S. (2024). Zero-Shot Generalization of Vision-Based RL Without Data Augmentation. arXiv: 2410.07441."},{"key":"10.1016\/j.eswa.2026.132126_bib0006","doi-asserted-by":"crossref","first-page":"30693","DOI":"10.52202\/068431-2225","article-title":"Look where you look! Saliency-guided Q-networks for generalization in visual Reinforcement Learning","volume":"35","author":"Bertoin","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132126_bib0007","series-title":"International conference on machine learning","first-page":"2048","article-title":"Leveraging procedural generation to benchmark reinforcement learning","author":"Cobbe","year":"2020"},{"key":"10.1016\/j.eswa.2026.132126_bib0008","series-title":"International conference on machine learning","first-page":"2020","article-title":"Phasic policy gradient","author":"Cobbe","year":"2021"},{"key":"10.1016\/j.eswa.2026.132126_bib0009","article-title":"Sinkhorn distances: Lightspeed computation of optimal transport","volume":"26","author":"Cuturi","year":"2013","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132126_bib0010","series-title":"International conference on machine learning","first-page":"1665","article-title":"Provably efficient rl with rich observations via latent state decoding","author":"Du","year":"2019"},{"issue":"6","key":"10.1016\/j.eswa.2026.132126_bib0011","doi-asserted-by":"crossref","first-page":"1662","DOI":"10.1137\/10080484X","article-title":"Bisimulation metrics for continuous Markov decision processes","volume":"40","author":"Ferns","year":"2011","journal-title":"SIAM Journal on Computing"},{"key":"10.1016\/j.eswa.2026.132126_bib0012","first-page":"25502","article-title":"Why generalization in rl is difficult: Epistemic pomdps and implicit partial observability","volume":"34","author":"Ghosh","year":"2021","journal-title":"Advances in Neural Information Processing Systems"},{"issue":"3","key":"10.1016\/j.eswa.2026.132126_bib0013","doi-asserted-by":"crossref","first-page":"419","DOI":"10.1111\/j.1751-5823.2002.tb00178.x","article-title":"On choosing and bounding probability metrics","volume":"70","author":"Gibbs","year":"2002","journal-title":"International Statistical Review"},{"key":"10.1016\/j.eswa.2026.132126_bib0014","unstructured":"Gmelin, K., Bahl, S., Mendonca, R., & Pathak, D. (2023). Efficient RL via disentangled environment and agent representations. arXiv: 2309.02435."},{"key":"10.1016\/j.eswa.2026.132126_bib0015","doi-asserted-by":"crossref","unstructured":"Grooten, B., Tomilin, T., Vasan, G., Taylor, M. E., Mahmood, A. R., Fang, M., Pechenizkiy, M., & Mocanu, D. C. (2023). MaDi: Learning to Mask Distractions for Generalization in Visual Deep Reinforcement Learning. arXiv: 2312.15339.","DOI":"10.65109\/YPTR7088"},{"key":"10.1016\/j.eswa.2026.132126_bib0016","series-title":"International conference on machine learning","first-page":"1861","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","author":"Haarnoja","year":"2018"},{"key":"10.1016\/j.eswa.2026.132126_bib0017","first-page":"3680","article-title":"Stabilizing deep q-learning with convnets and vision transformers under data augmentation","volume":"34","author":"Hansen","year":"2021","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132126_bib0018","series-title":"2021 IEEE international conference on robotics and automation (ICRA)","first-page":"13611","article-title":"Generalization in reinforcement learning by soft data augmentation","author":"Hansen","year":"2021"},{"key":"10.1016\/j.eswa.2026.132126_bib0019","unstructured":"Hu, J., Jiang, Y., & Weng, P. (2024). Revisiting Data Augmentation in Deep Reinforcement Learning. arXiv: 2402.12181."},{"key":"10.1016\/j.eswa.2026.132126_bib0020","doi-asserted-by":"crossref","first-page":"201","DOI":"10.1613\/jair.1.14174","article-title":"A survey of zero-shot generalisation in deep reinforcement learning","volume":"76","author":"Kirk","year":"2023","journal-title":"Journal of Artificial Intelligence Research"},{"key":"10.1016\/j.eswa.2026.132126_bib0021","unstructured":"Korkmaz, E. (2024). A Survey Analyzing Generalization in Deep Reinforcement Learning. arXiv: 2401.02349."},{"key":"10.1016\/j.eswa.2026.132126_bib0022","unstructured":"Kostrikov, I., Yarats, D., & Fergus, R. (2020). Image augmentation is all you need: Regularizing deep reinforcement learning from pixels. arXiv: 2004.13649."},{"key":"10.1016\/j.eswa.2026.132126_bib0023","unstructured":"Lan, C. L., Tu, S., Oberman, A., Agarwal, R., & Bellemare, M. G. (2022). On the generalization of representations in reinforcement learning. arXiv: 2203.00543."},{"key":"10.1016\/j.eswa.2026.132126_bib0024","first-page":"19884","article-title":"Reinforcement learning with augmented data","volume":"33","author":"Laskin","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132126_bib0025","series-title":"International conference on machine learning","first-page":"5639","article-title":"Curl: Contrastive unsupervised representations for reinforcement learning","author":"Laskin","year":"2020"},{"key":"10.1016\/j.eswa.2026.132126_bib0026","unstructured":"Li, L., Lyu, J., Ma, G., Wang, Z., Yang, Z., Li, X., & Li, Z. (2023). Normalization enhances generalization in visual reinforcement learning. arXiv: 2306.00656."},{"key":"10.1016\/j.eswa.2026.132126_bib0027","doi-asserted-by":"crossref","unstructured":"Liang, A., Thomason, J., & B\u0131y\u0131k, E. (2024). ViSaRL: Visual Reinforcement Learning Guided by Human Saliency. arXiv: 2403.10940.","DOI":"10.1109\/IROS58592.2024.10801388"},{"key":"10.1016\/j.eswa.2026.132126_bib0028","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"23436","article-title":"Improving generalization in visual reinforcement learning via conflict-aware gradient agreement augmentation","author":"Liu","year":"2023"},{"key":"10.1016\/j.eswa.2026.132126_bib0029","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1613\/jair.1.16422","article-title":"Understanding what affects the generalization gap in visual reinforcement learning: Theory and empirical evidence","volume":"81","author":"Lyu","year":"2024","journal-title":"Journal of Artificial Intelligence Research"},{"key":"10.1016\/j.eswa.2026.132126_bib0030","first-page":"34689","article-title":"Understanding the generalization benefit of normalization layers: Sharpness reduction","volume":"35","author":"Lyu","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132126_bib0031","unstructured":"Montenegro, A., Mussi, M., Metelli, A. M., & Papini, M. (2024). Learning Optimal Deterministic Policies with Stochastic Policy Gradients. arXiv: 2405.02235."},{"key":"10.1016\/j.eswa.2026.132126_bib0032","doi-asserted-by":"crossref","DOI":"10.1016\/j.array.2022.100258","article-title":"Data augmentation: A comprehensive survey of modern approaches","volume":"16","author":"Mumuni","year":"2022","journal-title":"Array"},{"key":"10.1016\/j.eswa.2026.132126_bib0033","unstructured":"Raileanu, R., Goldstein, M., Yarats, D., Kostrikov, I., & Fergus, R. (2020). Automatic data augmentation for generalization in deep reinforcement learning. arXiv: 2006.12862."},{"key":"10.1016\/j.eswa.2026.132126_bib0034","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., Radford, A., & Klimov, O. (2017). Proximal policy optimization algorithms. arXiv: 1707.06347."},{"issue":"4","key":"10.1016\/j.eswa.2026.132126_bib0035","doi-asserted-by":"crossref","first-page":"2443","DOI":"10.3390\/app13042443","article-title":"Reinforcement learning in game industry-review, prospects and challenges","volume":"13","author":"Souchleris","year":"2023","journal-title":"Applied Sciences"},{"key":"10.1016\/j.eswa.2026.132126_bib0036","series-title":"International conference on machine learning","first-page":"9870","article-title":"Decoupling representation learning from reinforcement learning","author":"Stooke","year":"2021"},{"key":"10.1016\/j.eswa.2026.132126_bib0037","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"2366","article-title":"Rethinking data augmentation for single-source domain generalization in medical image segmentation","volume":"vol. 37","author":"Su","year":"2023"},{"key":"10.1016\/j.eswa.2026.132126_bib0038","series-title":"2015 ieee information theory workshop (itw)","first-page":"1","article-title":"Deep learning and the information bottleneck principle","author":"Tishby","year":"2015"},{"key":"10.1016\/j.eswa.2026.132126_bib0039","series-title":"2014 information theory and applications workshop (ITA)","first-page":"1","article-title":"Total variation distance and the distribution of relative information","author":"Verd\u00fa","year":"2014"},{"key":"10.1016\/j.eswa.2026.132126_bib0040","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"5590","article-title":"What effects the generalization in visual reinforcement learning: policy consistency with truncated return prediction","volume":"vol. 38","author":"Wang","year":"2024"},{"key":"10.1016\/j.eswa.2026.132126_bib0041","series-title":"Proceedings of the thirty-third international joint conference on artificial intelligence","first-page":"1389","article-title":"How to learn domain-invariant representations for visual reinforcement learning: an information-theoretical perspective","author":"Wang","year":"2024"},{"key":"10.1016\/j.eswa.2026.132126_bib0042","series-title":"Proceedings of the 2024 4th international conference on internet of things and machine learning","first-page":"211","article-title":"Research on autonomous driving decision-making strategies based deep reinforcement learning","author":"Wang","year":"2024"},{"key":"10.1016\/j.eswa.2026.132126_bib0043","unstructured":"Wang, Z., Ze, Y., Sun, Y., Yuan, Z., & Xu, H. (2023). Generalizable visual reinforcement learning with segment anything model. arXiv: 2312.17116."},{"key":"10.1016\/j.eswa.2026.132126_bib0044","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"8683","article-title":"Generalizing reinforcement learning through fusing self-supervised learning into intrinsic motivation","volume":"vol. 36","author":"Wu","year":"2022"},{"key":"10.1016\/j.eswa.2026.132126_bib0045","unstructured":"Yarats, D., Fergus, R., Lazaric, A., & Pinto, L. (2021). Mastering visual continuous control: Improved data-augmented reinforcement learning. arXiv: 2107.09645."},{"key":"10.1016\/j.eswa.2026.132126_bib0046","doi-asserted-by":"crossref","unstructured":"Yuan, Z., Ma, G., Mu, Y., Xia, B., Yuan, B., Wang, X., Luo, P., & Xu, H. (2022a). Don\u2019t Touch What Matters: Task-Aware Lipschitz Data Augmentation for Visual Reinforcement Learning. arXiv: 2202.09982.","DOI":"10.24963\/ijcai.2022\/514"},{"key":"10.1016\/j.eswa.2026.132126_bib0047","first-page":"13022","article-title":"Pre-trained image encoder for generalizable visual reinforcement learning","volume":"35","author":"Yuan","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132126_bib0048","article-title":"Rl-vigen: A reinforcement learning benchmark for visual generalization","volume":"36","author":"Yuan","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132126_bib0049","series-title":"International conference on machine learning","first-page":"11214","article-title":"Invariant causal prediction for block mdps","author":"Zhang","year":"2020"},{"key":"10.1016\/j.eswa.2026.132126_bib0050","unstructured":"Zhang, A., McAllister, R., Calandra, R., Gal, Y., & Levine, S. (2020b). Learning invariant representations for reinforcement learning without reconstruction. arXiv: 2006.10742."},{"issue":"3","key":"10.1016\/j.eswa.2026.132126_bib0051","first-page":"3421","article-title":"Masked contrastive representation learning for reinforcement learning","volume":"45","author":"Zhu","year":"2022","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426010390?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426010390?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,11]],"date-time":"2026-06-11T15:54:55Z","timestamp":1781193295000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426010390"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":51,"alternative-id":["S0957417426010390"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132126","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Improving generalization in visual reinforcement learning through data distribution","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132126","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Published by Elsevier Ltd.","name":"copyright","label":"Copyright"}],"article-number":"132126"}}