{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T12:04:28Z","timestamp":1784203468413,"version":"3.55.0"},"reference-count":53,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100005046","name":"Natural Science Foundation of Heilongjiang Province","doi-asserted-by":"publisher","award":["YQ2024F007"],"award-info":[{"award-number":["YQ2024F007"]}],"id":[{"id":"10.13039\/501100005046","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100005046","name":"Natural Science Foundation of Heilongjiang Province","doi-asserted-by":"publisher","award":["YESS 20240415"],"award-info":[{"award-number":["YESS 20240415"]}],"id":[{"id":"10.13039\/501100005046","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100005046","name":"Natural Science Foundation of Heilongjiang Province","doi-asserted-by":"publisher","award":["2024QNRC001"],"award-info":[{"award-number":["2024QNRC001"]}],"id":[{"id":"10.13039\/501100005046","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62306088"],"award-info":[{"award-number":["62306088"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Neural Networks"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.neunet.2026.109019","type":"journal-article","created":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T23:35:47Z","timestamp":1776900947000},"page":"109019","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Curriculum reinforcement learning with measurable task representation learning"],"prefix":"10.1016","volume":"202","author":[{"given":"Yongyan","family":"Wen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7965-598X","authenticated-orcid":false,"given":"Siyuan","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4920-7235","authenticated-orcid":false,"given":"Mingjian","family":"Fu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-8541-1178","authenticated-orcid":false,"given":"Yiqin","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xun","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Peng","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.neunet.2026.109019_bib0001","unstructured":"Akkaya, I., Andrychowicz, M., Chociej, M., Litwin, M., McGrew, B., Petron, A., Paino, A., Plappert, M., Powell, G., Ribas, R., et al. (2019). Solving rubik\u2019s cube with a robot hand. arXiv: 1910.07113."},{"key":"10.1016\/j.neunet.2026.109019_bib0002","series-title":"Proceedings of the 39th international conference on machine learning","first-page":"822","article-title":"EAT-C: Environment-adversarial sub-task curriculum for efficient reinforcement learning","author":"Ao","year":"2022"},{"key":"10.1016\/j.neunet.2026.109019_bib0003","series-title":"Proceedings of the 40th international conference on machine learning","first-page":"1361","article-title":"CLUTR: Curriculum learning via unsupervised task representation learning","volume":"vol. 202","author":"Azad","year":"2023"},{"key":"10.1016\/j.neunet.2026.109019_bib0004","series-title":"Proceedings of the 26th annual international conference on machine learning","first-page":"41","article-title":"Curriculum learning","author":"Bengio","year":"2009"},{"key":"10.1016\/j.neunet.2026.109019_bib0005","series-title":"Proceedings of the 20th SIGNLL conference on computational natural language learning","first-page":"10","article-title":"Generating sentences from a continuous space","author":"Bowman","year":"2016"},{"key":"10.1016\/j.neunet.2026.109019_bib0006","series-title":"International conference on learning representations","article-title":"Learning with AMIGo: Adversarially motivated intrinsic goals","author":"Campero","year":"2020"},{"key":"10.1016\/j.neunet.2026.109019_bib0007","series-title":"Proceedings of the european conference on computer vision (ECCV)","first-page":"132","article-title":"Deep clustering for unsupervised learning of visual features","author":"Caron","year":"2018"},{"key":"10.1016\/j.neunet.2026.109019_bib0008","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"10069","article-title":"Scalable methods for computing state similarity in deterministic markov decision processes","volume":"vol. 34","author":"Castro","year":"2020"},{"key":"10.1016\/j.neunet.2026.109019_bib0009","first-page":"9681","article-title":"Variational automatic curriculum learning for sparse-reward cooperative multi-agent problems","volume":"34","author":"Chen","year":"2021","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.neunet.2026.109019_bib0010","doi-asserted-by":"crossref","unstructured":"Chevalier-Boisvert, M., Dai, B., Towers, M., de Lazcano, R., Willems, L., Lahlou, S., Pal, S., Castro, P. S., & Terry, J. (2023). Minigrid & miniworld: Modular & customizable reinforcement learning environments for goal-oriented tasks. CoRR. arXiv:abs\/2306.13831.","DOI":"10.52202\/075280-3209"},{"key":"10.1016\/j.neunet.2026.109019_bib0011","series-title":"The eleventh international conference on learning representations","article-title":"Outcome-directed reinforcement learning by uncertainty\\& temporal distance-aware curriculum goal generation","author":"Cho","year":"2022"},{"key":"10.1016\/j.neunet.2026.109019_bib0012","series-title":"Proceedings of the 7th conference on robot learning","first-page":"2184","article-title":"Seeing-eye quadruped navigation with force responsive locomotion control","volume":"vol. 229","author":"DeFazio","year":"2023"},{"key":"10.1016\/j.neunet.2026.109019_bib0013","first-page":"13049","article-title":"Emergent complexity and zero-shot transfer via unsupervised environment design","volume":"33","author":"Dennis","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.neunet.2026.109019_bib0014","series-title":"International conference on learning representations","article-title":"Adaptive procedural task generation for hard-exploration problems","author":"Fang","year":"2020"},{"key":"10.1016\/j.neunet.2026.109019_bib0015","series-title":"International conference on machine learning","first-page":"1515","article-title":"Automatic goal generation for reinforcement learning agents","author":"Florensa","year":"2018"},{"key":"10.1016\/j.neunet.2026.109019_bib0016","series-title":"Conference on robot learning","first-page":"482","article-title":"Reverse curriculum generation for reinforcement learning","author":"Florensa","year":"2017"},{"key":"10.1016\/j.neunet.2026.109019_bib0017","series-title":"Advances in neural information processing systems","article-title":"Generative adversarial nets","volume":"vol. 27","author":"Goodfellow","year":"2014"},{"key":"10.1016\/j.neunet.2026.109019_bib0018","series-title":"International conference on artificial neural networks","first-page":"799","article-title":"Bidirectional LSTM networks for improved phoneme classification and recognition","author":"Graves","year":"2005"},{"key":"10.1016\/j.neunet.2026.109019_bib0019","series-title":"International conference on machine learning","first-page":"1861","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","author":"Haarnoja","year":"2018"},{"key":"10.1016\/j.neunet.2026.109019_bib0020","unstructured":"Hallak, A., Di Castro, D., & Mannor, S. (2015). Contextual markov decision processes. arXiv: 1502.02259."},{"key":"10.1016\/j.neunet.2026.109019_bib0021","doi-asserted-by":"crossref","first-page":"10656","DOI":"10.52202\/068431-0774","article-title":"Curriculum reinforcement learning using optimal transport via gradual domain adaptation","volume":"35","author":"Huang","year":"2022","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.neunet.2026.109019_bib0022","article-title":"Unsupervised curricula for visual meta-reinforcement learning","volume":"32","author":"Jabri","year":"2019","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.neunet.2026.109019_bib0023","series-title":"International conference on machine learning","first-page":"4940","article-title":"Prioritized level replay","author":"Jiang","year":"2021"},{"key":"10.1016\/j.neunet.2026.109019_bib0024","series-title":"International conference on machine learning","first-page":"16668","article-title":"Variational curriculum reinforcement learning for unsupervised discovery of skills","author":"Kim","year":"2023"},{"key":"10.1016\/j.neunet.2026.109019_bib0025","unstructured":"Kingma, D. P., & Welling, M. (2013). Auto-encoding variational bayes. arXiv: 1312.6114."},{"issue":"1","key":"10.1016\/j.neunet.2026.109019_bib0026","article-title":"A probabilistic interpretation of self-paced learning with applications to reinforcement learning","volume":"22","author":"Klink","year":"2021","journal-title":"Journal of Machine Learning Research"},{"key":"10.1016\/j.neunet.2026.109019_bib0027","series-title":"Proceedings of the conference on robot learning","first-page":"513","article-title":"Self-paced contextual reinforcement learning","volume":"vol. 100","author":"Klink","year":"2020"},{"key":"10.1016\/j.neunet.2026.109019_bib0028","first-page":"9216","article-title":"Self-paced deep reinforcement learning","volume":"33","author":"Klink","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.neunet.2026.109019_bib0029","series-title":"International conference on machine learning","first-page":"11341","article-title":"Curriculum reinforcement learning via constrained optimal transport","author":"Klink","year":"2022"},{"key":"10.1016\/j.neunet.2026.109019_bib0030","series-title":"International conference on machine learning","first-page":"20412","article-title":"Understanding the complexity gains of single-task rl with a curriculum","author":"Li","year":"2023"},{"key":"10.1016\/j.neunet.2026.109019_bib0031","series-title":"Decision awareness in reinforcement learning workshop at ICML 2022","article-title":"Task factorization in curriculum learning","author":"Mirsky","year":"2022"},{"key":"10.1016\/j.neunet.2026.109019_bib0032","doi-asserted-by":"crossref","first-page":"529","DOI":"10.1038\/nature14236","article-title":"Human-level control through deep reinforcement learning","volume":"518","author":"Mnih","year":"2015","journal-title":"Nature"},{"issue":"1","key":"10.1016\/j.neunet.2026.109019_bib0033","first-page":"7382","article-title":"Curriculum learning for reinforcement learning domains: A framework and survey","volume":"21","author":"Narvekar","year":"2020","journal-title":"The Journal of Machine Learning Research"},{"key":"10.1016\/j.neunet.2026.109019_bib0034","series-title":"International conference on machine learning","first-page":"17473","article-title":"Evolving curricula with regret-based environment design","author":"Parker-Holder","year":"2022"},{"key":"10.1016\/j.neunet.2026.109019_bib0035","series-title":"Conference on robot learning","first-page":"835","article-title":"Teacher algorithms for curriculum learning of deep rl in continuously parameterized environments","author":"Portelas","year":"2020"},{"key":"10.1016\/j.neunet.2026.109019_bib0036","series-title":"International conference on learning representations","article-title":"Automated curriculum generation through setter-solver interactions","author":"Racaniere","year":"2019"},{"issue":"268","key":"10.1016\/j.neunet.2026.109019_bib0037","first-page":"1","article-title":"Stable-baselines3: Reliable reinforcement learning implementations","volume":"22","author":"Raffin","year":"2021","journal-title":"Journal of Machine Learning Research"},{"key":"10.1016\/j.neunet.2026.109019_bib0038","series-title":"Proceedings of the 37th international conference on machine learning","first-page":"7920","article-title":"Fast adaptation to new environments via policy-dynamics value functions","author":"Raileanu","year":"2020"},{"key":"10.1016\/j.neunet.2026.109019_bib0039","series-title":"Proceedings of the 36th international conference on machine learning","first-page":"5331","article-title":"Efficient off-policy meta-reinforcement learning via probabilistic context variables","author":"Rakelly","year":"2019"},{"key":"10.1016\/j.neunet.2026.109019_bib0040","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., Radford, A., & Klimov, O., et al. (2017). Proximal Policy Optimization Algorithms."},{"issue":"6419","key":"10.1016\/j.neunet.2026.109019_bib0041","doi-asserted-by":"crossref","first-page":"1140","DOI":"10.1126\/science.aar6404","article-title":"A general reinforcement learning algorithm that masters chess, shogi, and go through self-play","volume":"362","author":"Silver","year":"2018","journal-title":"Science"},{"key":"10.1016\/j.neunet.2026.109019_bib0042","series-title":"Proceedings of the 30th international conference on machine learning","first-page":"1139","article-title":"On the importance of initialization and momentum in deep learning","volume":"vol. 28","author":"Sutskever","year":"2013"},{"key":"10.1016\/j.neunet.2026.109019_bib0043","series-title":"Reinforcement learning: An introduction","author":"Sutton","year":"2018"},{"key":"10.1016\/j.neunet.2026.109019_bib0044","series-title":"2012\u202fIEEE\/RSJ International conference on intelligent robots and systems","first-page":"5026","article-title":"Mujoco: A physics engine for model-based control","author":"Todorov","year":"2012"},{"key":"10.1016\/j.neunet.2026.109019_bib0045","unstructured":"Wang, R., Lehman, J., Clune, J., & Stanley, K. O. (2019). Paired open-ended trailblazer (poet): Endlessly generating increasingly complex and diverse learning environments and their solutions. arXiv: 1901.01753."},{"key":"10.1016\/j.neunet.2026.109019_bib0046","article-title":"Robust imitation of diverse behaviors","volume":"30","author":"Wang","year":"2017","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.neunet.2026.109019_bib0047","series-title":"Proceedings of the 39th international conference on machine learning","first-page":"24177","article-title":"Robust deep reinforcement learning through bootstrapped opportunistic curriculum","author":"Wu","year":"2022"},{"key":"10.1016\/j.neunet.2026.109019_bib0048","series-title":"International conference on learning representations","article-title":"Learning invariant representations for reinforcement learning without reconstruction","author":"Zhang","year":"2021"},{"key":"10.1016\/j.neunet.2026.109019_bib0049","series-title":"International conference on learning representations","article-title":"C-Planning: An automatic curriculum for learning goal-reaching tasks","author":"Zhang","year":"2021"},{"key":"10.1016\/j.neunet.2026.109019_bib0050","first-page":"7648","article-title":"Automatic curriculum learning through value disagreement","volume":"33","author":"Zhang","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.neunet.2026.109019_bib0051","series-title":"2019 International conference on robotics and automation (ICRA)","first-page":"3651","article-title":"Dexterous manipulation with deep reinforcement learning: Efficient, general, and low-cost","author":"Zhu","year":"2019"},{"key":"10.1016\/j.neunet.2026.109019_bib0052","series-title":"Conference on robot learning","first-page":"73","article-title":"Robot parkour learning","author":"Zhuang","year":"2023"},{"key":"10.1016\/j.neunet.2026.109019_bib0053","series-title":"International conference on learning representations","article-title":"VariBAD: A very good method for bayes-adaptive deep RL via meta-learning","author":"Zintgraf","year":"2020"}],"container-title":["Neural Networks"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S089360802600479X?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S089360802600479X?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T11:19:50Z","timestamp":1784200790000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S089360802600479X"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":53,"alternative-id":["S089360802600479X"],"URL":"https:\/\/doi.org\/10.1016\/j.neunet.2026.109019","relation":{},"ISSN":["0893-6080"],"issn-type":[{"value":"0893-6080","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Curriculum reinforcement learning with measurable task representation learning","name":"articletitle","label":"Article Title"},{"value":"Neural Networks","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.neunet.2026.109019","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"109019"}}