{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T01:31:06Z","timestamp":1784511066121,"version":"3.55.0"},"reference-count":34,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"5","license":[{"start":{"date-parts":[[2025,10,1]],"date-time":"2025-10-01T00:00:00Z","timestamp":1759276800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,10,1]],"date-time":"2025-10-01T00:00:00Z","timestamp":1759276800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,10,1]],"date-time":"2025-10-01T00:00:00Z","timestamp":1759276800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100000923","name":"Australian Research Council","doi-asserted-by":"publisher","award":["DP220100803"],"award-info":[{"award-number":["DP220100803"]}],"id":[{"id":"10.13039\/501100000923","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100000923","name":"Australian Research Council","doi-asserted-by":"publisher","award":["DP250103612"],"award-info":[{"award-number":["DP250103612"]}],"id":[{"id":"10.13039\/501100000923","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Australian Cooperative Research Centres Projects","award":["CRCPXI000007"],"award-info":[{"award-number":["CRCPXI000007"]}]},{"name":"Australia Defence Innovation Hub","award":["P18-650825"],"award-info":[{"award-number":["P18-650825"]}]},{"name":"AFOSR &#x2013; DST Australian Autonomy Initiative","award":["ID10134"],"award-info":[{"award-number":["ID10134"]}]},{"DOI":"10.13039\/501100021670","name":"NSW Defence Innovation Network","doi-asserted-by":"publisher","award":["DINPP2019 S1-03\/09"],"award-info":[{"award-number":["DINPP2019 S1-03\/09"]}],"id":[{"id":"10.13039\/501100021670","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Emerg. Top. Comput. Intell."],"published-print":{"date-parts":[[2025,10]]},"DOI":"10.1109\/tetci.2025.3595684","type":"journal-article","created":{"date-parts":[[2025,8,13]],"date-time":"2025-08-13T17:36:16Z","timestamp":1755106576000},"page":"3719-3726","source":"Crossref","is-referenced-by-count":4,"title":["Contrastive Learning-Based Agent Modeling for Deep Reinforcement Learning"],"prefix":"10.1109","volume":"9","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2350-4597","authenticated-orcid":false,"given":"Wenhao","family":"Ma","sequence":"first","affiliation":[{"name":"Australian AI Institute, School of Computer Science, Faculty of Engineering and Information Technology, University of Technology Sydney, Ultimo, NSW, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9244-0318","authenticated-orcid":false,"given":"Yu-Chen","family":"Chang","sequence":"additional","affiliation":[{"name":"Australian AI Institute, School of Computer Science, Faculty of Engineering and Information Technology, University of Technology Sydney, Ultimo, NSW, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3109-472X","authenticated-orcid":false,"given":"Jie","family":"Yang","sequence":"additional","affiliation":[{"name":"Australian AI Institute, School of Computer Science, Faculty of Engineering and Information Technology, University of Technology Sydney, Ultimo, NSW, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8390-2664","authenticated-orcid":false,"given":"Yu-Kai","family":"Wang","sequence":"additional","affiliation":[{"name":"Australian AI Institute, School of Computer Science, Faculty of Engineering and Information Technology, University of Technology Sydney, Ultimo, NSW, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8371-8197","authenticated-orcid":false,"given":"Chin-Teng","family":"Lin","sequence":"additional","affiliation":[{"name":"Australian AI Institute, School of Computer Science, Faculty of Engineering and Information Technology, University of Technology Sydney, Ultimo, NSW, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1177\/0018720820960865"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1145\/3627676.3627678"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1016\/j.artint.2006.02.006"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1016\/j.artint.2018.01.002"},{"key":"ref5","article-title":"A survey of progress on cooperative multi-agent reinforcement learning in open environment","author":"Yuan","year":"2023"},{"key":"ref6","first-page":"1804","article-title":"Opponent modeling in deep reinforcement learning","volume-title":"Proc. 33rd Int. Conf. Mach. Learn.","author":"He","year":"2016"},{"key":"ref7","first-page":"1802","article-title":"Learning policy representations in multiagent systems","volume-title":"Proc. 35th Int. Conf. Mach. Learn.","author":"Grover","year":"2018"},{"key":"ref8","first-page":"19210","article-title":"Agent modelling under partial observability for deep reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Papoudakis","year":"2021"},{"key":"ref9","volume-title":"Reinforcement Learning: An Introduction","author":"Sutton","year":"2018"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.5555\/3524938.3525087"},{"key":"ref12","first-page":"4218","article-title":"Machine theory of mind","volume-title":"Proc. 35th Int. Conf. Mach. Learn.","author":"Rabinowitz","year":"2018"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2016.7759578"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.2307\/j.ctt4cgngj.10"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2022.3176413"},{"key":"ref16","article-title":"Temporal abstraction in reinforcement learning","author":"Precup","year":"2000"},{"key":"ref17","article-title":"Improving context-based meta-reinforcement learning with self-supervised trajectory contrastive learning","author":"Wang","year":"2021"},{"key":"ref18","article-title":"Representation learning with contrastive predictive coding","author":"Oord","year":"2018"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.3390\/technologies9010002"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"ref21","first-page":"9912","article-title":"Unsupervised learning of visual features by contrasting cluster assignments","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Caron","year":"2020"},{"key":"ref22","first-page":"709","article-title":"Dynamic programming for partially observable stochastic games","volume-title":"Proc. AAAI Conf. Artif. Intell.","author":"Hansen","year":"2004"},{"key":"ref23","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2018"},{"key":"ref24","article-title":"Improving language understanding by generative pre-training","author":"Radford","year":"2018"},{"key":"ref25","article-title":"An image is worth 16 x 16 words: Transformers for image recognition at scale","author":"Dosovitskiy","year":"2020"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01607"},{"key":"ref27","article-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017"},{"key":"ref28","article-title":"Playing Atari with deep reinforcement learning","author":"Mnih","year":"2013"},{"key":"ref29","first-page":"10707","article-title":"Shared experience actor-critic for multi-agent reinforcement learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Christianos","year":"2020"},{"key":"ref30","article-title":"Benchmarking multi-agent deep reinforcement learning algorithms in cooperative tasks","author":"Papoudakis","year":"2020"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11492"},{"key":"ref32","article-title":"The StarCraft multi-agent challenge","author":"Samvelyan","year":"2019"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i10.26388"},{"issue":"11","key":"ref34","first-page":"2579","article-title":"Visualizing data using t-SNE","volume":"9","author":"Maaten","year":"2008","journal-title":"J. Mach. Learn. Res."}],"container-title":["IEEE Transactions on Emerging Topics in Computational Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/7433297\/11177632\/11123723.pdf?arnumber=11123723","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,30]],"date-time":"2025-09-30T16:09:22Z","timestamp":1759248562000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11123723\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10]]},"references-count":34,"journal-issue":{"issue":"5"},"URL":"https:\/\/doi.org\/10.1109\/tetci.2025.3595684","relation":{},"ISSN":["2471-285X"],"issn-type":[{"value":"2471-285X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,10]]}}}