{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T15:03:28Z","timestamp":1781535808150,"version":"3.54.5"},"reference-count":52,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,10,14]],"date-time":"2024-10-14T00:00:00Z","timestamp":1728864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,10,14]],"date-time":"2024-10-14T00:00:00Z","timestamp":1728864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,10,14]]},"DOI":"10.1109\/iros58592.2024.10802484","type":"proceedings-article","created":{"date-parts":[[2024,12,25]],"date-time":"2024-12-25T19:17:39Z","timestamp":1735154259000},"page":"1443-1450","source":"Crossref","is-referenced-by-count":3,"title":["Multimodal Evolutionary Encoder for Continuous Vision-Language Navigation"],"prefix":"10.1109","author":[{"given":"Zongtao","family":"He","sequence":"first","affiliation":[{"name":"Tongji University,Department of Control Science and Engineering,Shanghai,China,201804"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Liuyi","family":"Wang","sequence":"additional","affiliation":[{"name":"Tongji University,Department of Control Science and Engineering,Shanghai,China,201804"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lu","family":"Chen","sequence":"additional","affiliation":[{"name":"Tongji University,Department of Control Science and Engineering,Shanghai,China,201804"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shu","family":"Li","sequence":"additional","affiliation":[{"name":"Tongji University,Department of Control Science and Engineering,Shanghai,China,201804"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qingqing","family":"Yan","sequence":"additional","affiliation":[{"name":"Tongji University,Department of Control Science and Engineering,Shanghai,China,201804"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chengju","family":"Liu","sequence":"additional","affiliation":[{"name":"Tongji University,Department of Control Science and Engineering,Shanghai,China,201804"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qijun","family":"Chen","sequence":"additional","affiliation":[{"name":"Tongji University,Department of Control Science and Engineering,Shanghai,China,201804"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00387"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58604-1_7"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/3DV.2017.00081"},{"key":"ref4","first-page":"3318","article-title":"Speaker-follower models for vision-and-language navigation","author":"Fried","year":"2018","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58539-6_16"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00166"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00169"},{"key":"ref9","first-page":"5834","article-title":"History aware multimodal transformer for vision-and-language navigation","volume":"34","author":"Chen","year":"2021","journal-title":"Advances in neural information processing systems"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01604"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2022.3233554"},{"key":"ref12","article-title":"Microsoft coco captions: Data collection and evaluation server","author":"Chen","year":"2015"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.670"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1238"},{"key":"ref15","article-title":"LAION-5b: An open large-scale dataset for training next generation image-text models","author":"Schuhmann","year":"2022","journal-title":"Thirty-sixth NeurIPS Datasets and Benchmarks Track"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.271"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01282"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2022.3141150"},{"key":"ref19","first-page":"4171","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","volume-title":"Proceedings of NAACL-HLT, Volume 1 (Long and Short Papers)","author":"Devlin"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2010.11929"},{"key":"ref21","article-title":"Vilbert: Pretraining task-agnostic visiolinguistic representations for vision-and-language tasks","volume":"32","author":"Lu","year":"2019","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01488"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19842-7_34"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01500"},{"key":"ref25","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"Proceedings of ICML","volume":"139","author":"Radford"},{"key":"ref26","article-title":"Dd-ppo: Learning near-perfect pointgoal navigators from 2.5 billion frames","volume-title":"Proceedings of ICLR","author":"Wijmans"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/W14-4012"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-016-0981-7"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00356"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.261"},{"issue":"1","key":"ref33","first-page":"5485","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","volume":"21","author":"Raffel","year":"2020","journal-title":"The Journal of Machine Learning Research"},{"key":"ref34","article-title":"Habitat-matterport 3d dataset (hm3d): 1000 large-scale 3d environments for embodied ai","author":"Ramakrishnan","year":"2021"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2013.458"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2012.6385773"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.292"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-33715-4_54"},{"key":"ref39","article-title":"DIODE: A Dense Indoor and Outdoor DEpth Dataset","author":"Vasiljevic","year":"2019","journal-title":"CoRR"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01041"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-long.41"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1287"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01075"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N19-1268"},{"key":"ref45","first-page":"627","article-title":"A reduction of imitation learning and structured prediction to no-regret online learning","volume-title":"Proceedings of AISTATS","volume":"15","author":"Ross"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01112"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/ICPR56361.2022.9956561"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01502"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.328"},{"issue":"86","key":"ref50","first-page":"2579","article-title":"Visualizing data using t-sne","volume":"9","author":"van der Maaten","year":"2008","journal-title":"Journal of Machine Learning Research"},{"key":"ref51","article-title":"Zoedepth: Zero-shot transfer by combining relative and metric depth","author":"Bhat","year":"2023"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2024.3435937"}],"event":{"name":"2024 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)","location":"Abu Dhabi, United Arab Emirates","start":{"date-parts":[[2024,10,14]]},"end":{"date-parts":[[2024,10,18]]}},"container-title":["2024 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10801246\/10801290\/10802484.pdf?arnumber=10802484","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,1,14]],"date-time":"2025-01-14T19:38:40Z","timestamp":1736883520000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10802484\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,14]]},"references-count":52,"URL":"https:\/\/doi.org\/10.1109\/iros58592.2024.10802484","relation":{},"subject":[],"published":{"date-parts":[[2024,10,14]]}}}