{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T16:41:46Z","timestamp":1777653706156,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":52,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,11,25]],"date-time":"2022-11-25T00:00:00Z","timestamp":1669334400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"State Grid Shandong Electric Power Company","award":["520612220007"],"award-info":[{"award-number":["520612220007"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,11,25]]},"DOI":"10.1145\/3577164.3577178","type":"proceedings-article","created":{"date-parts":[[2023,4,4]],"date-time":"2023-04-04T22:10:10Z","timestamp":1680646210000},"page":"87-92","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":7,"title":["Object Goal Navigation in Eobodied AI: A Survey"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2117-1648","authenticated-orcid":false,"given":"Baosheng","family":"Li","sequence":"first","affiliation":[{"name":"State Grid Shandong Electric Power Company Laiwu Power Supply Company, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4328-8959","authenticated-orcid":false,"given":"Jishui","family":"Han","sequence":"additional","affiliation":[{"name":"State Grid Shandong Electric Power Company Laiwu Power Supply Company, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1646-9175","authenticated-orcid":false,"given":"Yuan","family":"Cheng","sequence":"additional","affiliation":[{"name":"State Grid Shandong Electric Power Company Laiwu Power Supply Company, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5770-3653","authenticated-orcid":false,"given":"Chong","family":"Tan","sequence":"additional","affiliation":[{"name":"State Grid Shandong Electric Power Company Laiwu Power Supply Company, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7140-6521","authenticated-orcid":false,"given":"Peng","family":"Qi","sequence":"additional","affiliation":[{"name":"State Grid Shandong Electric Power Company Laiwu Power Supply Company, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6658-8148","authenticated-orcid":false,"given":"Jianping","family":"Zhang","sequence":"additional","affiliation":[{"name":"State Grid Shandong Electric Power Company Laiwu Power Supply Company, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9848-9200","authenticated-orcid":false,"given":"Xiaolei","family":"Li","sequence":"additional","affiliation":[{"name":"School of Control Science and Engineering, Shandong University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,4,4]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2020.2965078"},{"key":"e_1_3_2_1_2_1","volume-title":"Ai2-thor: An interactive 3d environment for visual ai. arXiv preprint arXiv:1712.05474","author":"Kolve E.","year":"2017","unstructured":"Kolve, E., Mottaghi, R., Han, W., VanderBilt, E., Weihs, L., Herrasti, A., ... & Farhadi, A. (2017). Ai2-thor: An interactive 3d environment for visual ai. arXiv preprint arXiv:1712.05474."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00943"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00945"},{"key":"e_1_3_2_1_5_1","volume-title":"Matterport3d: Learning from rgb-d data in indoor environments. arXiv preprint arXiv:1709.06158","author":"Chang A.","year":"2017","unstructured":"Chang, A., Dai, A., Funkhouser, T., Halber, M., Niessner, M., Savva, M., ... & Zhang, Y. (2017). Matterport3d: Learning from rgb-d data in indoor environments. arXiv preprint arXiv:1709.06158."},{"key":"e_1_3_2_1_6_1","first-page":"9700","article-title":"Multion: Benchmarking semantic map memory using multi-object navigation","volume":"33","author":"Wani S.","year":"2020","unstructured":"Wani, S., Patel, S., Jain, U., Chang, A., & Savva, M. (2020). Multion: Benchmarking semantic map memory using multi-object navigation. Advances in Neural Information Processing Systems, 33, 9700-9712.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2021.3064461"},{"key":"e_1_3_2_1_8_1","volume-title":"On evaluation of embodied navigation agents. arXiv preprint arXiv:1807.06757","author":"Anderson P.","year":"2018","unstructured":"Anderson, P., Chang, A., Chaplot, D. S., Dosovitskiy, A., Gupta, S., Koltun, V., ... & Zamir, A. R. (2018). On evaluation of embodied navigation agents. arXiv preprint arXiv:1807.06757."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01003"},{"key":"e_1_3_2_1_10_1","first-page":"1","article-title":"Improving target-driven visual navigation with attention on 3D spatial relationships","author":"Lyu Y.","year":"2022","unstructured":"Lyu, Y., Shi, Y., & Zhang, X. (2022). Improving target-driven visual navigation with attention on 3D spatial relationships. Neural Processing Letters, 1-20.","journal-title":"Neural Processing Letters"},{"key":"e_1_3_2_1_11_1","volume-title":"Conference on Robot Learning (pp. 517-528)","author":"Pal A.","year":"2021","unstructured":"Pal, A., Qiu, Y., & Christensen, H. (2021, October). Learning hierarchical relationships for object-goal navigation. In Conference on Robot Learning (pp. 517-528). PMLR."},{"key":"e_1_3_2_1_12_1","volume-title":"European Conference on Computer Vision (pp. 19-34)","author":"Du H.","year":"2020","unstructured":"Du, H., Yu, X., & Zheng, L. (2020, August). Learning object relation graph and tentative policy for visual navigation. In European Conference on Computer Vision (pp. 19-34). Springer, Cham."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01485"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01250"},{"key":"e_1_3_2_1_15_1","volume-title":"Attention is all you need. Advances in neural information processing systems, 30","author":"Vaswani A.","year":"2017","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A. N., ... & Polosukhin, I. (2017). Attention is all you need. Advances in neural information processing systems, 30."},{"key":"e_1_3_2_1_16_1","volume-title":"European conference on computer vision (pp. 213-229)","author":"Carion N.","year":"2020","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., & Zagoruyko, S. (2020, August). End-to-end object detection with transformers. In European conference on computer vision (pp. 213-229). Springer, Cham."},{"key":"e_1_3_2_1_17_1","volume-title":"An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929","author":"Dosovitskiy A.","year":"2020","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., ... & Houlsby, N. (2020). An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01526"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01662"},{"key":"e_1_3_2_1_20_1","volume-title":"VTNet: Visual transformer network for object goal navigation. arXiv preprint arXiv:2105.09447","author":"Du H.","year":"2021","unstructured":"Du, H., Yu, X., & Zheng, L. (2021). VTNet: Visual transformer network for object goal navigation. arXiv preprint arXiv:2105.09447."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48506.2021.9561058"},{"key":"e_1_3_2_1_22_1","volume-title":"2020 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS) (pp. 5879-5885)","author":"Zhu K.","year":"2020","unstructured":"Zhu, K., Chen, W., Zhang, W., Song, R., & Li, Y. (2020, October). Autonomous robot navigation based on multi-camera perception. In 2020 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS) (pp. 5879-5885). IEEE."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00297"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01581"},{"key":"e_1_3_2_1_25_1","volume-title":"Long short-term memory. Neural computation, 9(8), 1735-1780","author":"Hochreiter S.","year":"1997","unstructured":"Hochreiter, S., & Schmidhuber, J. (1997). Long short-term memory. Neural computation, 9(8), 1735-1780."},{"key":"e_1_3_2_1_26_1","volume-title":"Learning phrase representations using RNN encoder-decoder for statistical machine translation. arXiv preprint arXiv:1406.1078","author":"Cho K.","year":"2014","unstructured":"Cho, K., Van Merri\u00ebnboer, B., Gulcehre, C., Bahdanau, D., Bougares, F., Schwenk, H., & Bengio, Y. (2014). Learning phrase representations using RNN encoder-decoder for statistical machine translation. arXiv preprint arXiv:1406.1078."},{"key":"e_1_3_2_1_27_1","volume-title":"Memory-augmented reinforcement learning for image-goal navigation. arXiv preprint arXiv:2101.05181","author":"Mezghani L.","year":"2021","unstructured":"Mezghani, L., Sukhbaatar, S., Lavril, T., Maksymets, O., Batra, D., Bojanowski, P., & Alahari, K. (2021). Memory-augmented reinforcement learning for image-goal navigation. arXiv preprint arXiv:2101.05181."},{"key":"e_1_3_2_1_28_1","volume-title":"Object Memory Transformer for Object Goal Navigation. arXiv preprint arXiv:2203.14708","author":"Fukushima R.","year":"2022","unstructured":"Fukushima, R., Ota, K., Kanezaki, A., Sasaki, Y., & Yoshiyasu, Y. (2022). Object Memory Transformer for Object Goal Navigation. arXiv preprint arXiv:2203.14708."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01559"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIM.2022.3158384"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00691"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01509"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2022.3140795"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01540"},{"key":"e_1_3_2_1_35_1","first-page":"5296","article-title":"Counterfactual vision-and-language navigation: Unravelling the unseen","volume":"33","author":"Parvaneh A.","year":"2020","unstructured":"Parvaneh, A., Abbasnejad, E., Teney, D., Shi, J. Q., & van den Hengel, A. (2020). Counterfactual vision-and-language navigation: Unravelling the unseen. Advances in Neural Information Processing Systems, 33, 5296-5307.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/TRO.2020.2994002"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01501"},{"key":"e_1_3_2_1_38_1","volume-title":"Learning to explore using active neural slam. arXiv preprint arXiv:2004.05155","author":"Chaplot D. S.","year":"2020","unstructured":"Chaplot, D. S., Gandhi, D., Gupta, S., Gupta, A., & Salakhutdinov, R. (2020). Learning to explore using active neural slam. arXiv preprint arXiv:2004.05155."},{"key":"e_1_3_2_1_39_1","first-page":"4247","article-title":"Object goal navigation using goal-oriented semantic exploration","volume":"33","author":"Chaplot D. S.","year":"2020","unstructured":"Chaplot, D. S., Gandhi, D. P., Gupta, A., & Salakhutdinov, R. R. (2020). Object goal navigation using goal-oriented semantic exploration. Advances in Neural Information Processing Systems, 33, 4247-4258.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_40_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (pp. 12875-12884)","author":"Chaplot D. S.","year":"2020","unstructured":"Chaplot, D. S., Salakhutdinov, R., Gupta, A., & Gupta, S. (2020). Neural topological slam for visual navigation. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (pp. 12875-12884)."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01445"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2022.3178810"},{"key":"e_1_3_2_1_43_1","volume-title":"Learning to map for active semantic goal navigation. arXiv preprint arXiv:2106.15648","author":"Georgakis G.","year":"2021","unstructured":"Georgakis, G., Bucher, B., Schmeckpeper, K., Singh, S., & Daniilidis, K. (2021). Learning to map for active semantic goal navigation. arXiv preprint arXiv:2106.15648."},{"key":"e_1_3_2_1_44_1","volume-title":"Navigating to Objects in Unseen Environments by Distance Prediction. arXiv preprint arXiv:2202.03735","author":"Zhu M.","year":"2022","unstructured":"Zhu, M., Zhao, B., & Kong, T. (2022). Navigating to Objects in Unseen Environments by Distance Prediction. arXiv preprint arXiv:2202.03735."},{"key":"e_1_3_2_1_45_1","first-page":"13086","article-title":"SEAL: Self-supervised embodied active learning using exploration and 3d consistency","volume":"34","author":"Chaplot D. S.","year":"2021","unstructured":"Chaplot, D. S., Dalal, M., Gupta, S., Malik, J., & Salakhutdinov, R. R. (2021). SEAL: Self-supervised embodied active learning using exploration and 3d consistency. Advances in Neural Information Processing Systems, 34, 13086-13098.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01832"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2020.3034524"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2022.3145964"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01565"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2022.3146912"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48506.2021.9561631"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/TRO.2016.2624754"}],"event":{"name":"VSIP 2022: 2022 4th International Conference on Video, Signal and Image Processing","location":"Shanghai China","acronym":"VSIP 2022"},"container-title":["Proceedings of the 2022 4th International Conference on Video, Signal and Image Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3577164.3577178","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3577164.3577178","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T17:51:10Z","timestamp":1750182670000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3577164.3577178"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,11,25]]},"references-count":52,"alternative-id":["10.1145\/3577164.3577178","10.1145\/3577164"],"URL":"https:\/\/doi.org\/10.1145\/3577164.3577178","relation":{},"subject":[],"published":{"date-parts":[[2022,11,25]]},"assertion":[{"value":"2023-04-04","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}