{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T05:03:41Z","timestamp":1750309421805,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":45,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,21]],"date-time":"2024-10-21T00:00:00Z","timestamp":1729468800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,21]]},"DOI":"10.1145\/3627673.3679740","type":"proceedings-article","created":{"date-parts":[[2024,10,20]],"date-time":"2024-10-20T19:34:21Z","timestamp":1729452861000},"page":"3453-3462","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Learning Cross-modal Knowledge Reasoning and Heuristic-prompt for Visual-language Navigation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5669-1408","authenticated-orcid":false,"given":"Dongming","family":"Zhou","sequence":"first","affiliation":[{"name":"School of Computer Science, National University of Defense Technology, Changsha, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2046-5043","authenticated-orcid":false,"given":"Zhengbin","family":"Pang","sequence":"additional","affiliation":[{"name":"School of Computer Science, National University of Defence Technology, Changsha, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-1058-3287","authenticated-orcid":false,"given":"Wei","family":"Li","sequence":"additional","affiliation":[{"name":"School of Computer and Electronic Information, Guangxi University, Nanning, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,10,21]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Matterport3d: Learning from rgb-d data in indoor environments. arXiv preprint arXiv:1709.06158","author":"Chang Angel","year":"2017","unstructured":"Angel Chang, Angela Dai, Thomas Funkhouser, Maciej Halber, Matthias Niessner, Manolis Savva, Shuran Song, Andy Zeng, and Yinda Zhang. 2017. Matterport3d: Learning from rgb-d data in indoor environments. arXiv preprint arXiv:1709.06158 (2017)."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01501"},{"key":"e_1_3_2_1_3_1","volume-title":"History aware multimodal transformer for vision-and-language navigation. Advances in neural information processing systems","author":"Chen Shizhe","year":"2021","unstructured":"Shizhe Chen, Pierre-Louis Guhur, Cordelia Schmid, and Ivan Laptev. 2021. History aware multimodal transformer for vision-and-language navigation. Advances in neural information processing systems, Vol. 34 (2021), 5834--5847."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01604"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01106"},{"key":"e_1_3_2_1_6_1","volume-title":"Enhancing machine vision: the impact of a novel innovative technology on video question-answering. Soft Computing","author":"Dan Songjian","year":"2024","unstructured":"Songjian Dan and Wei Feng. 2024. Enhancing machine vision: the impact of a novel innovative technology on video question-answering. Soft Computing (2024), 1--14."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3561533"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.ipm.2023.103637"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.ipm.2023.103639"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00166"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2023.110096"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01315"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00169"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10160969"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00324"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2023.102122"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.engappai.2022.105618"},{"key":"e_1_3_2_1_18_1","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Li Jialu","year":"2024","unstructured":"Jialu Li and Mohit Bansal. 2024. Panogen: Text-conditioned panoramic environment generation for vision-and-language navigation. Advances in Neural Information Processing Systems, Vol. 36 (2024)."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i1.25223"},{"key":"e_1_3_2_1_20_1","volume-title":"Visual-language navigation pretraining via prompt-based environmental self-exploration. arXiv preprint arXiv:2203.04006","author":"Liang Xiwen","year":"2022","unstructured":"Xiwen Liang, Fengda Zhu, Lingling Li, Hang Xu, and Xiaodan Liang. 2022. Visual-language navigation pretraining via prompt-based environmental self-exploration. arXiv preprint arXiv:2203.04006 (2022)."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2023.3273594"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01496"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00696"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01544"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01411"},{"key":"e_1_3_2_1_26_1","volume-title":"Prompt-based Context-and Domain-aware Pretraining for Vision and Language Navigation. arXiv preprint arXiv:2309.03661","author":"Liu Ting","year":"2023","unstructured":"Ting Liu, Wansen Wu, Yue Hu, Youkai Wang, Kai Xu, and Quanjun Yin. 2023. Prompt-based Context-and Domain-aware Pretraining for Vision and Language Navigation. arXiv preprint arXiv:2309.03661 (2023)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2023.3284038"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2021.11.031"},{"key":"e_1_3_2_1_29_1","volume-title":"Self-monitoring navigation agent via auxiliary progress estimation. arXiv preprint arXiv:1901.03035","author":"Ma Chih-Yao","year":"2019","unstructured":"Chih-Yao Ma, Jiasen Lu, Zuxuan Wu, Ghassan AlRegib, Zsolt Kira, Richard Socher, and Caiming Xiong. 2019. Self-monitoring navigation agent via auxiliary progress estimation. arXiv preprint arXiv:1901.03035 (2019)."},{"volume-title":"European Conference on Computer Vision. Springer, 303--317","author":"Qi Yuankai","key":"e_1_3_2_1_30_1","unstructured":"Yuankai Qi, Zizheng Pan, Shengping Zhang, Anton van den Hengel, and Qi Wu. 2020. Object-and-action aware model for visual language navigation. In European Conference on Computer Vision. Springer, 303--317."},{"key":"e_1_3_2_1_31_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 15418--15427","author":"Qiao Yanyuan","year":"2022","unstructured":"Yanyuan Qiao, Yuankai Qi, Yicong Hong, Zheng Yu, Peng Wang, and Qi Wu. 2022. Hop: history-and-order aware pre-training for vision-and-language navigation. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 15418--15427."},{"key":"e_1_3_2_1_32_1","volume-title":"International conference on machine learning. PMLR, 8748--8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al. 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PMLR, 8748--8763."},{"key":"e_1_3_2_1_33_1","volume-title":"Conference on Robot Learning. PMLR, 492--504","author":"Shah Dhruv","year":"2023","unstructured":"Dhruv Shah, B\u0142a.zej Osi'nski, Sergey Levine, et al. 2023. Lm-nav: Robotic navigation with large pre-trained models of language, vision, and action. In Conference on Robot Learning. PMLR, 492--504."},{"key":"e_1_3_2_1_34_1","volume-title":"Thirty-seventh Conference on Neural Information Processing Systems.","author":"Shinn Noah","year":"2023","unstructured":"Noah Shinn, Federico Cassano, Ashwin Gopinath, Karthik R Narasimhan, and Shunyu Yao. 2023. Reflexion: Language agents with verbal reinforcement learning. In Thirty-seventh Conference on Neural Information Processing Systems."},{"key":"e_1_3_2_1_35_1","volume-title":"Learning to navigate unseen environments: Back translation with environmental dropout. arXiv preprint arXiv:1904.04195","author":"Tan Hao","year":"2019","unstructured":"Hao Tan, Licheng Yu, and Mohit Bansal. 2019. Learning to navigate unseen environments: Back translation with environmental dropout. arXiv preprint arXiv:1904.04195 (2019)."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"crossref","unstructured":"Siyu Teng Xuemin Hu Peng Deng Bai Li Yuchen Li Yunfeng Ai Dongsheng Yang Lingxi Li Zhe Xuanyuan Fenghua Zhu et al. 2023. Motion planning for autonomous driving: The state of the art and future perspectives. IEEE Transactions on Intelligent Vehicles (2023).","DOI":"10.1109\/TIV.2023.3274536"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00998"},{"key":"e_1_3_2_1_38_1","volume-title":"A Dual Semantic-Aware Recurrent Global-Adaptive Network For Vision-and-Language Navigation. arXiv preprint arXiv:2305.03602","author":"Wang Liuyi","year":"2023","unstructured":"Liuyi Wang, Zongtao He, Jiagui Tang, Ronghao Dang, Naijia Wang, Chengju Liu, and Qijun Chen. 2023. A Dual Semantic-Aware Recurrent Global-Adaptive Network For Vision-and-Language Navigation. arXiv preprint arXiv:2305.03602 (2023)."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00679"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01103"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01432"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"crossref","unstructured":"Liqiang Zhang Long Yu Shengwei Tian and Qimeng Yang. 2022. Sequential verb metaphor detection with linguistic theories. 2765--2775 pages.","DOI":"10.3233\/JIFS-210381"},{"key":"e_1_3_2_1_43_1","volume-title":"Sub-Instruction and Local Map Relationship Enhanced Model for Vision and Language Navigation. In International Conference on Neural Information Processing. Springer, 518--529","author":"Zhang Yong","year":"2023","unstructured":"Yong Zhang, Yinlin Li, Jihe Bai, Yi Feng, and Mo Tao. 2023. Sub-Instruction and Local Map Relationship Enhanced Model for Vision and Language Navigation. In International Conference on Neural Information Processing. Springer, 518--529."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01250"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01003"}],"event":{"name":"CIKM '24: The 33rd ACM International Conference on Information and Knowledge Management","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"],"location":"Boise ID USA","acronym":"CIKM '24"},"container-title":["Proceedings of the 33rd ACM International Conference on Information and Knowledge Management"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3627673.3679740","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3627673.3679740","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T00:58:27Z","timestamp":1750294707000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3627673.3679740"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,21]]},"references-count":45,"alternative-id":["10.1145\/3627673.3679740","10.1145\/3627673"],"URL":"https:\/\/doi.org\/10.1145\/3627673.3679740","relation":{},"subject":[],"published":{"date-parts":[[2024,10,21]]},"assertion":[{"value":"2024-10-21","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}