{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,3]],"date-time":"2026-08-03T18:57:37Z","timestamp":1785783457885,"version":"3.56.0"},"publisher-location":"Cham","reference-count":40,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031533044","type":"print"},{"value":"9783031533051","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024]]},"DOI":"10.1007\/978-3-031-53305-1_15","type":"book-chapter","created":{"date-parts":[[2024,1,27]],"date-time":"2024-01-27T21:37:36Z","timestamp":1706391456000},"page":"187-200","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["PANDA: Prompt-Based Context- and Indoor-Aware Pretraining for\u00a0Vision and\u00a0Language Navigation"],"prefix":"10.1007","author":[{"given":"Ting","family":"Liu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yue","family":"Hu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wansen","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Youkai","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kai","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Quanjun","family":"Yin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,1,28]]},"reference":[{"key":"15_CR1","doi-asserted-by":"crossref","unstructured":"Das, A., Datta, S., Gkioxari, G., Lee, S., Parikh, D., Batra, D.: Embodied question answering. In: Proceedings of CVPR, pp. 1\u201310 (2018)","DOI":"10.1109\/CVPR.2018.00008"},{"key":"15_CR2","doi-asserted-by":"crossref","unstructured":"Qi, Y., Wu, Q., Anderson, P., et al.: Reverie: remote embodied visual referring expression in real indoor environments. In: Proceedings of CVPR, pp. 9982\u20139991 (2020)","DOI":"10.1109\/CVPR42600.2020.01000"},{"key":"15_CR3","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"259","DOI":"10.1007\/978-3-030-58539-6_16","volume-title":"Computer Vision \u2013 ECCV 2020","author":"A Majumdar","year":"2020","unstructured":"Majumdar, A., Shrivastava, A., Lee, S., Anderson, P., Parikh, D., Batra, D.: Improving vision-and-language navigation with image-text pairs from the web. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12351, pp. 259\u2013274. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58539-6_16"},{"key":"15_CR4","unstructured":"Hao, W., Li, C., Li, X., Carin, L., et al.: Towards learning a generic agent for vision-and-language navigation via pre-trainin. In: CVPR 2022, pp. 13134\u201313143. IEEE (2022)"},{"key":"15_CR5","doi-asserted-by":"crossref","unstructured":"Guhur, P.-L., Tapaswi, M., Chen, S., et al.: Airbert: in-domain pretraining for vision-and-language navigation. In: Proceedings of ICCV, pp. 1634\u20131643. IEEE (2021)","DOI":"10.1109\/ICCV48922.2021.00166"},{"key":"15_CR6","doi-asserted-by":"crossref","unstructured":"Lin, B., Zhu, Y., Chen, Z., et al.: ADAPT: vision-language navigation with modality-aligned action prompts. In: CVPR, pp. 15375\u201315385. IEEE (2022)","DOI":"10.1109\/CVPR52688.2022.01496"},{"key":"15_CR7","unstructured":"Liu, P., Yuan, W., Fu, J., Jiang, Z., Hayashi, H., Neubig, G.: Pre-train, prompt, and predict: a systematic survey of prompting methods in natural language processing, CoRR, vol. abs\/ arXiv: 2107.13586 (2021)"},{"key":"15_CR8","doi-asserted-by":"crossref","unstructured":"Lester, B., Al-Rfou, R., Constant, N.: The power of scale for parameter-efficient prompt tuning. In: EMNLP (1), pp. 3045\u20133059. ACL (2021)","DOI":"10.18653\/v1\/2021.emnlp-main.243"},{"key":"15_CR9","unstructured":"Radford, A., Kim, J.W., et al.: Learning transferable visual models from natural language supervision. In: ICML, pp. 8748\u20138763. PMLR (2021)"},{"key":"15_CR10","unstructured":"Yao, Y., Zhang, A., Liu, Z., et al.: CPT: colorful prompt tuning for pre-trained vision-language models, CoRR, vol. abs\/ arXiv: 2109.11797 (2021)"},{"key":"15_CR11","unstructured":"Brown, T.B., Mann, B., et al.: Language models are few-shot learners. In: NeurIPS (2020)"},{"issue":"9","key":"15_CR12","doi-asserted-by":"publisher","first-page":"2337","DOI":"10.1007\/s11263-022-01653-1","volume":"130","author":"K Zhou","year":"2022","unstructured":"Zhou, K., Yang, J., Loy, C.C., Liu, Z.: Learning to prompt for vision-language models. Int. J. Comput. Vis. 130(9), 2337\u20132348 (2022)","journal-title":"Int. J. Comput. Vis."},{"key":"15_CR13","doi-asserted-by":"publisher","unstructured":"Jia, M., et al.: Visual prompt tuning. In: Avidan, S., Brostow, G.J., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) Computer Vision - ECCV 2022\u201317th European Conference, Tel Aviv, Israel, 23\u201327 October 2022, Proceedings, Part XXXIII, vol. 13693. LNCS, pp. 709\u2013727, Springer (2022). https:\/\/doi.org\/10.1007\/978-3-031-19827-4_41","DOI":"10.1007\/978-3-031-19827-4_41"},{"key":"15_CR14","doi-asserted-by":"crossref","unstructured":"Liu, T., Hu, Y., Wu, W., Wang, Y., Xu, K., Yin, Q.: Dap: domain-aware prompt learning for vision-and-language navigation (2023)","DOI":"10.1109\/ICASSP48485.2024.10446504"},{"key":"15_CR15","doi-asserted-by":"crossref","unstructured":"Hao, W., Li, C., Li, X., Carin, L., et al.: Towards learning a generic agent for vision-and-language navigation via pre-training. In: Proceedings of CVPR, pp. 13137\u201313146 (2020)","DOI":"10.1109\/CVPR42600.2020.01315"},{"key":"15_CR16","doi-asserted-by":"crossref","unstructured":"Hong, Y., Wu, Q., et al.: VLN BERT: a recurrent vision-and-language BERT for navigation. In: CVPR, pp. 1643\u20131653. IEEE (2021)","DOI":"10.1109\/CVPR46437.2021.00169"},{"key":"15_CR17","unstructured":"Devlin, J., Chang, M., et al.: BERT: pre-training of deep bidirectional transformers for language understanding. In: NAACL-HLT (1), pp. 4171\u20134186 (2019)"},{"key":"15_CR18","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"121","DOI":"10.1007\/978-3-030-58577-8_8","volume-title":"Computer Vision \u2013 ECCV 2020","author":"X Li","year":"2020","unstructured":"Li, X., et al.: Oscar: object-semantics aligned pre-training for vision-language tasks. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12375, pp. 121\u2013137. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58577-8_8"},{"key":"15_CR19","doi-asserted-by":"crossref","unstructured":"Liu, X., Huang, S., Kang, Y., Chen, H., Wang, D.: VGDiffZero: text-to-image diffusion models can be zero-shot visual grounders (2023)","DOI":"10.1109\/ICASSP48485.2024.10445945"},{"key":"15_CR20","doi-asserted-by":"crossref","unstructured":"Petroni, F., et al.: Language models as knowledge bases? EMNLP\/IJCNLP (1), pp. 2463\u20132473. ACL (2019)","DOI":"10.18653\/v1\/D19-1250"},{"key":"15_CR21","unstructured":"Liu, X.: GPT understands, too, CoRR, vol. abs\/ arXiv: 2103.10385 (2021)"},{"issue":"9","key":"15_CR22","doi-asserted-by":"publisher","first-page":"2337","DOI":"10.1007\/s11263-022-01653-1","volume":"130","author":"K Zhou","year":"2022","unstructured":"Zhou, K., Yang, J., Loy, C.C., Liu, Z.: Learning to prompt for vision-language models. Int. J. Comput. Vision 130(9), 2337\u20132348 (2022)","journal-title":"Int. J. Comput. Vision"},{"key":"15_CR23","first-page":"200","volume":"34","author":"M Tsimpoukelli","year":"2021","unstructured":"Tsimpoukelli, M., Menick, J.L., Cabi, S., Eslami, S., Vinyals, O., Hill, F.: Multimodal few-shot learning with frozen language models. Adv. Neural. Inf. Process. Syst. 34, 200\u2013212 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"15_CR24","doi-asserted-by":"crossref","unstructured":"Chang, A.X., Dai, A., Funkhouser, T.A., Halber, M., Nie\u00dfner, M., et al.: Matterport3d: learning from RGB-D data in indoor environments. In: 3DV, 667\u2013676. IEEE (2017)","DOI":"10.1109\/3DV.2017.00081"},{"key":"15_CR25","unstructured":"Anderson, P., Chang, A.X., Chaplot, D.S., et al.: On evaluation of embodied navigation agents, CoRR, vol. abs\/ arXiv: 1807.06757 (2018)"},{"key":"15_CR26","doi-asserted-by":"crossref","unstructured":"Li, M., et al.: Bridge-prompt: Towards ordinal action understanding in instructional videos. In: Proceedings of CVPR, pp. 19880\u201319889 (2022)","DOI":"10.1109\/CVPR52688.2022.01926"},{"key":"15_CR27","doi-asserted-by":"crossref","unstructured":"Hong, Y., Opazo, C.R., Wu, Q., Gould, S.: Sub-instruction aware vision-and-language navigation. In: EMNLP (1), pp. 3360\u20133376. Association for Computational Linguistics (2020)","DOI":"10.18653\/v1\/2020.emnlp-main.271"},{"key":"15_CR28","doi-asserted-by":"crossref","unstructured":"Li, X., Li, C., Xia, Q., Bisk, Y., Celikyilmaz, A., et al.: Robust navigation with language pretraining and stochastic sampling. In: EMNLP\/IJCNLP (1), pp. 1494\u20131499. ACL (2019)","DOI":"10.18653\/v1\/D19-1159"},{"key":"15_CR29","doi-asserted-by":"crossref","unstructured":"Tan, H., Yu, L., et al.: Learning to navigate unseen environments: back translation with environmental dropout. In: NAACL, pp. 2610\u20132621. ACL (2019)","DOI":"10.18653\/v1\/N19-1268"},{"key":"15_CR30","doi-asserted-by":"crossref","unstructured":"Liu, C., Zhu, F., Chang, X., et al.: Vision-language navigation with random environmental mixup. In: ICCV, pp. 1624\u20131634. IEEE (2021)","DOI":"10.1109\/ICCV48922.2021.00167"},{"key":"15_CR31","doi-asserted-by":"crossref","unstructured":"Zhu, F., Zhu, Y., Chang, X., et al.: Vision-language navigation with self-supervised auxiliary reasoning tasks. In: Proceedings of CVPR, pp. 10012\u201310022. IEEE (2020)","DOI":"10.1109\/CVPR42600.2020.01003"},{"key":"15_CR32","doi-asserted-by":"crossref","unstructured":"Qi, Y., Pan, Z., Hong, Y., Wu, Q., et al.: The road to know-where: an object-and-room informed sequential BERT for indoor vision-language navigation. In: ICCV, pp. 1635\u20131644. IEEE (2021)","DOI":"10.1109\/ICCV48922.2021.00168"},{"key":"15_CR33","doi-asserted-by":"crossref","unstructured":"An, D., Qi, Y., Wu, Q., et al.: Neighbor-view enhanced model for vision and language navigation. In: ACM MM, pp. 5101\u20135109. ACM (2021)","DOI":"10.1145\/3474085.3475282"},{"key":"15_CR34","doi-asserted-by":"crossref","unstructured":"Chen, J., Gao, C., Meng, E., et al.: Reinforced structured state-evolution for vision-language navigation. In: Proceedings of CVPR, pp. 15429\u201315438. IEEE (2022)","DOI":"10.1109\/CVPR52688.2022.01501"},{"key":"15_CR35","doi-asserted-by":"crossref","unstructured":"Liang, X., Zhu, F., Li, L., Xu, H., Liang, X.: Visual-language navigation pretraining via prompt-based environmental self-exploration. In: ACL (1), pp. 4837\u20134851. ACL (2022)","DOI":"10.18653\/v1\/2022.acl-long.332"},{"key":"15_CR36","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Qi, S., Zhou, Z., et al.: Reinforced vision-and-language navigation based on historical BERT. In: ICSI, pp. 427\u2013438 (2023)","DOI":"10.1007\/978-3-031-36625-3_34"},{"key":"15_CR37","doi-asserted-by":"crossref","unstructured":"Anderson, P., Wu, Q., Teney, D., Bruce, J., Johnson, M., et al.: \"Vision-and-language navigation: interpreting visually-grounded navigation instructions in real environments. Proc. CVPR 22, 3674\u20133683 (2018)","DOI":"10.1109\/CVPR.2018.00387"},{"key":"15_CR38","doi-asserted-by":"crossref","unstructured":"Wang, X., Huang, Q., Celikyilmaz, A., Gao, J., et al.: Reinforced cross-modal matching and self-supervised imitation learning for vision-language navigation. In: Proceedings of CVPR, pp. 6629\u20136638. IEEE (2019)","DOI":"10.1109\/CVPR.2019.00679"},{"key":"15_CR39","unstructured":"Ma, C., Lu, J., Wu, Z., AlRegib, G., Kira, Z., et al.: Self-monitoring navigation agent via auxiliary progress estimation. In: ICLR (2019)"},{"key":"15_CR40","doi-asserted-by":"crossref","unstructured":"Ke, L., Li, X., et al.: Tactical rewind: self-correction via backtracking in vision-and-language navigation. In: Proceedings of IEEE CVPR, pp. 6741\u20136749 (2019)","DOI":"10.1109\/CVPR.2019.00690"}],"container-title":["Lecture Notes in Computer Science","MultiMedia Modeling"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-53305-1_15","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,9]],"date-time":"2024-11-09T09:56:46Z","timestamp":1731146206000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-53305-1_15"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"ISBN":["9783031533044","9783031533051"],"references-count":40,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-53305-1_15","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024]]},"assertion":[{"value":"28 January 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"MMM","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Multimedia Modeling","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Amsterdam","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"The Netherlands","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 January 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2 February 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"mmm2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"ConfTool Pro","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"297","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"112","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"38% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.2","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.2","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}