{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,27]],"date-time":"2025-03-27T09:43:29Z","timestamp":1743068609860,"version":"3.40.3"},"publisher-location":"Cham","reference-count":64,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031731945"},{"type":"electronic","value":"9783031731952"}],"license":[{"start":{"date-parts":[[2024,11,27]],"date-time":"2024-11-27T00:00:00Z","timestamp":1732665600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,27]],"date-time":"2024-11-27T00:00:00Z","timestamp":1732665600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-73195-2_9","type":"book-chapter","created":{"date-parts":[[2024,11,26]],"date-time":"2024-11-26T09:35:44Z","timestamp":1732613744000},"page":"143-160","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Compositional Substitutivity of\u00a0Visual Reasoning for\u00a0Visual Question Answering"],"prefix":"10.1007","author":[{"given":"Chuanhao","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhen","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chenchen","family":"Jing","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuwei","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mingliang","family":"Zhai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yunde","family":"Jia","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,11,27]]},"reference":[{"key":"9_CR1","unstructured":"Hudson, D.A., Manning, C.D.: Compositional attention networks for machine reasoning. In: International Conference on Learning Representations (2018)"},{"key":"9_CR2","doi-asserted-by":"crossref","unstructured":"Hudson, A., Manning, C.D.: GQA: a new dataset for real-world visual reasoning and compositional question answering. In: IEEE Conference on Computer Vision and Pattern Recognition, pp. 6700\u20136709 (2019)","DOI":"10.1109\/CVPR.2019.00686"},{"key":"9_CR3","doi-asserted-by":"crossref","unstructured":"Tan, H., Bansal, M.: Lxmert: learning cross-modality encoder representations from transformers. In: Proceedings of the Conference on Empirical Methods in Natural Language Processing and the International Joint Conference on Natural Language Processing, pp. 5100\u20135111 (2019)","DOI":"10.18653\/v1\/D19-1514"},{"key":"9_CR4","doi-asserted-by":"publisher","first-page":"726","DOI":"10.1162\/tacl_a_00343","volume":"8","author":"Y Liu","year":"2020","unstructured":"Liu, Y., et al.: Multilingual denoising pre-training for neural machine translation. Trans. Assoc. Comput. Linguist. 8, 726\u2013742 (2020)","journal-title":"Trans. Assoc. Comput. Linguist."},{"key":"9_CR5","doi-asserted-by":"crossref","unstructured":"Sennrich, R., Haddow, B., Birch, A.: Improving neural machine translation models with monolingual data. In: Proceedings of the Annual Meeting of the Association for Computational Linguistics, pp. 86\u201396 (2016)","DOI":"10.18653\/v1\/P16-1009"},{"key":"9_CR6","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763 (2021)"},{"key":"9_CR7","unstructured":"Patrick, M., et al.: Support-set bottlenecks for video-text representation learning. arXiv preprint arXiv:2010.02824 (2020)"},{"key":"9_CR8","doi-asserted-by":"publisher","first-page":"757","DOI":"10.1613\/jair.1.11674","volume":"67","author":"D Hupkes","year":"2020","unstructured":"Hupkes, D., Dankers, V., Mul, M., Bruni, E.: Compositionality decomposed: how do neural networks generalise? J. Artif. Intell. Res. 67, 757\u2013795 (2020)","journal-title":"J. Artif. Intell. Res."},{"key":"9_CR9","doi-asserted-by":"crossref","unstructured":"Li, Y., Zhao, L., Wang, J., Hestness, J.: Compositional generalization for primitive substitutions. In: Proceedings of the Conference on Empirical Methods in Natural Language Processing and the International Joint Conference on Natural Language Processing, pp. 4293\u20134302 (2019)","DOI":"10.18653\/v1\/D19-1438"},{"key":"9_CR10","doi-asserted-by":"crossref","unstructured":"Ren, S., Deng, Y., He, K., Che, W.: Generating natural language adversarial examples through probability weighted word saliency. In: Proceedings of the Annual Meeting of the Association for Computational Linguistics, pp. 1085\u20131097 (2019)","DOI":"10.18653\/v1\/P19-1103"},{"key":"9_CR11","doi-asserted-by":"crossref","unstructured":"Zhou, Y., Zheng, X., Hsieh, C.-J., Chang, K.-W., Huang, X.-J.: Defense against synonym substitution-based adversarial attacks via dirichlet neighborhood ensemble. In: Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers), pp. 5482\u20135492 (2021)","DOI":"10.18653\/v1\/2021.acl-long.426"},{"key":"9_CR12","unstructured":"Yang, Y., Wang, X., He, K.: Robust textual embedding against word-level adversarial attacks. In: Uncertainty in Artificial Intelligence, pp. 2214\u20132224 (2022)"},{"key":"9_CR13","doi-asserted-by":"crossref","unstructured":"Dankers, V., Bruni, E., Hupkes, D.: The paradox of the compositionality of natural language: a neural machine translation case study. In: Proceedings of the Annual Meeting of the Association for Computational Linguistics, pp. 4154\u20134175 (2022)","DOI":"10.18653\/v1\/2022.acl-long.286"},{"key":"9_CR14","doi-asserted-by":"crossref","unstructured":"Kudo, K., et al.: Do deep neural networks capture compositionality in arithmetic reasoning? In: Proceedings of the Conference of the European Chapter of the Association for Computational Linguistics, pp. 1343\u20131354 (2023)","DOI":"10.18653\/v1\/2023.eacl-main.98"},{"key":"9_CR15","unstructured":"Whitehead, S., Wu, H., Fung, Y.R., Ji, H., Feris, R., Saenko, K.: Learning from lexical perturbations for consistent visual question answering. arXiv preprint arXiv:2011.13406 (2020)"},{"key":"9_CR16","unstructured":"Gou, W., Shi, W., Lou, J., Huang, L., Zhou, P., Li, R.: Sneak: synonymous sentences-aware adversarial attack on natural language video localization. arXiv preprint arXiv:2112.04154 (2021)"},{"key":"9_CR17","unstructured":"Klinger, T., et al.: A study of compositional generalization in neural models. arXiv preprint arXiv:2006.09437 (2020)"},{"key":"9_CR18","doi-asserted-by":"crossref","unstructured":"Pantazopoulos, G., Suglia, A., Eshghi, A.: Combine to describe: evaluating compositional generalization in image captioning. In: Proceedings of the Association for Computational Linguistics: Student Research Workshop, pp. 115\u2013131 (2022)","DOI":"10.18653\/v1\/2022.acl-srw.11"},{"key":"9_CR19","unstructured":"Stoikou, T., Lymperaiou, M., Stamou, G.: Knowledge-based counterfactual queries for visual question answering. arXiv preprint arXiv:2303.02601 (2023)"},{"issue":"1\u20132","key":"9_CR20","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1016\/0010-0277(88)90031-5","volume":"28","author":"JA Fodor","year":"1988","unstructured":"Fodor, J.A., Pylyshyn, Z.W.: Connectionism and cognitive architecture: a critical analysis. Cognition 28(1\u20132), 3\u201371 (1988)","journal-title":"Cognition"},{"key":"9_CR21","doi-asserted-by":"publisher","first-page":"195","DOI":"10.1162\/tacl_a_00361","volume":"9","author":"B Bogin","year":"2021","unstructured":"Bogin, B., Subramanian, S., Gardner, M., Berant, J.: Latent compositional representations improve systematic generalization in grounded question answering. Trans. Assoc. Comput. Linguist. 9, 195\u2013210 (2021)","journal-title":"Trans. Assoc. Comput. Linguist."},{"key":"9_CR22","doi-asserted-by":"crossref","unstructured":"Shi, J., Zhang, H., Li, J.: Explainable and explicit visual reasoning over scene graphs. In: IEEE Conference on Computer Vision and Pattern Recognition, pp. 8376\u20138384 (2019)","DOI":"10.1109\/CVPR.2019.00857"},{"key":"9_CR23","unstructured":"Akula, A., Jampani, V., Changpinyo, S., Zhu, S.-C.: Robust visual reasoning via language guided neural module networks. In: Advances in Neural Information Processing Systems, pp. 11041\u201311053 (2021)"},{"key":"9_CR24","doi-asserted-by":"crossref","unstructured":"Jiang, J., Liu, Z., Liu, Y., Nan, Z., Zheng, N.: X-GGM: graph generative modeling for out-of-distribution generalization in visual question answering. In: ACM International Conference on Multimedia, pp. 199\u2013208 (2021)","DOI":"10.1145\/3474085.3475350"},{"key":"9_CR25","doi-asserted-by":"crossref","unstructured":"Anderson, P., et al.: Bottom-up and top-down attention for image captioning and visual question answering. In: IEEE Conference on Computer Vision and Pattern Recognition, pp. 6077\u20136086 (2018)","DOI":"10.1109\/CVPR.2018.00636"},{"key":"9_CR26","doi-asserted-by":"crossref","unstructured":"Chen, W., Gan, Z., Li, L., Cheng, Y., Wang, W., Liu, J.: Meta module network for compositional visual reasoning. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 655\u2013664 (2021)","DOI":"10.1109\/WACV48630.2021.00070"},{"key":"9_CR27","doi-asserted-by":"crossref","unstructured":"Kamath, A., Singh, M., LeCun, Y., Synnaeve, G., Misra, I., Carion, N.: MDETR-modulated detection for end-to-end multi-modal understanding. In: International Conference on Computer Vision, pp. 1780\u20131790 (2021)","DOI":"10.1109\/ICCV48922.2021.00180"},{"key":"9_CR28","doi-asserted-by":"crossref","unstructured":"Jing, C., Jia, Y., Wu, Y., Liu, X., Wu, Q.: Maintaining reasoning consistency in compositional visual question answering. In: IEEE Conference on Computer Vision and Pattern Recognition, pp. 5099\u20135108 (2022)","DOI":"10.1109\/CVPR52688.2022.00504"},{"key":"9_CR29","doi-asserted-by":"crossref","unstructured":"Hu, R., Rohrbach, A., Darrell, T., Saenko, K.: Language-conditioned graph networks for relational reasoning. In: International Conference on Computer Vision, pp. 10294\u201310303 (2019)","DOI":"10.1109\/ICCV.2019.01039"},{"key":"9_CR30","volume-title":"WordNet: An Electronic Lexical Database","author":"GA Miller","year":"1998","unstructured":"Miller, G.A.: WordNet: An Electronic Lexical Database. MIT Press, Cambridge (1998)"},{"key":"9_CR31","doi-asserted-by":"crossref","unstructured":"Ontanon, S., Ainslie, J., Fisher, Z., Cvicek, V.: Making transformers solve compositional tasks. In: Proceedings of the Annual Meeting of the Association for Computational Linguistics, pp. 3591\u20133607 (2022)","DOI":"10.18653\/v1\/2022.acl-long.251"},{"key":"9_CR32","doi-asserted-by":"crossref","unstructured":"Zheng, H., Lapata, M.: Disentangled sequence to sequence learning for compositional generalization. In: Proceedings of the Annual Meeting of the Association for Computational Linguistics, pp. 4256\u20134268 (2022)","DOI":"10.18653\/v1\/2022.acl-long.293"},{"key":"9_CR33","doi-asserted-by":"crossref","unstructured":"Saphra, N., Lopez, A.: LSTMs compose\u2014and learn\u2014bottom-up. In: Proceedings of the Conference on Empirical Methods in Natural Language Processing, pp. 2797\u20132809 (2020)","DOI":"10.18653\/v1\/2020.findings-emnlp.252"},{"key":"9_CR34","doi-asserted-by":"crossref","unstructured":"Dankers, V., Langedijk, A., McCurdy, K., Williams, A., Hupkes, D.: Generalising to German plural noun classes, from the perspective of a recurrent neural network. In: CoNLL, pp. 94\u2013108 (2021)","DOI":"10.18653\/v1\/2021.conll-1.8"},{"key":"9_CR35","doi-asserted-by":"crossref","unstructured":"Li, Y., et al.: Gligen: open-set grounded text-to-image generation. In: IEEE Conference on Computer Vision and Pattern Recognition, pp. 22511\u201322521 (2023)","DOI":"10.1109\/CVPR52729.2023.02156"},{"key":"9_CR36","unstructured":"Awadalla, A., et al.: Openflamingo: an open-source framework for training large autoregressive vision-language models. arXiv preprint arXiv:2308.01390 (2023)"},{"key":"9_CR37","unstructured":"Li, J., Li, D., Savarese, S., Hoi, S.: Blip-2: bootstrapping language-image pre-training with frozen image encoders and large language models. arXiv preprint arXiv:2301.12597 (2023)"},{"key":"9_CR38","unstructured":"Cho, J., Lei, J., Tan, H., Bansal, M.: Unifying vision-and-language tasks via text generation. In: International Conference on Machine Learning, pp. 1931\u20131942 (2021)"},{"key":"9_CR39","doi-asserted-by":"crossref","unstructured":"Shah, M., Chen, X., Rohrbach, M., Parikh, D.: Cycle-consistency for robust visual question answering. In: IEEE Conference on Computer Vision and Pattern Recognition, pp. 6642\u20136651 (2019)","DOI":"10.1109\/CVPR.2019.00681"},{"key":"9_CR40","doi-asserted-by":"crossref","unstructured":"Ray, A., Sikka, K., Divakaran, A., Lee, S., Burachas, G.: Sunny and dark outside?! improving answer consistency in VQA through entailed question generation. In: Proceedings of the Conference on Empirical Methods in Natural Language Processing and the International Joint Conference on Natural Language Processing, pp. 5860\u20135865 (2019)","DOI":"10.18653\/v1\/D19-1596"},{"key":"9_CR41","doi-asserted-by":"crossref","unstructured":"Ribeiro, M.T., Guestrin, C., Singh, S.: Are red roses red? Evaluating consistency of question-answering models. In: Proceedings of the Annual Meeting of the Association for Computational Linguistics, pp. 6174\u20136184 (2019)","DOI":"10.18653\/v1\/P19-1621"},{"key":"9_CR42","doi-asserted-by":"crossref","unstructured":"Goyal, Y., Khot, T., Summers-Stay, D., Batra, D., Parikh, D.: Making the V in VQA matter: elevating the role of image understanding in visual question answering. In: IEEE Conference on Computer Vision and Pattern Recognition, pp. 6904\u20136913 (2017)","DOI":"10.1109\/CVPR.2017.670"},{"key":"9_CR43","doi-asserted-by":"crossref","unstructured":"Tascon-Morales, S., M\u00e1rquez-Neila, P., Sznitman, R.: Logical implications for visual question answering consistency. In: IEEE Conference on Computer Vision and Pattern Recognition, pp. 6725\u20136735 (2023)","DOI":"10.1109\/CVPR52729.2023.00650"},{"key":"9_CR44","doi-asserted-by":"crossref","unstructured":"Selvaraju, R.R., et al.: Squinting at VQA models: introspecting VQA models with sub-questions. In: IEEE Conference on Computer Vision and Pattern Recognition, pp. 10003\u201310011 (2020)","DOI":"10.1109\/CVPR42600.2020.01002"},{"key":"9_CR45","doi-asserted-by":"crossref","unstructured":"Yuan, Y., Wang, S., Jiang, M., Chen, T.Y.: Perception matters: detecting perception failures of VQA models using metamorphic testing. In: IEEE Conference on Computer Vision and Pattern Recognition, pp. 16908\u201316917 (2021)","DOI":"10.1109\/CVPR46437.2021.01663"},{"key":"9_CR46","unstructured":"Kim, W., Son, B., Kim, I.: VILT: vision-and-language transformer without convolution or region supervision. In: International Conference on Machine Learning, pp. 5583\u20135594 (2021)"},{"key":"9_CR47","unstructured":"Kim, J.-H., Jun, J., Zhang, B.-T.: Bilinear attention networks. Advances in Neural Information Processing Systems, vol.\u00a031 (2018)"},{"key":"9_CR48","doi-asserted-by":"crossref","unstructured":"Kant, Y., Moudgil, A., Batra, D., Parikh, D., Agrawal, H.: Contrast and classify: training robust VQA models. In: International Conference on Computer Vision, pp. 1604\u20131613 (2021)","DOI":"10.1109\/ICCV48922.2021.00163"},{"key":"9_CR49","unstructured":"Li, L., Gan, Z., Liu, J.: A closer look at the robustness of vision-and-language pre-trained models. arXiv preprint arXiv:2012.08673 (2020)"},{"key":"9_CR50","unstructured":"Zhu, D., Chen, J., Shen, X., Li, X., Elhoseiny, M.: Minigpt-4: enhancing vision-language understanding with advanced large language models. arXiv preprint arXiv:2304.10592 (2023)"},{"key":"9_CR51","doi-asserted-by":"crossref","unstructured":"Li, C., Li, Z., Jing, C., Jia, Y., Wu, Y.: Exploring the effect of primitives for compositional generalization in vision-and-language. In: IEEE Conference on Computer Vision and Pattern Recognition, pp. 19092\u201319101 (2023)","DOI":"10.1109\/CVPR52729.2023.01830"},{"key":"9_CR52","doi-asserted-by":"crossref","unstructured":"Yang, L., Kong, Q., Yang, H.-K., Kehl, W., Sato, Y., Kobori, N.: Deco: decomposition and reconstruction for compositional temporal grounding via coarse-to-fine contrastive ranking. In: IEEE Conference on Computer Vision and Pattern Recognition, pp. 23130\u201323140 (2023)","DOI":"10.1109\/CVPR52729.2023.02215"},{"key":"9_CR53","doi-asserted-by":"crossref","unstructured":"Li, J., et al.: Variational cross-graph reasoning and adaptive structured semantics learning for compositional temporal grounding. IEEE Trans. Pattern Anal. Mach. Intell. (2023)","DOI":"10.1109\/TPAMI.2023.3274139"},{"key":"9_CR54","doi-asserted-by":"crossref","unstructured":"Li, C., et al.: mPLUG: effective and efficient vision-language learning by cross-modal skip-connections. In: Proceedings of the Conference on Empirical Methods in Natural Language Processing, pp. 7241\u20137259 (2022)","DOI":"10.18653\/v1\/2022.emnlp-main.488"},{"key":"9_CR55","doi-asserted-by":"crossref","unstructured":"Wang, W., et al.: Image as a foreign language: BEiT pretraining for vision and vision-language tasks. In: IEEE Conference on Computer Vision and Pattern Recognition (2023)","DOI":"10.1109\/CVPR52729.2023.01838"},{"key":"9_CR56","unstructured":"Dong, X., et al.: Internlm-xcomposer2: mastering free-form text-image composition and comprehension in vision-language large model. arXiv preprint arXiv:2401.16420 (2024)"},{"key":"9_CR57","unstructured":"Bai, J., et al.: Qwen-VL: a frontier large vision-language model with versatile abilities. arXiv preprint arXiv:2308.12966 (2023)"},{"key":"9_CR58","unstructured":"Wang, W., et al.: Cogvlm: visual expert for pretrained language models. arXiv preprint arXiv:2311.03079 (2023)"},{"key":"9_CR59","doi-asserted-by":"crossref","unstructured":"Ye, Q., et al.: mPLUG-Owl2: revolutionizing multi-modal large language model with modality collaboration. In: IEEE Conference on Computer Vision and Pattern Recognition, pp. 13040\u201313051 (2024)","DOI":"10.1109\/CVPR52733.2024.01239"},{"key":"9_CR60","unstructured":"XTuner Contributors: Xtuner: a toolkit for efficiently fine-tuning LLM (2023). https:\/\/github.com\/InternLM\/xtuner"},{"key":"9_CR61","unstructured":"DataCanvas Ltd.: mmalaya (2024). https:\/\/github.com\/DataCanvasIO\/MMAlaya"},{"key":"9_CR62","unstructured":"OpenCompass Contributors: Opencompass: a universal evaluation platform for foundation models (2023). https:\/\/github.com\/open-compass\/opencompass"},{"key":"9_CR63","doi-asserted-by":"crossref","unstructured":"Agrawal, A., Batra, D., Parikh, D., Kembhavi, A.: Don\u2019t just assume; look and answer: overcoming priors for visual question answering. In: IEEE Conference on Computer Vision and Pattern Recognition, pp. 4971\u20134980 (2018)","DOI":"10.1109\/CVPR.2018.00522"},{"key":"9_CR64","unstructured":"Team, G., et al.: Gemini: a family of highly capable multimodal models. arXiv preprint arXiv:2312.11805 (2023)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-73195-2_9","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,26]],"date-time":"2024-11-26T10:06:40Z","timestamp":1732615600000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-73195-2_9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,27]]},"ISBN":["9783031731945","9783031731952"],"references-count":64,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-73195-2_9","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2024,11,27]]},"assertion":[{"value":"27 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}