{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,29]],"date-time":"2026-05-29T11:21:09Z","timestamp":1780053669077,"version":"3.54.0"},"publisher-location":"Cham","reference-count":81,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031733369","type":"print"},{"value":"9783031733376","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T00:00:00Z","timestamp":1730332800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T00:00:00Z","timestamp":1730332800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-73337-6_8","type":"book-chapter","created":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T23:02:27Z","timestamp":1730329347000},"page":"129-147","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["GENIXER: Empowering Multimodal Large Language Model as\u00a0a\u00a0Powerful Data Generator"],"prefix":"10.1007","author":[{"given":"Henry Hengyuan","family":"Zhao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Pan","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mike Zheng","family":"Shou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,10,31]]},"reference":[{"key":"8_CR1","doi-asserted-by":"crossref","unstructured":"Agrawal, H., et al: nocaps: novel object captioning at scale. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00904"},{"key":"8_CR2","unstructured":"Anil, R., et\u00a0al.: Palm 2 technical report. arXiv:2305.10403 (2023)"},{"key":"8_CR3","unstructured":"Bai, J., et al.: Qwen-vl: a versatile vision-language model for understanding, localization, text reading, and beyond. arXiv preprint arXiv:2308.12966 (2023)"},{"key":"8_CR4","unstructured":"Bai, J., et\u00a0al.: Ofasys: a multi-modal multi-task learning system for building generalist models. arXiv:2212.04408 (2022)"},{"key":"8_CR5","unstructured":"Bavishi, R., et al.: Introducing our multimodal models (2023). https:\/\/www.adept.ai\/blog\/fuyu-8b"},{"key":"8_CR6","unstructured":"Brown, T., et\u00a0al.: Language models are few-shot learners. In: NeurIPS (2020)"},{"key":"8_CR7","doi-asserted-by":"crossref","unstructured":"Cha, J., Kang, W., Mun, J., Roh, B.: Honeybee: locality-enhanced projector for multimodal llm. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2024)","DOI":"10.1109\/CVPR52733.2024.01311"},{"key":"8_CR8","unstructured":"Chen, C., et al.: Position-enhanced visual instruction tuning for multimodal large language models. arXiv preprint arXiv:2308.13437 (2023)"},{"key":"8_CR9","unstructured":"Chen, J., et al.: Minigpt-v2: large language model as a unified interface for vision-language multi-task learning. arXiv preprint arXiv:2310.09478 (2023)"},{"key":"8_CR10","unstructured":"Chen, K., Zhang, Z., Zeng, W., Zhang, R., Zhu, F., Zhao, R.: Shikra: unleashing multimodal llm\u2019s referential dialogue magic. arXiv:2306.15195 (2023)"},{"key":"8_CR11","doi-asserted-by":"crossref","unstructured":"Chen, L., et al.: Sharegpt4v: improving large multi-modal models with better captions. arXiv preprint arXiv:2311.12793 (2023)","DOI":"10.1007\/978-3-031-72643-9_22"},{"key":"8_CR12","unstructured":"Chen, X., et al.: Microsoft coco captions: Data collection and evaluation server. arXiv:1504.00325 (2015)"},{"key":"8_CR13","doi-asserted-by":"publisher","first-page":"104","DOI":"10.1007\/978-3-030-58577-8_7","volume-title":"Computer Vision \u2013 ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XXX","author":"Y-C Chen","year":"2020","unstructured":"Chen, Y.-C., et al.: UNITER: UNiversal image-TExt representation learning. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) Computer Vision \u2013 ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XXX, pp. 104\u2013120. Springer International Publishing, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58577-8_7"},{"key":"8_CR14","doi-asserted-by":"crossref","unstructured":"Chen, Z., et\u00a0al.: Internvl: scaling up vision foundation models and aligning for generic visual-linguistic tasks. arXiv preprint arXiv:2312.14238 (2023)","DOI":"10.1109\/CVPR52733.2024.02283"},{"key":"8_CR15","unstructured":"Chiang, W.L., et al.: Vicuna: an open-source chatbot impressing gpt-4 with 90%* chatgpt quality (March 2023). https:\/\/lmsys.org\/blog\/2023-03-30-vicuna\/"},{"key":"8_CR16","unstructured":"Chung, H.W., et\u00a0al.: Scaling instruction-finetuned language models. arXiv preprint arXiv:2210.11416 (2022)"},{"key":"8_CR17","unstructured":"Dai, W., et al.: Instructblip: towards general-purpose vision-language models with instruction tuning. In: NeurIPS (2023)"},{"key":"8_CR18","unstructured":"Fu, C., et\u00a0al.: Mme: a comprehensive evaluation benchmark for multimodal large language models. arXiv:2306.13394 (2023)"},{"key":"8_CR19","first-page":"6616","volume":"33","author":"Z Gan","year":"2020","unstructured":"Gan, Z., Chen, Y.C., Li, L., Zhu, C., Cheng, Y., Liu, J.: Large-scale adversarial training for vision-and-language representation learning. Adv. Neural. Inf. Process. Syst. 33, 6616\u20136628 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"8_CR20","unstructured":"Gao, P., et\u00a0al.: Llama-adapter v2: parameter-efficient visual instruction model. arXiv:2304.15010 (2023)"},{"key":"8_CR21","unstructured":"Gao, P., et\u00a0al.: Sphinx-x: scaling data and parameters for a family of multi-modal large language models. arXiv preprint arXiv:2402.05935 (2024)"},{"key":"8_CR22","doi-asserted-by":"crossref","unstructured":"Goyal, Y., Khot, T., Summers-Stay, D., Batra, D., Parikh, D.: Making the v in vqa matter: elevating the role of image understanding in visual question answering. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.670"},{"key":"8_CR23","doi-asserted-by":"crossref","unstructured":"Gurari, D., et al.: Vizwiz grand challenge: answering visual questions from blind people. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00380"},{"key":"8_CR24","unstructured":"He, M., et al.: Efficient multimodal learning from data-centric perspective. arXiv preprint arXiv:2402.11530 (2024)"},{"key":"8_CR25","unstructured":"Huang, S., et\u00a0al.: Language is not all you need: Aligning perception with language models. arXiv:2302.14045 (2023)"},{"key":"8_CR26","doi-asserted-by":"crossref","unstructured":"Hudson, D.A., Manning, C.D.: Gqa: a new dataset for real-world visual reasoning and compositional question answering. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00686"},{"key":"8_CR27","unstructured":"Ilharco, G., et al.: Openclip (2021). https:\/\/doi.org\/10.5281\/zenodo.5143773"},{"key":"8_CR28","doi-asserted-by":"crossref","unstructured":"Kamath, A., Singh, M., LeCun, Y., Synnaeve, G., Misra, I., Carion, N.: Mdetr-modulated detection for end-to-end multi-modal understanding. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1780\u20131790 (2021)","DOI":"10.1109\/ICCV48922.2021.00180"},{"key":"8_CR29","doi-asserted-by":"publisher","first-page":"662","DOI":"10.1007\/978-3-031-20059-5_38","volume-title":"Computer Vision \u2013 ECCV 2022: 17th European Conference, Tel Aviv, Israel, October 23\u201327, 2022, Proceedings, Part XXXVI","author":"A Kamath","year":"2022","unstructured":"Kamath, A., Clark, C., Gupta, T., Kolve, E., Hoiem, D., Kembhavi, A.: Webly supervised concept expansion for\u00a0general purpose vision models. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) Computer Vision \u2013 ECCV 2022: 17th European Conference, Tel Aviv, Israel, October 23\u201327, 2022, Proceedings, Part XXXVI, pp. 662\u2013681. Springer Nature Switzerland, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-20059-5_38"},{"key":"8_CR30","doi-asserted-by":"crossref","unstructured":"Kazemzadeh, S., Ordonez, V., Matten, M., Berg, T.: Referitgame: Referring to objects in photographs of natural scenes. In: EMNLP (2014)","DOI":"10.3115\/v1\/D14-1086"},{"key":"8_CR31","doi-asserted-by":"crossref","unstructured":"Krishna, R., et\u00a0al.: Visual genome: connecting language and vision using crowdsourced dense image annotations. In: IJCV (2017)","DOI":"10.1007\/s11263-016-0981-7"},{"key":"8_CR32","doi-asserted-by":"crossref","unstructured":"Lai, X., et al.: Lisa: reasoning segmentation via large language model. arXiv preprint arXiv:2308.00692 (2023)","DOI":"10.1109\/CVPR52733.2024.00915"},{"key":"8_CR33","unstructured":"Lauren\u00e7on, H., et\u00a0al.: Obelics: an open web-scale filtered dataset of interleaved image-text documents. In: Thirty-seventh Conference on Neural Information Processing Systems Datasets and Benchmarks Track (2023)"},{"key":"8_CR34","unstructured":"Li, B., Zhang, Y., Chen, L., Wang, J., Yang, J., Liu, Z.: Otter: a multi-modal model with in-context instruction tuning. arXiv:2305.03726 (2023)"},{"key":"8_CR35","doi-asserted-by":"crossref","unstructured":"Li, B., Wang, R., Wang, G., Ge, Y., Ge, Y., Shan, Y.: Seed-bench: benchmarking multimodal llms with generative comprehension (2023)","DOI":"10.1109\/CVPR52733.2024.01263"},{"key":"8_CR36","unstructured":"Li, J., Li, D., Savarese, S., Hoi, S.: Blip-2: bootstrapping language-image pre-training with frozen image encoders and large language models. arXiv:2301.12597 (2023)"},{"key":"8_CR37","doi-asserted-by":"crossref","unstructured":"Li, Y., Du, Y., Zhou, K., Wang, J., Zhao, W.X., Wen, J.R.: Evaluating object hallucination in large vision-language models. arXiv:2305.10355 (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.20"},{"key":"8_CR38","doi-asserted-by":"crossref","unstructured":"Lin, T.Y., et al.: Microsoft coco: common objects in context. In: ECCV (2014)","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"8_CR39","doi-asserted-by":"crossref","unstructured":"Liu, H., Li, C., Li, Y., Lee, Y.J.: Improved baselines with visual instruction tuning (2023)","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"8_CR40","unstructured":"Liu, H., Li, C., Wu, Q., Lee, Y.J.: Visual instruction tuning. In: NeurIPS (2023)"},{"key":"8_CR41","doi-asserted-by":"crossref","unstructured":"Liu, Y., et\u00a0al.: Mmbench: is your multi-modal model an all-around player? arXiv preprint arXiv:2307.06281 (2023)","DOI":"10.1007\/978-3-031-72658-3_13"},{"key":"8_CR42","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. In: International Conference on Learning Representations (2019). https:\/\/openreview.net\/forum?id=Bkg6RiCqY7"},{"key":"8_CR43","unstructured":"Lu, H., et\u00a0al.: Deepseek-vl: towards real-world vision-language understanding. arXiv preprint arXiv:2403.05525 (2024)"},{"key":"8_CR44","doi-asserted-by":"crossref","unstructured":"Lu, J., et al.: Unified-io 2: scaling autoregressive multimodal models with vision, language, audio, and action. arXiv preprint arXiv:2312.17172 (2023)","DOI":"10.1109\/CVPR52733.2024.02497"},{"key":"8_CR45","unstructured":"Lu, J., Clark, C., Zellers, R., Mottaghi, R., Kembhavi, A.: Unified-io: a unified model for vision, language, and multi-modal tasks. arXiv:2206.08916 (2022)"},{"key":"8_CR46","unstructured":"Lu, P., et al.: Learn to explain: multimodal reasoning via thought chains for science question answering. In: NeurIPS (2022)"},{"key":"8_CR47","unstructured":"Luo, G., Zhou, Y., Ren, T., Chen, S., Sun, X., Ji, R.: Cheap and quick: Efficient vision-language instruction tuning for large language models. NeurIPS (2023)"},{"key":"8_CR48","unstructured":"Mani, A., Yoo, N., Hinthorn, W., Russakovsky, O.: Point and ask: incorporating pointing into visual question answering. arXiv preprint arXiv:2011.13681 (2020)"},{"key":"8_CR49","doi-asserted-by":"crossref","unstructured":"Mao, J., Huang, J., Toshev, A., Camburu, O., Yuille, A.L., Murphy, K.: Generation and comprehension of unambiguous object descriptions. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.9"},{"key":"8_CR50","doi-asserted-by":"crossref","unstructured":"Marino, K., Rastegari, M., Farhadi, A., Mottaghi, R.: Ok-vqa: a visual question answering benchmark requiring external knowledge. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00331"},{"key":"8_CR51","doi-asserted-by":"crossref","unstructured":"Moon, S., et\u00a0al.: Anymal: an efficient and scalable any-modality augmented language model. arXiv preprint arXiv:2309.16058 (2023)","DOI":"10.18653\/v1\/2024.emnlp-industry.98"},{"key":"8_CR52","unstructured":"OpenAI: Gpt-4 technical report (2023)"},{"key":"8_CR53","unstructured":"OpenAI: Gpt-4v(ision) system card (2023). https:\/\/cdn.openai.com\/papers\/GPTV_System_Card.pdf"},{"key":"8_CR54","unstructured":"Ordonez, V., Kulkarni, G., Berg, T.: Im2text: describing images using 1 million captioned photographs. In: NeurIPS (2011)"},{"key":"8_CR55","unstructured":"Peng, Z., et al.: Kosmos-2: grounding multimodal large language models to the world. arXiv:2306.14824 (2023)"},{"key":"8_CR56","doi-asserted-by":"publisher","first-page":"74","DOI":"10.1007\/s11263-016-0965-7","volume":"123","author":"BA Plummer","year":"2015","unstructured":"Plummer, B.A., Wang, L., Cervantes, C.M., Caicedo, J.C., Hockenmaier, J., Lazebnik, S.: Flickr30k entities: Collecting region-to-phrase correspondences for richer image-to-sentence models. Int. J. Comput. Vision 123, 74\u201393 (2015)","journal-title":"Int. J. Comput. Vision"},{"key":"8_CR57","unstructured":"Radford, A., et\u00a0al.: Learning transferable visual models from natural language supervision. In: ICML (2021)"},{"key":"8_CR58","doi-asserted-by":"crossref","unstructured":"Rasheed, H., et al.: Glamm: pixel grounding large multimodal model. The IEEE\/CVF Conference on Computer Vision and Pattern Recognition (2024)","DOI":"10.1109\/CVPR52733.2024.01236"},{"key":"8_CR59","doi-asserted-by":"crossref","unstructured":"Ren, Z., et al.: Pixellm: pixel reasoning with large multimodal model (2023)","DOI":"10.1109\/CVPR52733.2024.02491"},{"key":"8_CR60","unstructured":"Schuhmann, C., et\u00a0al.: Laion-5b: an open large-scale dataset for training next generation image-text models. arXiv:2210.08402 (2022)"},{"key":"8_CR61","doi-asserted-by":"crossref","unstructured":"Sharma, P., Ding, N., Goodman, S., Soricut, R.: Conceptual captions: A cleaned, hypernymed, image alt-text dataset for automatic image captioning. In: ACL (2018)","DOI":"10.18653\/v1\/P18-1238"},{"key":"8_CR62","doi-asserted-by":"crossref","unstructured":"Singh, A., et al.: Towards vqa models that can read. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (June 2019)","DOI":"10.1109\/CVPR.2019.00851"},{"key":"8_CR63","unstructured":"Su, Y., Lan, T., Li, H., Xu, J., Wang, Y., Cai, D.: Pandagpt: one model to instruction-follow them all. arXiv:2305.16355 (2023)"},{"key":"8_CR64","unstructured":"Tang, J., et\u00a0al.: Textsquare: scaling up text-centric visual instruction tuning. arXiv preprint arXiv:2404.12803 (2024)"},{"key":"8_CR65","unstructured":"Touvron, H., et\u00a0al.: Llama: open and efficient foundation language models. arXiv:2302.13971 (2023)"},{"key":"8_CR66","unstructured":"Touvron, H., et\u00a0al.: Llama 2: open foundation and fine-tuned chat models. arXiv preprint arXiv:2307.09288 (2023)"},{"key":"8_CR67","unstructured":"Wang, P., et al.: Ofa: Unifying architectures, tasks, and modalities through a simple sequence-to-sequence learning framework. In: ICML (2022)"},{"key":"8_CR68","unstructured":"Wang, W., et al.: Cogvlm: Visual expert for pretrained language models. arXiv (2023)"},{"key":"8_CR69","unstructured":"Wang, W., et\u00a0al.: Visionllm: large language model is also an open-ended decoder for vision-centric tasks. arXiv preprint arXiv:2305.11175 (2023)"},{"key":"8_CR70","doi-asserted-by":"publisher","unstructured":"Wang, Y., et al.: Self-instruct: aligning language models with self-generated instructions. In: Rogers, A., Boyd-Graber, J., Okazaki, N. (eds.) Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers). pp. 13484\u201313508. Association for Computational Linguistics, Toronto, Canada (Jul 2023). https:\/\/doi.org\/10.18653\/v1\/2023.acl-long.754, https:\/\/aclanthology.org\/2023.acl-long.754","DOI":"10.18653\/v1\/2023.acl-long.754"},{"key":"8_CR71","unstructured":"Xu, C., et al.: WizardLM: empowering large pre-trained language models to follow complex instructions. In: The Twelfth International Conference on Learning Representations (2024). https:\/\/openreview.net\/forum?id=CfXh93NDgH"},{"key":"8_CR72","unstructured":"Xu, J., et al.: Pixel Aligned Language Models. arXiv preprint arXiv: 2312.09237 (2023)"},{"key":"8_CR73","doi-asserted-by":"publisher","unstructured":"Yang, Z., et al.: Unitab: unifying text and box outputs for grounded vision-language modeling. In: European Conference on Computer Vision. pp. 521\u2013539. Springer (2022). https:\/\/doi.org\/10.1007\/978-3-031-20059-5_30","DOI":"10.1007\/978-3-031-20059-5_30"},{"key":"8_CR74","unstructured":"Ye, Q., et\u00a0al.: mplug-owl: modularization empowers large language models with multimodality. arXiv:2304.14178 (2023)"},{"key":"8_CR75","unstructured":"You, H., et al.: Ferret: refer and ground anything anywhere at any granularity. In: The Twelfth International Conference on Learning Representations (2024). https:\/\/openreview.net\/forum?id=2msbbX3ydD"},{"key":"8_CR76","doi-asserted-by":"crossref","unstructured":"Young, P., Lai, A., Hodosh, M., Hockenmaier, J.: From image descriptions to visual denotations: New similarity metrics for semantic inference over event descriptions. In: ACL (2014)","DOI":"10.1162\/tacl_a_00166"},{"key":"8_CR77","unstructured":"Yu, W., et al.: Mm-vet: evaluating large multimodal models for integrated capabilities. In: International conference on machine learning. PMLR (2024)"},{"key":"8_CR78","unstructured":"Zhang, R., et al.: Llama-adapter: efficient fine-tuning of language models with zero-init attention. arXiv:2303.16199 (2023)"},{"key":"8_CR79","unstructured":"Zhang, S., et\u00a0al.: Opt: Open pre-trained transformer language models. arXiv:2205.01068 (2022)"},{"key":"8_CR80","unstructured":"Zhao, H.H., Zhou, P., Gao, D., Shou, M.Z.: Lova3: learning to visual question answering, asking and assessment (2024)"},{"key":"8_CR81","unstructured":"Zhu, D., Chen, J., Shen, X., Li, X., Elhoseiny, M.: Minigpt-4: enhancing vision-language understanding with advanced large language models. arXiv:2304.10592 (2023)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-73337-6_8","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,30]],"date-time":"2024-11-30T15:01:23Z","timestamp":1732978883000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-73337-6_8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,31]]},"ISBN":["9783031733369","9783031733376"],"references-count":81,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-73337-6_8","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,31]]},"assertion":[{"value":"31 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}