{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,12]],"date-time":"2026-08-12T16:57:59Z","timestamp":1786553879042,"version":"build-2736575974"},"publisher-location":"Cham","reference-count":60,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031729126","type":"print"},{"value":"9783031729133","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,12,2]],"date-time":"2024-12-02T00:00:00Z","timestamp":1733097600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,12,2]],"date-time":"2024-12-02T00:00:00Z","timestamp":1733097600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72913-3_1","type":"book-chapter","created":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T21:44:51Z","timestamp":1733089491000},"page":"1-17","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":11,"title":["EgoCVR: An Egocentric Benchmark for\u00a0Fine-Grained Composed Video Retrieval"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3201-360X","authenticated-orcid":false,"given":"Thomas","family":"Hummel","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-5554-1375","authenticated-orcid":false,"given":"Shyamgopal","family":"Karthik","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mariana-Iuliana","family":"Georgescu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1432-7747","authenticated-orcid":false,"given":"Zeynep","family":"Akata","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,12,2]]},"reference":[{"key":"1_CR1","doi-asserted-by":"crossref","unstructured":"Anwaar, M.U., Labintcev, E., Kleinsteuber, M.: Compositional learning of image-text query for image retrieval. In: WACV (2021)","DOI":"10.1109\/WACV48630.2021.00118"},{"key":"1_CR2","doi-asserted-by":"crossref","unstructured":"Ashutosh, K., Xue, Z., Nagarajan, T., Grauman, K.: Detours for navigating instructional videos. In: CVPR (2024)","DOI":"10.1109\/CVPR52733.2024.01779"},{"key":"1_CR3","unstructured":"Xu, X., et al.: Sentence-level prompts benefit composed image retrieval. In: ICLR (2024)"},{"key":"1_CR4","doi-asserted-by":"crossref","unstructured":"Bain, M., Nagrani, A., Varol, G., Zisserman, A.: Frozen in time: a joint video and image encoder for end-to-end retrieval. In: ICCV (2021)","DOI":"10.1109\/ICCV48922.2021.00175"},{"key":"1_CR5","unstructured":"Bain, M., Nagrani, A., Varol, G., Zisserman, A.: A clip-hitchhiker\u2019s guide to long video retrieval. arXiv preprint arXiv:2205.08508 (2022)"},{"key":"1_CR6","doi-asserted-by":"crossref","unstructured":"Baldrati, A., Agnolucci, L., Bertini, M., Del\u00a0Bimbo, A.: Zero-shot composed image retrieval with textual inversion. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.01407"},{"key":"1_CR7","doi-asserted-by":"crossref","unstructured":"Baldrati, A., Bertini, M., Uricchio, T., Del\u00a0Bimbo, A.: Effective conditioned and composed image retrieval combining clip-based features. In: CVPR Workshops (2022)","DOI":"10.1109\/CVPR52688.2022.02080"},{"key":"1_CR8","unstructured":"Bommasani, R., et\u00a0al.: On the opportunities and risks of foundation models. arXiv preprint arXiv:2108.07258 (2021)"},{"key":"1_CR9","unstructured":"Chen, J., Lai, H.: Pretrain like you inference: masked tuning improves zero-shot composed image retrieval. arXiv preprint arXiv:2311.07622 (2023)"},{"key":"1_CR10","unstructured":"Chen, S., et al.: Vast: a vision-audio-subtitle-text omni-modality foundation model and dataset. In: NeurIPS (2023)"},{"key":"1_CR11","doi-asserted-by":"crossref","unstructured":"Chen, Y., Bazzani, L.: Learning joint visual semantic matching embeddings for language-guided retrieval. In: ECCV (2020)","DOI":"10.1007\/978-3-030-58542-6_9"},{"key":"1_CR12","doi-asserted-by":"crossref","unstructured":"Chen, Y., Gong, S., Bazzani, L.: Image search with text feedback by visiolinguistic attention learning. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00307"},{"key":"1_CR13","unstructured":"Delmas, G., Rezende, R.S., Csurka, G., Larlus, D.: ARTEMIS: attention-based retrieval with text-explicit matching and implicit similarity. In: ICLR (2022)"},{"key":"1_CR14","doi-asserted-by":"crossref","unstructured":"Dong, J., Li, X., Snoek, C.G.: Predicting visual features from text for image and video caption retrieval. IEEE Trans. Multimedia (2018)","DOI":"10.1109\/TMM.2018.2832602"},{"key":"1_CR15","unstructured":"Dong, Q., et al.: A survey on in-context learning (2023)"},{"key":"1_CR16","unstructured":"Dosovitskiy, A., et\u00a0al.: An image is worth 16x16 words: transformers for image recognition at scale. In: ICLR (2021)"},{"key":"1_CR17","doi-asserted-by":"crossref","unstructured":"Forbes, M., Kaeser-Chen, C., Sharma, P., Belongie, S.: Neural naturalist: generating fine-grained image comparisons. In: EMNLP (2019)","DOI":"10.18653\/v1\/D19-1065"},{"key":"1_CR18","unstructured":"Gadre, S.Y., et\u00a0al.: Datacomp: in search of the next generation of multimodal datasets. In: NeurIPS (2023)"},{"key":"1_CR19","doi-asserted-by":"crossref","unstructured":"Gatti, P., Parikh, K.G., Paul, D.P., Gupta, M., Mishra, A.: Composite sketch+ text queries for retrieving objects with elusive names and complex interactions. In: AAAI (2024)","DOI":"10.1609\/aaai.v38i3.27956"},{"key":"1_CR20","doi-asserted-by":"crossref","unstructured":"Girdhar, R., et al.: Imagebind: one embedding space to bind them all. In: CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.01457"},{"key":"1_CR21","unstructured":"Grauman, K., et\u00a0al.: Ego4D: around the world in 3,000 hours of egocentric video. In: CVPR (2022)"},{"key":"1_CR22","unstructured":"Grauman, K., et\u00a0al.: Ego-exo4d: understanding skilled human activity from first- and third-person perspectives. In: CVPR (2024)"},{"key":"1_CR23","unstructured":"Gu, G., Chun, S., Kim, W., Jun, H., Kang, Y., Yun, S.: Compodiff: versatile composed image retrieval with latent diffusion. arXiv preprint arXiv:2303.11916 (2023)"},{"key":"1_CR24","doi-asserted-by":"crossref","unstructured":"Gu, G., Chun, S., Kim, W., Kang, Y., Yun, S.: Language-only efficient training of zero-shot composed image retrieval. In: CVPR (2024)","DOI":"10.1109\/CVPR52733.2024.01256"},{"key":"1_CR25","doi-asserted-by":"crossref","unstructured":"Han, X., et al.: Automatic spatially-aware fashion concept discovery. In: ICCV (2017)","DOI":"10.1109\/ICCV.2017.163"},{"key":"1_CR26","doi-asserted-by":"publisher","unstructured":"Ilharco, G., et al.: OpenCLIP. https:\/\/doi.org\/10.5281\/zenodo.5143773","DOI":"10.5281\/zenodo.5143773"},{"key":"1_CR27","unstructured":"Jia, C., et al.: Scaling up visual and vision-language representation learning with noisy text supervision. In: ICML (2021)"},{"key":"1_CR28","unstructured":"Karthik, S., Roth, K., Mancini, M., Akata, Z.: Vision-by-language for training-free compositional image retrieval. In: ICLR (2024)"},{"key":"1_CR29","doi-asserted-by":"crossref","unstructured":"Lee, S., Kim, D., Han, B.: Cosmo: content-style modulation for image retrieval with text feedback. In: CVPR (2021)","DOI":"10.1109\/CVPR46437.2021.00086"},{"key":"1_CR30","doi-asserted-by":"crossref","unstructured":"Levy, M., Ben-Ari, R., Darshan, N., Lischinski, D.: Data roaming and early fusion for composed image retrieval. In: AAAI (2024)","DOI":"10.1609\/aaai.v38i4.28081"},{"key":"1_CR31","unstructured":"Li, J., Li, D., Savarese, S., Hoi, S.: Blip-2: bootstrapping language-image pre-training with frozen image encoders and large language models. In: ICML (2023)"},{"key":"1_CR32","unstructured":"Li, J., Li, D., Xiong, C., Hoi, S.: Blip: bootstrapping language-image pre-training for unified vision-language understanding and generation. In: ICML (2022)"},{"key":"1_CR33","doi-asserted-by":"crossref","unstructured":"Lin, B., Zhu, B., Ye, Y., Ning, M., Jin, P., Yuan, L.: Video-llava: learning united visual representation by alignment before projection. arXiv preprint arXiv:2311.10122 (2023)","DOI":"10.18653\/v1\/2024.emnlp-main.342"},{"key":"1_CR34","unstructured":"Lin, K.Q., et\u00a0al.: Egocentric video-language pretraining. In: NeurIPS (2022)"},{"key":"1_CR35","doi-asserted-by":"crossref","unstructured":"Lin, T.Y., et al.: Microsoft coco: common objects in context. In: ECCV (2014)","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"1_CR36","unstructured":"Liu, Y., Yao, J., Zhang, Y., Wang, Y., Xie, W.: Zero-shot composed text-image retrieval. In: BMVC (2023)"},{"key":"1_CR37","doi-asserted-by":"crossref","unstructured":"Liu, Z., Rodriguez-Opazo, C., Teney, D., Gould, S.: Image retrieval on real-life images with pre-trained vision-and-language models. In: ICCV (2021)","DOI":"10.1109\/ICCV48922.2021.00213"},{"key":"1_CR38","doi-asserted-by":"crossref","unstructured":"Luo, H., et al.: Clip4clip: an empirical study of clip for end to end video clip retrieval and captioning. Neurocomputing (2022)","DOI":"10.1016\/j.neucom.2022.07.028"},{"key":"1_CR39","doi-asserted-by":"crossref","unstructured":"Miech, A., Zhukov, D., Alayrac, J.B., Tapaswi, M., Laptev, I., Sivic, J.: HowTo100M: learning a text-video embedding by watching hundred million narrated video clips. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00272"},{"key":"1_CR40","unstructured":"OpenAI: GPT-4 Technical Report. arXiv abs\/2303.08774 (2023)"},{"key":"1_CR41","doi-asserted-by":"crossref","unstructured":"Otani, M., Nakashima, Y., Rahtu, E., Heikkil\u00e4, J., Yokoya, N.: Learning joint representations of videos and sentences with web image search. In: ECCV Workshops (2016)","DOI":"10.1007\/978-3-319-46604-0_46"},{"key":"1_CR42","doi-asserted-by":"crossref","unstructured":"Pramanick, S., et al.: EgoVLPv2: egocentric video-language pre-training with fusion in the backbone. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.00487"},{"key":"1_CR43","unstructured":"Radford, A., et\u00a0al.: Learning transferable visual models from natural language supervision. In: ICML (2021)"},{"key":"1_CR44","doi-asserted-by":"crossref","unstructured":"Saito, K., et al.: Pic2word: mapping pictures to words for zero-shot composed image retrieval. In: CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.01850"},{"key":"1_CR45","unstructured":"Sun, S., Ye, F., Gong, S.: Training-free zero-shot composed image retrieval with local concept reranking. arXiv preprint arXiv:2312.08924 (2023)"},{"key":"1_CR46","doi-asserted-by":"crossref","unstructured":"Tang, Y., et al.: Context-i2w: mapping images to context-dependent words for accurate zero-shot composed image retrieval. In: AAAI (2024)","DOI":"10.1609\/aaai.v38i6.28324"},{"key":"1_CR47","doi-asserted-by":"crossref","unstructured":"Thawakar, O., et al.: Composed video retrieval via enriched context and discriminative embeddings. In: CVPR (2024)","DOI":"10.1109\/CVPR52733.2024.02540"},{"key":"1_CR48","unstructured":"Torabi, A., Tandon, N., Sigal, L.: Learning language-visual embedding for movie understanding with natural-language. arXiv preprint arXiv:1609.08124 (2016)"},{"key":"1_CR49","unstructured":"Touvron, H., et\u00a0al.: Llama 2: open foundation and fine-tuned chat models. arXiv preprint arXiv:2307.09288 (2023)"},{"key":"1_CR50","doi-asserted-by":"crossref","unstructured":"Vaze, S., Carion, N., Misra, I.: Genecis: a benchmark for general conditional image similarity. In: CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.00663"},{"key":"1_CR51","doi-asserted-by":"crossref","unstructured":"Ventura, L., Yang, A., Schmid, C., Varol, G.: CoVR: learning composed video retrieval from web video captions. In: AAAI (2024)","DOI":"10.1007\/s11263-024-02202-8"},{"key":"1_CR52","doi-asserted-by":"crossref","unstructured":"Vo, N., et al.: Composing text and image for image retrieval-an empirical odyssey. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00660"},{"key":"1_CR53","doi-asserted-by":"crossref","unstructured":"Wang, X., Wu, J., Chen, J., Li, L., Wang, Y.F., Wang, W.Y.: VATEX: a large-scale, high-quality multilingual dataset for video-and-language research. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00468"},{"key":"1_CR54","unstructured":"Wu, H., et al.: The fashion IQ dataset: retrieving images by combining side information and relative natural language feedback. In: CVPR (2021)"},{"key":"1_CR55","unstructured":"Xu, H., et\u00a0al.: mPLUG-2: a modularized multi-modal foundation model across text, image and video. In: ICML (2023)"},{"key":"1_CR56","doi-asserted-by":"crossref","unstructured":"Xu, J., Mei, T., Yao, T., Rui, Y.: MSR-VTT: a large video description dataset for bridging video and language. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.571"},{"key":"1_CR57","doi-asserted-by":"crossref","unstructured":"Xu, R., Xiong, C., Chen, W., Corso, J.: Jointly modeling deep video and compositional text to bridge vision and language in a unified framework. In: AAAI (2015)","DOI":"10.1609\/aaai.v29i1.9512"},{"key":"1_CR58","unstructured":"Zhao, L., et\u00a0al.: VideoPrism: a foundational visual encoder for video understanding. arXiv preprint arXiv:2402.13217 (2024)"},{"key":"1_CR59","doi-asserted-by":"crossref","unstructured":"Zhao, Y., Misra, I., Kr\u00e4henb\u00fchl, P., Girdhar, R.: Learning video representations from large language models. In: CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.00637"},{"key":"1_CR60","unstructured":"Zhu, B., et\u00a0al.: LanguageBind: extending video-language pretraining to n-modality by language-based semantic alignment. In: ICLR (2024)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72913-3_1","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T23:21:35Z","timestamp":1733095295000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72913-3_1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,2]]},"ISBN":["9783031729126","9783031729133"],"references-count":60,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72913-3_1","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,12,2]]},"assertion":[{"value":"2 December 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}