{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:50:34Z","timestamp":1765309834971,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":42,"publisher":"ACM","funder":[{"name":"Icelandic Research Fund","award":["239772-051"],"award-info":[{"award-number":["239772-051"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3760243","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T06:54:17Z","timestamp":1761375257000},"page":"14273-14279","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Interactive Retrieval System for Multi-Stream Collections: multiXview at CASTLE 2025 Interactive Grand Challenge"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9720-3645","authenticated-orcid":false,"given":"Omar Shahbaz","family":"Khan","sequence":"first","affiliation":[{"name":"Reykjavik University, Reykjavik, Iceland"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0285-1303","authenticated-orcid":false,"given":"Ujjwal","family":"Sharma","sequence":"additional","affiliation":[{"name":"University of Amsterdam, Amsterdam, Netherlands"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-1015-6845","authenticated-orcid":false,"given":"Gon\u00e7alo","family":"Marcelino","sequence":"additional","affiliation":[{"name":"University of Amsterdam, Amsterdam, Netherlands"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9825-1654","authenticated-orcid":false,"given":"Aaron","family":"Duane","sequence":"additional","affiliation":[{"name":"HCI, Churney, Copenhagen, Denmark"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1904-8736","authenticated-orcid":false,"given":"Stevan","family":"Rudinac","sequence":"additional","affiliation":[{"name":"University of Amsterdam, Amsterdam, Netherlands"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4097-4136","authenticated-orcid":false,"given":"Marcel","family":"Worring","sequence":"additional","affiliation":[{"name":"University of Amsterdam, Amsterdam, Netherlands"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0889-3491","authenticated-orcid":false,"given":"Bj\u00f6rn \u00de\u00f3r","family":"J\u00f3nsson","sequence":"additional","affiliation":[{"name":"Reykjavik University, Reykjav\u00edk, Iceland"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"doi-asserted-by":"publisher","key":"e_1_3_2_1_1_1","DOI":"10.1145\/3643489.3661130"},{"key":"e_1_3_2_1_2_1","volume-title":"TRECVID 2024 - Evaluating video search, captioning, and activity recognition. In Proceedings of TRECVID 2024. NIST, USA.","author":"Awad George","year":"2024","unstructured":"George Awad, Jonathan Fiscus, Afzal Godil, Lukas Diduch, Yvette Graham, and Georges Qu\u00e9not. 2024. TRECVID 2024 - Evaluating video search, captioning, and activity recognition. In Proceedings of TRECVID 2024. NIST, USA."},{"key":"e_1_3_2_1_3_1","volume-title":"Proceedings of the European Conference on Computer Vision (ECCV).","author":"Damen Dima","year":"2018","unstructured":"Dima Damen, Hazel Doughty, Giovanni Maria Farinella, Sanja Fidler, Antonino Furnari, Evangelos Kazakos, Davide Moltisanti, Jonathan Munro, Toby Perrett, Will Price, and Michael Wray. 2018. Scaling Egocentric Vision: The EPIC-KITCHENS Dataset. In Proceedings of the European Conference on Computer Vision (ECCV)."},{"key":"e_1_3_2_1_4_1","volume-title":"Antonino Furnari, Evangelos Kazakos, Jian Ma, Davide Moltisanti, Jonathan Munro, Toby Perrett, Will Price, et al.","author":"Damen Dima","year":"2022","unstructured":"Dima Damen, Hazel Doughty, Giovanni Maria Farinella, Antonino Furnari, Evangelos Kazakos, Jian Ma, Davide Moltisanti, Jonathan Munro, Toby Perrett, Will Price, et al., 2022. Rescaling egocentric vision: Collection, pipeline and challenges for epic-kitchens-100. International Journal of Computer Vision (2022), 1-23."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_5_1","DOI":"10.1109\/WACV51458.2022.00026"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_6_1","DOI":"10.1109\/TSMC.2016.2531671"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_7_1","DOI":"10.1109\/CVPR.2015.7299176"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_8_1","DOI":"10.1109\/TMM.2010.2052025"},{"volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","author":"Gan Chuang","unstructured":"Chuang Gan, Naiyan Wang, Yi Yang, Dit-Yan Yeung, and Alex G. Hauptmann. 2015. DevNet: A Deep Event Network for Multimedia Event Detection and Evidence Recounting. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","key":"e_1_3_2_1_9_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_10_1","DOI":"10.1145\/1878137.1878145"},{"key":"e_1_3_2_1_11_1","volume-title":"Lifelogging: Personal big data. Foundations and Trends\u00ae in information retrieval","author":"Gurrin Cathal","year":"2014","unstructured":"Cathal Gurrin, Alan F Smeaton, Aiden R Doherty, et al., 2014. Lifelogging: Personal big data. Foundations and Trends\u00ae in information retrieval, Vol. 8, 1 (2014), 1-125."},{"key":"e_1_3_2_1_12_1","first-page":"1334","volume-title":"LSC'24","author":"Gurrin Cathal","year":"2024","unstructured":"Cathal Gurrin, Liting Zhou, Graham Healy, Werner Bailer, Duc-Tien Dang Nguyen, Steve Hodges, Bj\u00f6rn \u00de\u00f3r J\u00f3nsson, Luca Rossetto, Minh-Triet Tran, and Klaus Sch\u00f6ffmann. 2024. Introduction to the Seventh Annual Lifelog Search Challenge, LSC'24. In Proceedings of the 2024 International Conference on Multimedia Retrieval (ICMR '24). ACM, 1334-1335."},{"key":"e_1_3_2_1_13_1","first-page":"17","volume-title":"Proceedings of the 22nd ACM International Conference on Multimedia (MM '14)","author":"Habibian Amirhossein","unstructured":"Amirhossein Habibian, Thomas Mensink, and Cees G.M. Snoek. 2014. VideoStory: A New Multimedia Embedding for Few-Example Recognition and Translation of Events. In Proceedings of the 22nd ACM International Conference on Multimedia (MM '14). ACM, 17-26."},{"key":"e_1_3_2_1_14_1","volume-title":"Working Notes Proceedings of the MediaEval 2023 Workshop, Amsterdam, The Netherlands and Online, 1-2 February 2024. CEUR Workshop Proceedings","volume":"3658","author":"Hicks Steven","unstructured":"Steven Hicks, Andreas Lommatzsch, Ali H\u00fcrriyetoglu, Romain Vuillemot, Mihai Gabriel Constantin, Vajira Thambawita, and Martha A. Larson (Eds.). 2024. Working Notes Proceedings of the MediaEval 2023 Workshop, Amsterdam, The Netherlands and Online, 1-2 February 2024. CEUR Workshop Proceedings, Vol. 3658. CEUR-WS.org. https:\/\/ceur-ws.org\/Vol-3658"},{"key":"e_1_3_2_1_15_1","volume-title":"European Conference on Computer Vision. Springer, 1-17","author":"Hummel Thomas","year":"2024","unstructured":"Thomas Hummel, Shyamgopal Karthik, Mariana-Iuliana Georgescu, and Zeynep Akata. 2024. Egocvr: An egocentric benchmark for fine-grained composed video retrieval. In European Conference on Computer Vision. Springer, 1-17."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_16_1","DOI":"10.1007\/s11042-020-08806-9"},{"key":"e_1_3_2_1_17_1","volume-title":"The Curious Case of High-Dimensional Indexing as a File Structure: A Case Study of eCP-FS. arXiv preprint arXiv:2507.21939","author":"Khan Omar Shahbaz","year":"2025","unstructured":"Omar Shahbaz Khan, Gylfi \u00de\u00f3r Gu\u00f0mundsson, and Bj\u00f6rn \u00de\u00f3r J\u00f3nsson. 2025. The Curious Case of High-Dimensional Indexing as a File Structure: A Case Study of eCP-FS. arXiv preprint arXiv:2507.21939 (2025)."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_18_1","DOI":"10.1007\/978-3-030-45439-5_33"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_19_1","DOI":"10.1145\/3643489.3661132"},{"volume-title":"MultiMedia Modeling","author":"Khan Omar Shahbaz","unstructured":"Omar Shahbaz Khan, Hongyi Zhu, Ujjwal Sharma, Evangelos Kanoulas, Stevan Rudinac, and Bj\u00f6rn \u00de\u00f3r J\u00f3nsson. 2024c. Exquisitor at the Video Browser Showdown 2024: Relevance Feedback Meets Conversational Search. In MultiMedia Modeling. Springer Nature Switzerland, 347-355.","key":"e_1_3_2_1_20_1"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_21_1","DOI":"10.1145\/3643489.3661121"},{"key":"e_1_3_2_1_22_1","volume-title":"2012 IEEE conference on computer vision and pattern recognition. IEEE, 1346-1353","author":"Lee Yong Jae","year":"2012","unstructured":"Yong Jae Lee, Joydeep Ghosh, and Kristen Grauman. 2012. Discovering important people and objects for egocentric video summarization. In 2012 IEEE conference on computer vision and pattern recognition. IEEE, 1346-1353."},{"key":"e_1_3_2_1_23_1","volume-title":"Proceedings of the 40th International Conference on Machine Learning (ICML'23)","author":"Li Junnan","year":"2023","unstructured":"Junnan Li, Dongxu Li, Silvio Savarese, and Steven Hoi. 2023. BLIP-2: bootstrapping language-image pre-training with frozen image encoders and large language models. In Proceedings of the 40th International Conference on Machine Learning (ICML'23). JMLR.org, Article 814, 13 pages."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_24_1","DOI":"10.1016\/j.knosys.2023.110986"},{"doi-asserted-by":"crossref","unstructured":"Jakub Lokos Stelios Andreadis Werner Bailer Aaron Duane Cathal Gurrin Zhixin Ma Nicola Messina Thao-Nhu Nguyen Ladislav Pe\u0161ka Luca Rossetto et al. 2023. Interactive video retrieval in the age of effective joint embedding deep models: lessons from the 11th VBS. Multimedia Systems (2023) 1-24.","key":"e_1_3_2_1_25_1","DOI":"10.1007\/s00530-023-01143-5"},{"key":"e_1_3_2_1_26_1","volume-title":"Learning Transferable Visual Models From Natural Language Supervision. In International Conference on Machine Learning. PMLR, 8748-8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al., 2021. Learning Transferable Visual Models From Natural Language Supervision. In International Conference on Machine Learning. PMLR, 8748-8763."},{"key":"e_1_3_2_1_27_1","volume-title":"International conference on machine learning. PMLR, 28492-28518","author":"Radford Alec","year":"2023","unstructured":"Alec Radford, Jong Wook Kim, Tao Xu, Greg Brockman, Christine McLeavey, and Ilya Sutskever. 2023. Robust speech recognition via large-scale weak supervision. In International conference on machine learning. PMLR, 28492-28518."},{"key":"e_1_3_2_1_28_1","volume-title":"The CASTLE 2024 Dataset: Advancing the Art of Multimodal Understanding. arXiv preprint arXiv:2503","author":"Rossetto Luca","year":"2025","unstructured":"Luca Rossetto, Werner Bailer, Duc-Tien Dang-Nguyen, Graham Healy, Bj\u00f6rn \u00de\u00f3r J\u00f3nsson, Onanong Kongmeesub, Hoang-Bao Le, Stevan Rudinac, Klaus Sch\u00f6ffmann, Florian Spiess, et al., 2025. The CASTLE 2024 Dataset: Advancing the Art of Multimodal Understanding. arXiv preprint arXiv:2503.17116 (2025)."},{"key":"e_1_3_2_1_29_1","volume-title":"MMM 2025, Nara, Japan, January 8-10, 2025, Proceedings, Part V (Lecture Notes in Computer Science","volume":"277","author":"Rossetto Luca","year":"2025","unstructured":"Luca Rossetto and Ralph Gasser. 2025. Feature-Driven Video Segmentation and Advanced Querying with vitrivr-Engine. In MultiMedia Modeling - 31st International Conference on Multimedia Modeling, MMM 2025, Nara, Japan, January 8-10, 2025, Proceedings, Part V (Lecture Notes in Computer Science, Vol. 15524). Springer, 272-277."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_30_1","DOI":"10.1007\/s13735-012-0018-0"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_31_1","DOI":"10.1109\/CVPRW67362.2025.00478"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_32_1","DOI":"10.1109\/ICIP.1997.638621"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_33_1","DOI":"10.1109\/CBMI.2019.8877397"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_34_1","DOI":"10.1109\/CVPRW67362.2025.00360"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_35_1","DOI":"10.1109\/CVPRW67362.2025.00359"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_36_1","DOI":"10.1007\/978-981-96-2074-6_31"},{"key":"e_1_3_2_1_37_1","volume-title":"The MediaMill TRECVID 2009 semantic video search engine. In TRECVID workshop.","author":"Snoek Cees","year":"2009","unstructured":"Cees Snoek, Kvd Sande, OD Rooij, Bouke Huurnink, J Uijlings, M v Liempt, M Bugalhoy, I Trancosoy, F Yan, M Tahir, et al., 2009. The MediaMill TRECVID 2009 semantic video search engine. In TRECVID workshop."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_38_1","DOI":"10.1007\/978-981-96-2074-6_39"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_39_1","DOI":"10.1109\/2.493456"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_40_1","DOI":"10.1109\/TPAMI.2020.2975798"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_41_1","DOI":"10.1145\/1290082.1290111"},{"key":"e_1_3_2_1_42_1","first-page":"1243","article-title":"Shared Multi-View Data Representation for Multi-Domain Event Detection","volume":"42","author":"Yang Zhenguo","year":"2020","unstructured":"Zhenguo Yang, Qing Li, Wenyin Liu, and Jianming Lv. 2020. Shared Multi-View Data Representation for Multi-Domain Event Detection. IEEE Transactions on Pattern Analysis and Machine Intelligence, Vol. 42, 5 (2020), 1243-1256.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"}],"event":{"sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"acronym":"MM '25","name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland"},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3760243","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:46:21Z","timestamp":1765309581000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3760243"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":42,"alternative-id":["10.1145\/3746027.3760243","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3760243","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}