{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T05:30:59Z","timestamp":1781587859890,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":69,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,5,30]],"date-time":"2024-05-30T00:00:00Z","timestamp":1717027200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"the Ministry of Education, Singapore, under its Academic Research Fund Tier 2","award":["Proposal ID: T2EP20222- 0047"],"award-info":[{"award-number":["Proposal ID: T2EP20222- 0047"]}]},{"name":"the CityU MF_EXT","award":["project no. 9678180"],"award-info":[{"award-number":["project no. 9678180"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,5,30]]},"DOI":"10.1145\/3652583.3658052","type":"proceedings-article","created":{"date-parts":[[2024,6,7]],"date-time":"2024-06-07T06:30:40Z","timestamp":1717741840000},"page":"73-82","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["Improving Interpretable Embeddings for Ad-hoc Video Search with Generative Captions and Multi-word Concept Bank"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-4074-3442","authenticated-orcid":false,"given":"Jiaxin","family":"Wu","sequence":"first","affiliation":[{"name":"Department of Computer Science, City University of Hong Kong, Hong Kong, Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4182-8261","authenticated-orcid":false,"given":"Chong-Wah","family":"Ngo","sequence":"additional","affiliation":[{"name":"School of Computing and Information Systems, Singapore Management University, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7726-6235","authenticated-orcid":false,"given":"Wing-Kwong","family":"Chan","sequence":"additional","affiliation":[{"name":"Department of Computer Science, City University of Hong Kong, Hong Kong, Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,6,7]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Proceedings of the TRECVid 2018 Workshop. 1--13","author":"Avgerinakis Konstantinos","year":"2018","unstructured":"Konstantinos Avgerinakis, Anastasia Moumtzidou, Damianos Galanopoulos, Georgios Orfanidis, Stelios Andreadis, Foteini Markatopoulou, Elissavet Batziou, Konstantinos Ioannidis, Stefanos Vrochidis, Vasileios Mezaris, and Ioannis Kompatsiaris. 2018. ITI-CERTH participation in TRECVid 2018. In Proceedings of the TRECVid 2018 Workshop. 1--13."},{"key":"e_1_3_2_1_2_1","volume-title":"Proceedings of TRECVid","author":"Awad George","year":"2021","unstructured":"George Awad, Asad A. Butt, Keith Curtis, Jonathan Fiscus, Afzal Godil, Yooyoung Lee, Andrew Delgado, Jesse Zhang, Eliot Godard, Baptiste Chocot, Lukas Diduch, Jeffrey Liu, Yvette Graham, Gareth J. F. Jones, and Georges Qu\u00e9not. 2021. Evaluating Multiple Video Understanding and Retrieval Tasks at TRECVid 2021. In Proceedings of TRECVid 2021. 1--55."},{"key":"e_1_3_2_1_3_1","volume-title":"Proceedings of TRECVid","author":"Awad George","year":"2020","unstructured":"George Awad, Asad A. Butt, Keith Curtis, Yooyoung Lee, Jonathan Fiscus, Afzal Godil, Andrew Delgado, Jesse Zhang, Eliot Godard, Lukas Diduch, Jeffrey Liu, Alan F. Smeaton, Yvette Graham, Gareth J. F. Jones, Wessel Kraaij, and Georges Qu\u00e9not. 2020. TRECVid 2020: comprehensive campaign for evaluating video retrieval tasks across multiple application domains. In Proceedings of TRECVid 2020. 1--55."},{"key":"e_1_3_2_1_4_1","volume-title":"Proceedings of TRECVID","author":"Awad George","year":"2023","unstructured":"George Awad, Keith Curtis, Asad A. Butt, Jonathan Fiscus, Afzal Godil, Yooyoung Lee, Andrew Delgado, Eliot Godard, Lukas Diduch, Yvette Graham, and Georges Qu\u00e9not. 2023. TRECVID 2023 - A series of evaluation tracks in video understanding. In Proceedings of TRECVID 2023. NIST, USA, 1--23."},{"key":"e_1_3_2_1_5_1","volume-title":"Proceedings of the TRECVid 2018 Workshop.","author":"Awad George","year":"2018","unstructured":"George Awad, Asad Gov, Asad Butt, Keith Curtis, Yooyoung Lee, yooyoung@nist Gov, Jonathan Fiscus, David Joy, Andrew Delgado, Alan Smeaton, Yvette Graham, Wessel Kraaij, Georges Quenot, Joao Magalhaes, and Saverio Blasi. 2018. TRECVid 2018: Benchmarking Video Activity Detection, Video Captioning and Matching, Video Storytelling Linking and Video Search. In Proceedings of the TRECVid 2018 Workshop."},{"key":"e_1_3_2_1_6_1","volume-title":"Proceedings of the TRECVid 2016 Workshop. 1--54","author":"Awad George","year":"2016","unstructured":"George Awad, Fiscus Jonathan, Joy David, Michel Martial, Smeaton Alan, Kraaij Wessel, Quenot Georges, Eskevich Maria, Aly Robin, Ordelman Roeland, Jones Gareth, Huet Benoit, and LarsonMartha. 2016. TRECVid 2016: Evaluating Video Search, Video Event Detection, Localization, and Hyperlinking. In Proceedings of the TRECVid 2016 Workshop. 1--54."},{"key":"e_1_3_2_1_7_1","volume-title":"Frozen in Time: A Joint Video and Image Encoder for End-to-End Retrieval. In IEEE International Conference on Computer Vision. 1--15","author":"Bain Max","year":"2021","unstructured":"Max Bain, Arsha Nagrani, G\u00fcl Varol, and Andrew Zisserman. 2021. Frozen in Time: A Joint Video and Image Encoder for End-to-End Retrieval. In IEEE International Conference on Computer Vision. 1--15."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/3323873.3325051"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3131288"},{"key":"e_1_3_2_1_10_1","volume-title":"A Short Note about Kinetics-600. ArXiv","author":"Carreira Jo","year":"2018","unstructured":"Jo ao Carreira, Eric Noland, Andras Banki-Horvath, Chloe Hillier, and Andrew Zisserman. 2018. A Short Note about Kinetics-600. ArXiv , Vol. abs\/1808.01340 (2018), 1--6."},{"key":"e_1_3_2_1_11_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 10635--10644","author":"Chen S.","unstructured":"S. Chen, Y. Zhao, Q. Jin, and Q. Wu. 2020. Fine-Grained Video-Text Retrieval With Hierarchical Graph Reasoning. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 10635--10644."},{"key":"e_1_3_2_1_12_1","volume-title":"Piotr Doll\u00e1 r, and C. Lawrence Zitnick","author":"Chen Xinlei","year":"2015","unstructured":"Xinlei Chen, Hao Fang, Tsung-Yi Lin, Ramakrishna Vedantam, Saurabh Gupta, Piotr Doll\u00e1 r, and C. Lawrence Zitnick. 2015. Microsoft COCO Captions: Data Collection and Evaluation Server. CoRR , Vol. abs\/1504.00325 (2015), 1--7."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2018.2832602"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00957"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3059295"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2022.3150959"},{"key":"e_1_3_2_1_17_1","volume-title":"Proceedings of the British Machine Vision Conference. 1--13","author":"Faghri Fartash","year":"2018","unstructured":"Fartash Faghri, David J. Fleet, Jamie Ryan Kiros, and Sanja Fidlere. 2018. VSE: Improving Visual-Semantic Embeddings with Hard Negatives. In Proceedings of the British Machine Vision Conference. 1--13."},{"key":"e_1_3_2_1_18_1","volume-title":"Proceedings of the TRECVid 2019 Workshop.","author":"Francis Danny","year":"2019","unstructured":"Danny Francis, Phuong Anh Nguyen, Benoit Huet, and Chong-Wah Ngo. 2019. EURECOM at TRECVid AVS 2019. In Proceedings of the TRECVid 2019 Workshop."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3372278.3390737"},{"key":"e_1_3_2_1_20_1","volume-title":"Armand Joulin, and Ishan Misra.","author":"Girdhar Rohit","year":"2023","unstructured":"Rohit Girdhar, Alaaeldin El-Nouby, Zhuang Liu, Mannat Singh, Kalyan Vasudev Alwala, Armand Joulin, and Ishan Misra. 2023. ImageBind: One Embedding Space To Bind Them All. In CVPR. 1--11."},{"key":"e_1_3_2_1_21_1","volume-title":"Proceedings of the ACM Conference on Multimedia. 17--26","author":"Habibian Amirhossein","unstructured":"Amirhossein Habibian, Thomas Mensink, and Cees G. M. Snoek. 2014. VideoStory: A New Multimedia Embedding for Few-Example Recognition and Translation of Events. In Proceedings of the ACM Conference on Multimedia. 17--26."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19781-9_26"},{"key":"e_1_3_2_1_23_1","volume-title":"Proceedings of the TRECVid 2018 Workshop. 1--10","author":"Huang Po-Yao","year":"2018","unstructured":"Po-Yao Huang, Junwei Liang, Vaibhav, Xiaojun Chang, and Alexander Hauptmann. 2018. Informedia @ TRECVid 2018: Ad-hoc Video Search with Discrete and Continuous Representations. In Proceedings of the TRECVid 2018 Workshop. 1--10."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/2671188.2749399"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2009.2036235"},{"key":"e_1_3_2_1_26_1","unstructured":"Ranjay Krishna Yuke Zhu Justin Johnson Kenji Hata Joshua Kravitz Stephanie Chen Michael S. Bernstein and Li Fei-Fei. [n. d.]. ( [n. d.])."},{"key":"e_1_3_2_1_27_1","volume-title":"ICML 2013 Workshop : Challenges in Representation Learning (WREPL) (07","author":"Lee Dong-Hyun","year":"2013","unstructured":"Dong-Hyun Lee. 2013. Pseudo-Label : The Simple and Efficient Semi-Supervised Learning Method for Deep Neural Networks. ICML 2013 Workshop : Challenges in Representation Learning (WREPL) (07 2013), 1--6."},{"key":"e_1_3_2_1_28_1","volume-title":"Hoi","author":"Li Junnan","year":"2023","unstructured":"Junnan Li, Dongxu Li, Silvio Savarese, and Steven C. H. Hoi. 2023. BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models. ArXiv , Vol. abs\/2301.12597 (2023), 1--13."},{"key":"e_1_3_2_1_29_1","volume-title":"BLIP: Bootstrapping Language-Image Pre-training for Unified Vision-Language Understanding and Generation. In ICML. 1--12.","author":"Li Junnan","year":"2022","unstructured":"Junnan Li, Dongxu Li, Caiming Xiong, and Steven Hoi. 2022. BLIP: Bootstrapping Language-Image Pre-training for Unified Vision-Language Understanding and Generation. In ICML. 1--12."},{"key":"e_1_3_2_1_30_1","volume-title":"Proceedings of the TRECVid 2018 Workshop. 1--6.","author":"Li Xirong","year":"2018","unstructured":"Xirong Li, Jianfeng Dong, Chaoxi Xu, Jing Cao, Xun Wang, and Gang Yang. 2018. Renmin University of China and Zhejiang Gongshang University at TRECVid 2018: Deep Cross-Modal Embeddings for Video-Text Retrieval. In Proceedings of the TRECVid 2018 Workshop. 1--6."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3343031.3350906"},{"key":"e_1_3_2_1_32_1","volume-title":"SEA: Sentence Encoder Assembly for Video Retrieval by Textual Queries","author":"Li Xirong","year":"2021","unstructured":"Xirong Li, Fangming Zhou, Chaoxi Xu, Jiaqi Ji, and Gang Yang. 2021. SEA: Sentence Encoder Assembly for Video Retrieval by Textual Queries. IEEE Transactions on Multimedia (2021), 4351--4362."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.502"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3414002"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/2911996.2912015"},{"key":"e_1_3_2_1_36_1","volume-title":"CLIP4Clip: An Empirical Study of CLIP for End to End Video Clip Retrieval. arXiv preprint arXiv:2104.08860","author":"Luo Huaishao","year":"2021","unstructured":"Huaishao Luo, Lei Ji, Ming Zhong, Yang Chen, Wen Lei, Nan Duan, and Tianrui Li. 2021. CLIP4Clip: An Empirical Study of CLIP for End to End Video Clip Retrieval. arXiv preprint arXiv:2104.08860 (2021), 1--14."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3078971.3079041"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"crossref","unstructured":"Antoine Miech Dimitri Zhukov Jean-Baptiste Alayrac Makarand Tapaswi Ivan Laptev and Josef Sivic. 2019. HowTo100M: Learning a Text-Video Embedding by Watching Hundred Million Narrated Video Clips. In ICCV. 1--11.","DOI":"10.1109\/ICCV.2019.00272"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/219717.219748"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/MMUL.2006.63"},{"key":"e_1_3_2_1_41_1","volume-title":"Proceedings of the TRECVid 2017 Workshop.","author":"Nguyen Phuong Anh","year":"2017","unstructured":"Phuong Anh Nguyen, Qing Li, Zhi-Qi Cheng, Yi-Jie Lu, Hao Zhang, Xiao Wu, and Chong-Wah Ngo. 2017. VIREO @ TRECVid 2017: Video-to-Text, Ad-hoc Video Search and Video Hyperlinking. In Proceedings of the TRECVid 2017 Workshop."},{"key":"e_1_3_2_1_42_1","volume-title":"Proceedings of the TRECVid 2019 Workshop. 1--8.","author":"Nguyen Phuong Anh","year":"2019","unstructured":"Phuong Anh Nguyen, Jiaxin Wu, Chong-Wah Ngo, Francis Danny, and Huet Benoit. 2019. VIREO-EURECOM @ TRECVid 2019: Ad-hoc Video Search. In Proceedings of the TRECVid 2019 Workshop. 1--8."},{"key":"e_1_3_2_1_43_1","volume-title":"Proceedings of the TRECVid 2016 Workshop. 1--4.","author":"Nguyen Vinh-Tiep","year":"2016","unstructured":"Vinh-Tiep Nguyen, Duy-Dinh Le, Benjamin Renoust, Thanh Duc Ngo, Minh-Triet Tran, Duc Anh Duong, and Shinichi Satoh. 2016. NII-HITACHI-UIT at TRECVid 2016 Ad-hoc Video Search: Enriching Semantic Features using Multiple Neural Networks. In Proceedings of the TRECVid 2016 Workshop. 1--4."},{"key":"e_1_3_2_1_44_1","volume-title":"TRECVID 2014 -- An Overview of the Goals, Tasks, Data, Evaluation Mechanisms, and Metrics.","author":"Over Paul","year":"2014","unstructured":"Paul Over, Jon Fiscus, Gregory Sanders, David Joy, Martial Michel, George Awad, Alan Smeaton, Wessel Kraaij, and Georges Qu\u00e9not. 2014. TRECVID 2014 -- An Overview of the Goals, Tasks, Data, Evaluation Mechanisms, and Metrics."},{"key":"e_1_3_2_1_45_1","volume-title":"Proceedings of the 38th International Conference on Machine Learning, ICML. 8748--8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. In Proceedings of the 38th International Conference on Machine Learning, ICML. 8748--8763."},{"key":"e_1_3_2_1_46_1","volume-title":"Thirty-sixth Conference on Neural Information Processing Systems. 1--50","author":"Schuhmann Christoph","year":"2022","unstructured":"Christoph Schuhmann, Romain Beaumont, Richard Vencu, Cade W Gordon, Ross Wightman, Mehdi Cherti, Theo Coombes, Aarush Katta, Clayton Mullis, Mitchell Wortsman, Patrick Schramowski, Srivatsa R Kundurthy, Katherine Crowson, Ludwig Schmidt, Robert Kaczmarczyk, and Jenia Jitsev. 2022. LAION-5B: An open large-scale dataset for training next generation image-text models. In Thirty-sixth Conference on Neural Information Processing Systems. 1--50."},{"key":"e_1_3_2_1_47_1","volume-title":"Proceedings of the TRECVid 2019 Workshop.","author":"Shirahama Kimiaki","year":"2019","unstructured":"Kimiaki Shirahama, Daichi Sakurai, Takashi Matsubara, and Kuniaki Uehara. 2019. Kindai University and Kobe University at TRECVid 2019 AVS Task. In Proceedings of the TRECVid 2019 Workshop."},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1145\/1178677.1178722"},{"key":"e_1_3_2_1_49_1","volume-title":"Proceedings of the TRECVid 2017 Workshop. 1--6.","author":"Snoek Cees G. M.","unstructured":"Cees G. M. Snoek, Xirong Li, Chaoxi Xu, and Dennis C. Koelma. 2017. University of Amsterdam and Renmin University at TRECVid 2017: Searching Video, Detecting Events and Describing Video. In Proceedings of the TRECVid 2017 Workshop. 1--6."},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1561\/1500000014"},{"key":"e_1_3_2_1_51_1","volume-title":"Proceedings of the 14th ACM International Conference on Multimedia. 421--430","author":"Snoek Cees G. M.","unstructured":"Cees G. M. Snoek, Marcel Worring, Jan C. van Gemert, Jan-Mark Geusebroek, and Arnold W. M. Smeulders. 2006. The Challenge Problem for Automated Detection of 101 Semantic Concepts in Multimedia. In Proceedings of the 14th ACM International Conference on Multimedia. 421--430."},{"key":"e_1_3_2_1_52_1","volume-title":"UCF101: A Dataset of 101 Human Actions Classes From Videos in The Wild. CoRR (12","author":"Soomro Khurram","year":"2012","unstructured":"Khurram Soomro, Amir Zamir, and Mubarak Shah. 2012. UCF101: A Dataset of 101 Human Actions Classes From Videos in The Wild. CoRR (12 2012)."},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1145\/2812802"},{"key":"e_1_3_2_1_54_1","volume-title":"Proceedings of the TRECVid 2017 Workshop. 1--8.","author":"Ueki Kazuya","year":"2017","unstructured":"Kazuya Ueki, Koji Hirakawa, Kotaro Kikuchi, Tetsuji Ogawa, and Tetsunori Kobayashi. 2017. Waseda Meisei at TRECVid 2017: Ad-hoc Video Search. In Proceedings of the TRECVid 2017 Workshop. 1--8."},{"key":"e_1_3_2_1_55_1","volume-title":"Proceedings of the TRECVid 2019 Workshop. 1--7.","author":"Ueki Kazuya","year":"2019","unstructured":"Kazuya Ueki, Takayuki Hori, and Tetsunori Kobayashi. 2019. Waseda Meisei SoftBank at TRECVid 2019: Ad-hoc Video Search. In Proceedings of the TRECVid 2019 Workshop. 1--7."},{"key":"e_1_3_2_1_56_1","volume-title":"Proceedings of the TRECVid 2016 Workshop. 1--5.","author":"Ueki Kazuya","year":"2016","unstructured":"Kazuya Ueki, Kotaro Kikuchi, Susumu Saito, and Tetsunori Kobayashi. 2016. Waseda at TRECVid 2016: Ad-hoc Video Search. In Proceedings of the TRECVid 2016 Workshop. 1--5."},{"key":"e_1_3_2_1_57_1","volume-title":"Proceedings of the TRECVid 2020 Workshop. 1--7.","author":"Ueki Kazuya","year":"2020","unstructured":"Kazuya Ueki, Ryo Mutou, Takayuki Hori, Yongbeom Kim, and Yuma Suzuki. 2020. Waseda Meisei SoftBank at TRECVid 2020: Ad-hoc Video Search. In Proceedings of the TRECVid 2020 Workshop. 1--7."},{"key":"e_1_3_2_1_58_1","volume-title":"Proceedings of the TRECVid 2018 Workshop. 1--7.","author":"Ueki Kazuya","year":"2018","unstructured":"Kazuya Ueki, Yu Nakagome, Koji Hirakawa, Kotaro Kikuchi, Yoshihiko Hayashi, Tetsuji Ogawa, and Tetsunori Kobayashi. 2018. Waseda Meisei at TRECVid 2018: Ad-hoc Video Search. In Proceedings of the TRECVid 2018 Workshop. 1--7."},{"key":"e_1_3_2_1_59_1","volume-title":"Proceedings of the TRECVid 2022 Workshop. 1--5.","author":"Ueki Kazuya","year":"2022","unstructured":"Kazuya Ueki, Yuma Suzuki, Hiroki Takushima, Hideaki Okamoto, Hayato Tanoue, , and Takayuki Hori. 2022. Waseda Meisei SoftBank at TRECVID 2022. In Proceedings of the TRECVid 2022 Workshop. 1--5."},{"key":"e_1_3_2_1_60_1","volume-title":"Proceedings of the TRECVid 2023 Workshop. 1--8.","author":"Ueki Kazuya","year":"2023","unstructured":"Kazuya Ueki, Yuma Suzuki, Hiroki Takushima, Haruki Sato, Takumi Takada, Hideaki Okamoto, Hayato Tanoue, Takayuki Hori, and Aiswariya Manoj Kumar3. 2023. Waseda Meisei SoftBank at TRECVID 2023. In Proceedings of the TRECVid 2023 Workshop. 1--8."},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00468"},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3088863"},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413916"},{"key":"e_1_3_2_1_64_1","doi-asserted-by":"publisher","DOI":"10.1145\/3586993"},{"key":"e_1_3_2_1_65_1","volume-title":"SQL-Like Interpretable Interactive Video Search. In International Conference on MultiMedia Modeling. 391--397","author":"Wu Jiaxin","year":"2021","unstructured":"Jiaxin Wu, Phuong Anh Nguyen, Zhixin Ma, and Chong-Wah Ngo. 2021. SQL-Like Interpretable Interactive Video Search. In International Conference on MultiMedia Modeling. 391--397."},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.571"},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00273"},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"crossref","unstructured":"P. Young M. Hodosh A. Lai and J. Hockenmaier. 2014. From image descriptions to visual denotations: New similarity metrics for semantic inference over event descriptions. Transactions of the Association for Computational Linguistics (2014) 67--78.","DOI":"10.1162\/tacl_a_00166"},{"key":"e_1_3_2_1_69_1","volume-title":"Proceedings of the International Conference on Neural Information. 487--495","author":"Zhou Bolei","year":"2014","unstructured":"Bolei Zhou, Agata Lapedriza, Jianxiong Xiao, Antonio Torralba, and Aude Oliva. 2014. Learning Deep Features for Scene Recognition Using Places Database. In Proceedings of the International Conference on Neural Information. 487--495."}],"event":{"name":"ICMR '24: International Conference on Multimedia Retrieval","location":"Phuket Thailand","acronym":"ICMR '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia","SIGSOFT ACM Special Interest Group on Software Engineering"]},"container-title":["Proceedings of the 2024 International Conference on Multimedia Retrieval"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3652583.3658052","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3652583.3658052","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T08:50:24Z","timestamp":1755766224000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3652583.3658052"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,5,30]]},"references-count":69,"alternative-id":["10.1145\/3652583.3658052","10.1145\/3652583"],"URL":"https:\/\/doi.org\/10.1145\/3652583.3658052","relation":{},"subject":[],"published":{"date-parts":[[2024,5,30]]},"assertion":[{"value":"2024-06-07","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}