{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T17:08:28Z","timestamp":1785604108477,"version":"3.56.0"},"publisher-location":"New York, NY, USA","reference-count":66,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,10,10]],"date-time":"2022-10-10T00:00:00Z","timestamp":1665360000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"the National Natural Science Foundation of China","award":["No. U21B2037, No. 62176222, No. 62176223, No. 62176226, No. 62072386, No. 62072387, No. 62072389, No. 62002305"],"award-info":[{"award-number":["No. U21B2037, No. 62176222, No. 62176223, No. 62176226, No. 62072386, No. 62072387, No. 62072389, No. 62002305"]}]},{"name":"the Natural Science Foundation of Fujian Province of China","award":["No.2021J01002"],"award-info":[{"award-number":["No.2021J01002"]}]},{"name":"Guangdong Basic and Applied Basic Research Foundatio","award":["No.2019B1515120049"],"award-info":[{"award-number":["No.2019B1515120049"]}]},{"name":"the National Science Fund for Distinguished Young Scholars","award":["No.62025603"],"award-info":[{"award-number":["No.62025603"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,10,10]]},"DOI":"10.1145\/3503161.3547910","type":"proceedings-article","created":{"date-parts":[[2022,10,10]],"date-time":"2022-10-10T15:43:01Z","timestamp":1665416581000},"page":"638-647","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":298,"title":["X-CLIP: End-to-End Multi-grained Contrastive Learning for Video-Text Retrieval"],"prefix":"10.1145","author":[{"given":"Yiwei","family":"Ma","sequence":"first","affiliation":[{"name":"Xiamen University, Xiamen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Guohai","family":"Xu","sequence":"additional","affiliation":[{"name":"Alibaba Group, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaoshuai","family":"Sun","sequence":"additional","affiliation":[{"name":"Xiamen University, Xiamen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ming","family":"Yan","sequence":"additional","affiliation":[{"name":"Alibaba Group, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ji","family":"Zhang","sequence":"additional","affiliation":[{"name":"Alibaba Group, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rongrong","family":"Ji","sequence":"additional","affiliation":[{"name":"Xiamen University, Xiamen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2022,10,10]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"Noise estimation using density estimation for self-supervised multimodal learning. arXiv preprint arXiv:2003.03186 8","author":"Amrani Elad","year":"2020","unstructured":"Elad Amrani , Rami Ben-Ari , Daniel Rotman , and Alex Bronstein . 2020. Noise estimation using density estimation for self-supervised multimodal learning. arXiv preprint arXiv:2003.03186 8 ( 2020 ). Elad Amrani, Rami Ben-Ari, Daniel Rotman, and Alex Bronstein. 2020. Noise estimation using density estimation for self-supervised multimodal learning. arXiv preprint arXiv:2003.03186 8 (2020)."},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.618"},{"key":"e_1_3_2_2_3_1","volume-title":"Vivit: A video vision transformer. arXiv preprint arXiv:2103.15691","author":"Arnab Anurag","year":"2021","unstructured":"Anurag Arnab , Mostafa Dehghani , Georg Heigold , Chen Sun , Mario Lucic , and Cordelia Schmid . 2021 . Vivit: A video vision transformer. arXiv preprint arXiv:2103.15691 (2021). Anurag Arnab, Mostafa Dehghani, Georg Heigold, Chen Sun, Mario Lucic, and Cordelia Schmid. 2021. Vivit: A video vision transformer. arXiv preprint arXiv:2103.15691 (2021)."},{"key":"e_1_3_2_2_4_1","volume-title":"Frozen in time: A joint video and image encoder for end-to-end retrieval. arXiv preprint arXiv:2104.00650","author":"Bain Max","year":"2021","unstructured":"Max Bain , Arsha Nagrani , G\u00fcl Varol , and Andrew Zisserman . 2021. Frozen in time: A joint video and image encoder for end-to-end retrieval. arXiv preprint arXiv:2104.00650 ( 2021 ). Max Bain, Arsha Nagrani, G\u00fcl Varol, and Andrew Zisserman. 2021. Frozen in time: A joint video and image encoder for end-to-end retrieval. arXiv preprint arXiv:2104.00650 (2021)."},{"key":"e_1_3_2_2_5_1","volume-title":"Is Space-Time At- tention All You Need for Video Understanding? arXiv preprint arXiv:2102.05095","author":"Bertasius Gedas","year":"2021","unstructured":"Gedas Bertasius , Heng Wang , and Lorenzo Torresani . 2021. Is Space-Time At- tention All You Need for Video Understanding? arXiv preprint arXiv:2102.05095 ( 2021 ). Gedas Bertasius, Heng Wang, and Lorenzo Torresani. 2021. Is Space-Time At- tention All You Need for Video Understanding? arXiv preprint arXiv:2102.05095 (2021)."},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298698"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.5555\/2002472.2002497"},{"key":"e_1_3_2_2_8_1","unstructured":"Ting Chen Simon Kornblith Mohammad Norouzi and Geoffrey Hinton. 2020. A simple framework for contrastive learning of visual representations. In Interna- tional conference on machine learning. PMLR 1597--1607.  Ting Chen Simon Kornblith Mohammad Norouzi and Geoffrey Hinton. 2020. A simple framework for contrastive learning of visual representations. In Interna- tional conference on machine learning. PMLR 1597--1607."},{"key":"e_1_3_2_2_9_1","volume-title":"Improved baselines with momentum contrastive learning. arXiv preprint arXiv:2003.04297","author":"Chen Xinlei","year":"2020","unstructured":"Xinlei Chen , Haoqi Fan , Ross Girshick , and Kaiming He. 2020. Improved baselines with momentum contrastive learning. arXiv preprint arXiv:2003.04297 ( 2020 ). Xinlei Chen, Haoqi Fan, Ross Girshick, and Kaiming He. 2020. Improved baselines with momentum contrastive learning. arXiv preprint arXiv:2003.04297 (2020)."},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00950"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01138"},{"key":"e_1_3_2_2_12_1","volume-title":"Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805","author":"Devlin Jacob","year":"2018","unstructured":"Jacob Devlin , Ming-Wei Chang , Kenton Lee , and Kristina Toutanova . 2018 . Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018). Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2018. Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)."},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00957"},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW53098.2021.00374"},{"key":"e_1_3_2_2_15_1","volume-title":"Proceedings, Part IV 16","author":"Gabeur Valentin","year":"2020","unstructured":"Valentin Gabeur , Chen Sun , Karteek Alahari , and Cordelia Schmid . 2020 . Multi- modal transformer for video retrieval. In Computer Vision--ECCV 2020: 16th Euro- pean Conference, Glasgow, UK, August 23--28, 2020 , Proceedings, Part IV 16 . Springer, 214--229. Valentin Gabeur, Chen Sun, Karteek Alahari, and Cordelia Schmid. 2020. Multi- modal transformer for video retrieval. In Computer Vision--ECCV 2020: 16th Euro- pean Conference, Glasgow, UK, August 23--28, 2020, Proceedings, Part IV 16. Springer, 214--229."},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3404835.3462927"},{"key":"e_1_3_2_2_17_1","volume-title":"PixelFolder: An Efficient Progressive Pixel Synthesis Network for Image Generation. arXiv preprint arXiv:2204.00833","author":"He Jing","year":"2022","unstructured":"Jing He , Yiyi Zhou , Qi Zhang , Jun Peng , Yunhang Shen , Xiaoshuai Sun , Chao Chen , and Rongrong Ji. 2022. PixelFolder: An Efficient Progressive Pixel Synthesis Network for Image Generation. arXiv preprint arXiv:2204.00833 ( 2022 ). Jing He, Yiyi Zhou, Qi Zhang, Jun Peng, Yunhang Shen, Xiaoshuai Sun, Chao Chen, and Rongrong Ji. 2022. PixelFolder: An Efficient Progressive Pixel Synthesis Network for Image Generation. arXiv preprint arXiv:2204.00833 (2022)."},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"e_1_3_2_2_19_1","volume-title":"et al","author":"Huo Yuqi","year":"2021","unstructured":"Yuqi Huo , Manli Zhang , Guangzhen Liu , Haoyu Lu , Yizhao Gao , Guoxing Yang , Jingyuan Wen , Heng Zhang , Baogui Xu , Weihao Zheng , et al . 2021 . WenLan : Bridging vision and language by large-scale multi-modal pre-training. arXiv preprint arXiv:2103.06561 (2021). Yuqi Huo, Manli Zhang, Guangzhen Liu, Haoyu Lu, Yizhao Gao, Guoxing Yang, Jingyuan Wen, Heng Zhang, Baogui Xu, Weihao Zheng, et al . 2021. WenLan: Bridging vision and language by large-scale multi-modal pre-training. arXiv preprint arXiv:2103.06561 (2021)."},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.149"},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i2.16258"},{"key":"e_1_3_2_2_22_1","volume-title":"Knowing What to Learn: A Metric-oriented Focal Mechanism for Image Captioning","author":"Ji Jiayi","year":"2022","unstructured":"Jiayi Ji , Yiwei Ma , Xiaoshuai Sun , Yiyi Zhou , Yongjian Wu , and Rongrong Ji. 2022. Knowing What to Learn: A Metric-oriented Focal Mechanism for Image Captioning . IEEE Transactions on Image Processing ( 2022 ), 1--1. https:\/\/doi.org\/ 10.1109\/TIP.2022.3183434 Jiayi Ji, Yiwei Ma, Xiaoshuai Sun, Yiyi Zhou, Yongjian Wu, and Rongrong Ji. 2022. Knowing What to Learn: A Metric-oriented Focal Mechanism for Image Captioning. IEEE Transactions on Image Processing (2022), 1--1. https:\/\/doi.org\/ 10.1109\/TIP.2022.3183434"},{"key":"e_1_3_2_2_23_1","volume-title":"International Conference on Machine Learning. PMLR, 4904--4916","author":"Jia Chao","year":"2021","unstructured":"Chao Jia , Yinfei Yang , Ye Xia , Yi-Ting Chen , Zarana Parekh , Hieu Pham , Quoc Le , Yun-Hsuan Sung , Zhen Li , and Tom Duerig . 2021 . Scaling up visual and vision- language representation learning with noisy text supervision . In International Conference on Machine Learning. PMLR, 4904--4916 . Chao Jia, Yinfei Yang, Ye Xia, Yi-Ting Chen, Zarana Parekh, Hieu Pham, Quoc Le, Yun-Hsuan Sung, Zhen Li, and Tom Duerig. 2021. Scaling up visual and vision- language representation learning with noisy text supervision. In International Conference on Machine Learning. PMLR, 4904--4916."},{"key":"e_1_3_2_2_24_1","volume-title":"Hierarchical Cross-Modal Graph Consistency Learning for Video- Text Retrieval. In SIGIR '21: The 44th International ACM SIGIR Conference on Research and Development in Information Retrieval","author":"Jin Weike","year":"2021","unstructured":"Weike Jin , Zhou Zhao , Pengcheng Zhang , Jieming Zhu , Xiuqiang He , and Yueting Zhuang . 2021 . Hierarchical Cross-Modal Graph Consistency Learning for Video- Text Retrieval. In SIGIR '21: The 44th International ACM SIGIR Conference on Research and Development in Information Retrieval , Virtual Event, Canada, July 11--15 , 2021, Fernando Diaz, Chirag Shah, Torsten Suel, Pablo Castells, Rosie Jones, and Tetsuya Sakai (Eds.). ACM, 1114--1124. https:\/\/doi.org\/10.1145\/3404835. 3462974 Weike Jin, Zhou Zhao, Pengcheng Zhang, Jieming Zhu, Xiuqiang He, and Yueting Zhuang. 2021. Hierarchical Cross-Modal Graph Consistency Learning for Video- Text Retrieval. In SIGIR '21: The 44th International ACM SIGIR Conference on Research and Development in Information Retrieval, Virtual Event, Canada, July 11--15, 2021, Fernando Diaz, Chirag Shah, Torsten Suel, Pablo Castells, Rosie Jones, and Tetsuya Sakai (Eds.). ACM, 1114--1124. https:\/\/doi.org\/10.1145\/3404835. 3462974"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00405"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3397271.3401075"},{"key":"e_1_3_2_2_27_1","volume-title":"ICLR 2015","author":"Kingma Diederik P","year":"2015","unstructured":"Diederik P Kingma and Jimmy Ba . 2015 . Adam: A method for stochastic opti- mization . ICLR 2015 . arXiv preprint arXiv:1412.6980 9 (2015). Diederik P Kingma and Jimmy Ba. 2015. Adam: A method for stochastic opti- mization. ICLR 2015. arXiv preprint arXiv:1412.6980 9 (2015)."},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00999"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01225-0_13"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00725"},{"key":"e_1_3_2_2_31_1","volume-title":"Align before fuse: Vision and language repre- sentation learning with momentum distillation. Advances in Neural Information Processing Systems 34","author":"Li Junnan","year":"2021","unstructured":"Junnan Li , Ramprasaath Selvaraju , Akhilesh Gotmare , Shafiq Joty , Caiming Xiong , and Steven Chu Hong Hoi . 2021. Align before fuse: Vision and language repre- sentation learning with momentum distillation. Advances in Neural Information Processing Systems 34 ( 2021 ). Junnan Li, Ramprasaath Selvaraju, Akhilesh Gotmare, Shafiq Joty, Caiming Xiong, and Steven Chu Hong Hoi. 2021. Align before fuse: Vision and language repre- sentation learning with momentum distillation. Advances in Neural Information Processing Systems 34 (2021)."},{"key":"e_1_3_2_2_32_1","volume-title":"Hero: Hierarchical encoder for video language omni-representation pre-training. arXiv preprint arXiv:2005.00200","author":"Li Linjie","year":"2020","unstructured":"Linjie Li , Yen-Chun Chen , Yu Cheng , Zhe Gan , Licheng Yu , and Jingjing Liu . 2020 . Hero: Hierarchical encoder for video language omni-representation pre-training. arXiv preprint arXiv:2005.00200 (2020). Linjie Li, Yen-Chun Chen, Yu Cheng, Zhe Gan, Licheng Yu, and Jingjing Liu. 2020. Hero: Hierarchical encoder for video language omni-representation pre-training. arXiv preprint arXiv:2005.00200 (2020)."},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58577-8_8"},{"key":"e_1_3_2_2_34_1","volume-title":"Hit: Hierarchical transformer with momentum contrast for video- text retrieval. arXiv preprint arXiv:2103.15049","author":"Liu Song","year":"2021","unstructured":"Song Liu , Haoqi Fan , Shengsheng Qian , Yiru Chen , Wenkui Ding , and Zhongyuan Wang . 2021 . Hit: Hierarchical transformer with momentum contrast for video- text retrieval. arXiv preprint arXiv:2103.15049 (2021). Song Liu, Haoqi Fan, Shengsheng Qian, Yiru Chen, Wenkui Ding, and Zhongyuan Wang. 2021. Hit: Hierarchical transformer with momentum contrast for video- text retrieval. arXiv preprint arXiv:2103.15049 (2021)."},{"key":"e_1_3_2_2_35_1","volume-title":"Use what you have: Video retrieval using representations from collaborative experts. arXiv preprint arXiv:1907.13487","author":"Liu Yang","year":"2019","unstructured":"Yang Liu , Samuel Albanie , Arsha Nagrani , and Andrew Zisserman . 2019. Use what you have: Video retrieval using representations from collaborative experts. arXiv preprint arXiv:1907.13487 ( 2019 ). Yang Liu, Samuel Albanie, Arsha Nagrani, and Andrew Zisserman. 2019. Use what you have: Video retrieval using representations from collaborative experts. arXiv preprint arXiv:1907.13487 (2019)."},{"key":"e_1_3_2_2_36_1","volume-title":"Sgdr: Stochastic gradient descent with warm restarts. arXiv preprint arXiv:1608.03983","author":"Loshchilov Ilya","year":"2016","unstructured":"Ilya Loshchilov and Frank Hutter . 2016 . Sgdr: Stochastic gradient descent with warm restarts. arXiv preprint arXiv:1608.03983 (2016). Ilya Loshchilov and Frank Hutter. 2016. Sgdr: Stochastic gradient descent with warm restarts. arXiv preprint arXiv:1608.03983 (2016)."},{"key":"e_1_3_2_2_37_1","volume-title":"Vilbert: Pretrain- ing task-agnostic visiolinguistic representations for vision-and-language tasks. Advances in neural information processing systems 32","author":"Lu Jiasen","year":"2019","unstructured":"Jiasen Lu , Dhruv Batra , Devi Parikh , and Stefan Lee . 2019 . Vilbert: Pretrain- ing task-agnostic visiolinguistic representations for vision-and-language tasks. Advances in neural information processing systems 32 (2019). Jiasen Lu, Dhruv Batra, Devi Parikh, and Stefan Lee. 2019. Vilbert: Pretrain- ing task-agnostic visiolinguistic representations for vision-and-language tasks. Advances in neural information processing systems 32 (2019)."},{"key":"e_1_3_2_2_38_1","volume-title":"Clip4clip: An empirical study of clip for end to end video clip retrieval. arXiv preprint arXiv:2104.08860","author":"Luo Huaishao","year":"2021","unstructured":"Huaishao Luo , Lei Ji , Ming Zhong , Yang Chen , Wen Lei , Nan Duan , and Tianrui Li. 2021. Clip4clip: An empirical study of clip for end to end video clip retrieval. arXiv preprint arXiv:2104.08860 ( 2021 ). Huaishao Luo, Lei Ji, Ming Zhong, Yang Chen, Wen Lei, Nan Duan, and Tianrui Li. 2021. Clip4clip: An empirical study of clip for end to end video clip retrieval. arXiv preprint arXiv:2104.08860 (2021)."},{"key":"e_1_3_2_2_39_1","volume-title":"Knowing what it is: Semantic-enhanced Dual Attention Transformer","author":"Ma Yiwei","year":"2022","unstructured":"Yiwei Ma , Jiayi Ji , Xiaoshuai Sun , Yiyi Zhou , Yongjian Wu , Feiyue Huang , and Rongrong Ji. 2022. Knowing what it is: Semantic-enhanced Dual Attention Transformer . IEEE Transactions on Multimedia ( 2022 ), 1--1. https:\/\/doi.org\/10. 1109\/TMM.2022.3164787 Yiwei Ma, Jiayi Ji, Xiaoshuai Sun, Yiyi Zhou, Yongjian Wu, Feiyue Huang, and Rongrong Ji. 2022. Knowing what it is: Semantic-enhanced Dual Attention Transformer. IEEE Transactions on Multimedia (2022), 1--1. https:\/\/doi.org\/10. 1109\/TMM.2022.3164787"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00990"},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3206025.3206064"},{"key":"e_1_3_2_2_42_1","volume-title":"Joao Henriques, and Andrea Vedaldi.","author":"Patrick Mandela","year":"2020","unstructured":"Mandela Patrick , Po-Yao Huang , Yuki Asano , Florian Metze , Alexander Haupt- mann , Joao Henriques, and Andrea Vedaldi. 2020 . Support-set bottlenecks for video-text representation learning. arXiv preprint arXiv:2010.02824 (2020). Mandela Patrick, Po-Yao Huang, Yuki Asano, Florian Metze, Alexander Haupt- mann, Joao Henriques, and Andrea Vedaldi. 2020. Support-set bottlenecks for video-text representation learning. arXiv preprint arXiv:2010.02824 (2020)."},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-77004-4_1"},{"key":"e_1_3_2_2_44_1","volume-title":"Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al.","author":"Radford Alec","year":"2021","unstructured":"Alec Radford , Jong Wook Kim , Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al. 2021 . Learning transferable visual models from natural language supervision. arXiv preprint arXiv:2103.00020 (2021). Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al. 2021. Learning transferable visual models from natural language supervision. arXiv preprint arXiv:2103.00020 (2021)."},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-24947-6_17"},{"key":"e_1_3_2_2_46_1","volume-title":"et al","author":"Rouditchenko Andrew","year":"2020","unstructured":"Andrew Rouditchenko , Angie Boggust , David Harwath , Brian Chen , Dhiraj Joshi , Samuel Thomas , Kartik Audhkhasi , Hilde Kuehne , Rameswar Panda , Rogerio Feris , et al . 2020 . Avlnet : Learning audio-visual language representations from instructional videos. arXiv preprint arXiv:2006.09199 (2020). Andrew Rouditchenko, Angie Boggust, David Harwath, Brian Chen, Dhiraj Joshi, Samuel Thomas, Kartik Audhkhasi, Hilde Kuehne, Rameswar Panda, Rogerio Feris, et al . 2020. Avlnet: Learning audio-visual language representations from instructional videos. arXiv preprint arXiv:2006.09199 (2020)."},{"key":"e_1_3_2_2_47_1","volume-title":"Colbertv2: Effective and efficient retrieval via lightweight late interaction. arXiv preprint arXiv:2112.01488","author":"Santhanam Keshav","year":"2021","unstructured":"Keshav Santhanam , Omar Khattab , Jon Saad-Falcon , Christopher Potts , and Matei Zaharia . 2021. Colbertv2: Effective and efficient retrieval via lightweight late interaction. arXiv preprint arXiv:2112.01488 ( 2021 ). Keshav Santhanam, Omar Khattab, Jon Saad-Falcon, Christopher Potts, and Matei Zaharia. 2021. Colbertv2: Effective and efficient retrieval via lightweight late interaction. arXiv preprint arXiv:2112.01488 (2021)."},{"key":"e_1_3_2_2_48_1","volume-title":"Neural machine translation of rare words with subword units. arXiv preprint arXiv:1508.07909","author":"Sennrich Rico","year":"2015","unstructured":"Rico Sennrich , Barry Haddow , and Alexandra Birch . 2015. Neural machine translation of rare words with subword units. arXiv preprint arXiv:1508.07909 ( 2015 ). Rico Sennrich, Barry Haddow, and Alexandra Birch. 2015. Neural machine translation of rare words with subword units. arXiv preprint arXiv:1508.07909 (2015)."},{"key":"e_1_3_2_2_49_1","volume-title":"Learning video representations using contrastive bidirectional transformer. arXiv preprint arXiv:1906.05743","author":"Sun Chen","year":"2019","unstructured":"Chen Sun , Fabien Baradel , Kevin Murphy , and Cordelia Schmid . 2019. Learning video representations using contrastive bidirectional transformer. arXiv preprint arXiv:1906.05743 ( 2019 ). Chen Sun, Fabien Baradel, Kevin Murphy, and Cordelia Schmid. 2019. Learning video representations using contrastive bidirectional transformer. arXiv preprint arXiv:1906.05743 (2019)."},{"key":"e_1_3_2_2_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00756"},{"key":"e_1_3_2_2_51_1","volume-title":"Lxmert: Learning cross-modality encoder representations from transformers. arXiv preprint arXiv:1908.07490","author":"Tan Hao","year":"2019","unstructured":"Hao Tan and Mohit Bansal . 2019 . Lxmert: Learning cross-modality encoder representations from transformers. arXiv preprint arXiv:1908.07490 (2019). Hao Tan and Mohit Bansal. 2019. Lxmert: Learning cross-modality encoder representations from transformers. arXiv preprint arXiv:1908.07490 (2019)."},{"key":"e_1_3_2_2_52_1","volume-title":"Attention is all you need. Advances in neural information processing systems 30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani , Noam Shazeer , Niki Parmar , Jakob Uszkoreit , Llion Jones , Aidan N Gomez , Lukasz Kaiser , and Illia Polosukhin . 2017. Attention is all you need. Advances in neural information processing systems 30 ( 2017 ). Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, Lukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_2_2_53_1","volume-title":"Translating videos to natural language using deep recurrent neural networks. arXiv preprint arXiv:1412.4729","author":"Venugopalan Subhashini","year":"2014","unstructured":"Subhashini Venugopalan , Huijuan Xu , Jeff Donahue , Marcus Rohrbach , Raymond Mooney , and Kate Saenko . 2014. Translating videos to natural language using deep recurrent neural networks. arXiv preprint arXiv:1412.4729 ( 2014 ). Subhashini Venugopalan, Huijuan Xu, Jeff Donahue, Marcus Rohrbach, Raymond Mooney, and Kate Saenko. 2014. Translating videos to natural language using deep recurrent neural networks. arXiv preprint arXiv:1412.4729 (2014)."},{"key":"e_1_3_2_2_54_1","volume-title":"T2VLAD: Global-Local Sequence Alignment for Text-Video Retrieval","author":"Wang Xiaohan","year":"2021","unstructured":"Xiaohan Wang , Linchao Zhu , and Yi Yang . 2021. T2VLAD: Global-Local Sequence Alignment for Text-Video Retrieval . In IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2021 , virtual, June 19--25, 2021. Computer Vision Foundation \/ IEEE, 5079--5088. https:\/\/openaccess.thecvf.com\/content\/ CVPR2021\/html\/Wang_T2VLAD_Global-Local_Sequence_Alignment_for_ Text-Video _Retrieval_CVPR_2021_paper.html Xiaohan Wang, Linchao Zhu, and Yi Yang. 2021. T2VLAD: Global-Local Sequence Alignment for Text-Video Retrieval. In IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2021, virtual, June 19--25, 2021. Computer Vision Foundation \/ IEEE, 5079--5088. https:\/\/openaccess.thecvf.com\/content\/ CVPR2021\/html\/Wang_T2VLAD_Global-Local_Sequence_Alignment_for_ Text-Video_Retrieval_CVPR_2021_paper.html"},{"key":"e_1_3_2_2_55_1","volume-title":"Proceedings of the 59th Annual Meeting of the Asso- ciation for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers)","author":"Xu Haiyang","year":"1865","unstructured":"Haiyang Xu , Ming Yan , Chenliang Li , Bin Bi , Songfang Huang , Wenming Xiao , and Fei Huang . 2021. E2E-VLP: End-to-End Vision-Language Pre-training En- hanced by Visual Learning . In Proceedings of the 59th Annual Meeting of the Asso- ciation for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers) . Association for Computa- tional Linguistics, Online, 503--513. https:\/\/doi.org\/10. 1865 3\/v1\/2021.acl-long.42 Haiyang Xu, Ming Yan, Chenliang Li, Bin Bi, Songfang Huang, Wenming Xiao, and Fei Huang. 2021. E2E-VLP: End-to-End Vision-Language Pre-training En- hanced by Visual Learning. In Proceedings of the 59th Annual Meeting of the Asso- ciation for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers). Association for Computa- tional Linguistics, Online, 503--513. https:\/\/doi.org\/10.18653\/v1\/2021.acl-long.42"},{"key":"e_1_3_2_2_56_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.571"},{"key":"e_1_3_2_2_57_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01136"},{"key":"e_1_3_2_2_58_1","doi-asserted-by":"publisher","DOI":"10.1145\/3397271.3401151"},{"key":"e_1_3_2_2_59_1","volume-title":"Zhenguo Li, Xin Jiang, and Chunjing Xu.","author":"Yao Lewei","year":"2021","unstructured":"Lewei Yao , Runhui Huang , Lu Hou , Guansong Lu , Minzhe Niu , Hang Xu , Xiao- dan Liang , Zhenguo Li, Xin Jiang, and Chunjing Xu. 2021 . FILIP : Fine-grained Interactive Language-Image Pre-Training . arXiv preprint arXiv:2111.07783 (2021). Lewei Yao, Runhui Huang, Lu Hou, Guansong Lu, Minzhe Niu, Hang Xu, Xiao- dan Liang, Zhenguo Li, Xin Jiang, and Chunjing Xu. 2021. FILIP: Fine-grained Interactive Language-Image Pre-Training. arXiv preprint arXiv:2111.07783 (2021)."},{"key":"e_1_3_2_2_60_1","volume-title":"Ernie-vil: Knowledge enhanced vision-language representations through scene graph. arXiv preprint arXiv:2006.16934","author":"Yu Fei","year":"2020","unstructured":"Fei Yu , Jiji Tang , Weichong Yin , Yu Sun , Hao Tian , Hua Wu , and Haifeng Wang . 2020 . Ernie-vil: Knowledge enhanced vision-language representations through scene graph. arXiv preprint arXiv:2006.16934 (2020). Fei Yu, Jiji Tang, Weichong Yin, Yu Sun, Hao Tian, Hua Wu, and Haifeng Wang. 2020. Ernie-vil: Knowledge enhanced vision-language representations through scene graph. arXiv preprint arXiv:2006.16934 (2020)."},{"key":"e_1_3_2_2_61_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01234-2_29"},{"key":"e_1_3_2_2_62_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.347"},{"key":"e_1_3_2_2_63_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01261-8_23"},{"key":"e_1_3_2_2_64_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01521"},{"key":"e_1_3_2_2_65_1","volume-title":"SeqTR: A Simple yet Universal Network for Visual Grounding. arXiv preprint arXiv:2203.16265","author":"Zhu Chaoyang","year":"2022","unstructured":"Chaoyang Zhu , Yiyi Zhou , Yunhang Shen , Gen Luo , Xingjia Pan , Mingbao Lin , Chao Chen , Liujuan Cao , Xiaoshuai Sun , and Rongrong Ji. 2022. SeqTR: A Simple yet Universal Network for Visual Grounding. arXiv preprint arXiv:2203.16265 ( 2022 ). Chaoyang Zhu, Yiyi Zhou, Yunhang Shen, Gen Luo, Xingjia Pan, Mingbao Lin, Chao Chen, Liujuan Cao, Xiaoshuai Sun, and Rongrong Ji. 2022. SeqTR: A Simple yet Universal Network for Visual Grounding. arXiv preprint arXiv:2203.16265 (2022)."},{"key":"e_1_3_2_2_66_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00877"}],"event":{"name":"MM '22: The 30th ACM International Conference on Multimedia","location":"Lisboa Portugal","acronym":"MM '22","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 30th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3503161.3547910","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3503161.3547910","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T19:00:30Z","timestamp":1750186830000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3503161.3547910"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,10,10]]},"references-count":66,"alternative-id":["10.1145\/3503161.3547910","10.1145\/3503161"],"URL":"https:\/\/doi.org\/10.1145\/3503161.3547910","relation":{},"subject":[],"published":{"date-parts":[[2022,10,10]]},"assertion":[{"value":"2022-10-10","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}