{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,21]],"date-time":"2026-06-21T10:48:33Z","timestamp":1782038913125,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":54,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,26]],"date-time":"2023-10-26T00:00:00Z","timestamp":1698278400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Shandong Provincial Natural Science and Foundation","award":["No.: ZR2020QF106"],"award-info":[{"award-number":["No.: ZR2020QF106"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["No.: 62176137, and No.: 62006140"],"award-info":[{"award-number":["No.: 62176137, and No.: 62006140"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,26]]},"DOI":"10.1145\/3581783.3612152","type":"proceedings-article","created":{"date-parts":[[2023,10,27]],"date-time":"2023-10-27T07:26:54Z","timestamp":1698391614000},"page":"557-566","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":13,"title":["RTQ: Rethinking Video-language Understanding Based on Image-text Model"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-4879-2169","authenticated-orcid":false,"given":"Xiao","family":"Wang","sequence":"first","affiliation":[{"name":"Harbin Institute of Technology, Shenzhen &amp; JD.com Inc., Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7362-7897","authenticated-orcid":false,"given":"Yaoyu","family":"Li","sequence":"additional","affiliation":[{"name":"JD.com Inc., Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3197-5698","authenticated-orcid":false,"given":"Tian","family":"Gan","sequence":"additional","affiliation":[{"name":"Shandong University, Qingdao, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-6391-4814","authenticated-orcid":false,"given":"Zheng","family":"Zhang","sequence":"additional","affiliation":[{"name":"JD.com Inc., Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-5518-7077","authenticated-orcid":false,"given":"Jingjing","family":"Lv","sequence":"additional","affiliation":[{"name":"JD.com Inc., Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1476-0273","authenticated-orcid":false,"given":"Liqiang","family":"Nie","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology, Shenzhen, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2023,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Jean-Baptiste Alayrac Jeff Donahue Pauline Luc Antoine Miech Iain Barr Yana Hasson Karel Lenc Arthur Mensch Katherine Millican Malcolm Reynolds Roman Ring Eliza Rutherford Serkan Cabi Tengda Han Zhitao Gong Sina Samangooei Marianne Monteiro Jacob L. Menick Sebastian Borgeaud Andy Brock Aida Nematzadeh Sahand Sharifzadeh Mikolaj Binkowski Ricardo Barreira Oriol Vinyals Andrew Zisserman and Kar\u00e9n Simonyan. 2022. Flamingo: a Visual Language Model for Few-Shot Learning. In Advances in Neural Information Processing Systems. Curran Associates 23716--23736."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.618"},{"key":"e_1_3_2_1_3_1","volume-title":"Frozen in Time: A Joint Video and Image Encoder for End-to-End Retrieval. In Conference on Computer Vision and Pattern Recognition. IEEE, 1708--1718","author":"Bain Max","year":"2021","unstructured":"Max Bain, Arsha Nagrani, G\u00fcl Varol, and Andrew Zisserman. 2021. Frozen in Time: A Joint Video and Image Encoder for End-to-End Retrieval. In Conference on Computer Vision and Pattern Recognition. IEEE, 1708--1718."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00293"},{"key":"e_1_3_2_1_5_1","volume-title":"Annual Meeting of the Association for Computational Linguistics. The Association for Computer Linguistics, 190--200","author":"Chen David","year":"2011","unstructured":"David Chen and William B Dolan. 2011. Collecting highly parallel data for paraphrase evaluation. In Annual Meeting of the Association for Computational Linguistics. The Association for Computer Linguistics, 190--200."},{"key":"e_1_3_2_1_6_1","unstructured":"Xing Cheng Hezheng Lin Xiangyu Wu Fan Yang and Dong Shen. 2021. Improving Video-Text Retrieval by Multi-Stream Corpus Alignment and Dual Softmax Loss. showeprint[arXiv]2109.04290"},{"key":"e_1_3_2_1_7_1","volume-title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In Conference of the North American","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In Conference of the North American Chapter of the Association for Computational Linguistics. Association for Computational Linguistics, 4171--4186."},{"key":"e_1_3_2_1_8_1","volume-title":"Stacked Hybrid-Attention and Group Collaborative Learning for Unbiased Scene Graph Generation. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition. IEEE","author":"Dong Xingning","year":"2022","unstructured":"Xingning Dong, Tian Gan, Xuemeng Song, Jianlong Wu, Yuan Cheng, and Liqiang Nie. 2022. Stacked Hybrid-Attention and Group Collaborative Learning for Unbiased Scene Graph Generation. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition. IEEE, 19405--19414."},{"key":"e_1_3_2_1_9_1","volume-title":"International Conference on Learning Representations. OpenReview.net.","author":"Dosovitskiy Alexey","year":"2021","unstructured":"Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, Xiaohua Zhai, Thomas Unterthiner, Mostafa Dehghani, Matthias Minderer, Georg Heigold, Sylvain Gelly, Jakob Uszkoreit, and Neil Houlsby. 2021. An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. In International Conference on Learning Representations. OpenReview.net."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3539597.3570405"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3300941"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01423"},{"key":"e_1_3_2_1_13_1","volume-title":"Bridging Video-Text Retrieval With Multiple Choice Questions. In Conference on Computer Vision and Pattern Recognition. IEEE, 16167--16176","author":"Ge Yuying","year":"2022","unstructured":"Yuying Ge, Yixiao Ge, Xihui Liu, Dian Li, Ying Shan, Xiaohu Qie, and Ping Luo. 2022. Bridging Video-Text Retrieval With Multiple Choice Questions. In Conference on Computer Vision and Pattern Recognition. IEEE, 16167--16176."},{"key":"e_1_3_2_1_14_1","volume-title":"X-Pool: Cross-Modal Language-Video Attention for Text-Video Retrieval. In Conference on Computer Vision and Pattern Recognition. IEEE, 4996--5005","author":"Gorti Satya Krishna","year":"2022","unstructured":"Satya Krishna Gorti, No\u00ebl Vouitsis, Junwei Ma, Keyvan Golestan, Maksims Volkovs, Animesh Garg, and Guangwei Yu. 2022. X-Pool: Cross-Modal Language-Video Attention for Text-Video Retrieval. In Conference on Computer Vision and Pattern Recognition. IEEE, 4996--5005."},{"key":"e_1_3_2_1_15_1","volume-title":"Text with Knowledge Graph Augmented Transformer for Video Captioning. In Conference on Computer Vision and Pattern Recognition. IEEE, 1175--1175","author":"Gu Xin","year":"2023","unstructured":"Xin Gu, Guang Chen, Yufei Wang, Libo Zhang, Tiejian Luo, and Longyin Wen. 2023. Text with Knowledge Graph Augmented Transformer for Video Captioning. In Conference on Computer Vision and Pattern Recognition. IEEE, 1175--1175."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.83"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00725"},{"key":"e_1_3_2_1_18_1","volume-title":"BLIP: Bootstrapping Language-Image Pre-training for Unified Vision-Language Understanding and Generation. In International Conference on Machine Learning. PMLR, 12888--12900","author":"Li Junnan","unstructured":"Junnan Li, Dongxu Li, Caiming Xiong, and Steven C. H. Hoi. 2022a. BLIP: Bootstrapping Language-Image Pre-training for Unified Vision-Language Understanding and Generation. In International Conference on Machine Learning. PMLR, 12888--12900."},{"key":"e_1_3_2_1_19_1","unstructured":"Junnan Li Ramprasaath R. Selvaraju Akhilesh Gotmare Shafiq R. Joty Caiming Xiong and Steven Chu-Hong Hoi. 2021. Align before Fuse: Vision and Language Representation Learning with Momentum Distillation. In Advances in Neural Information Processing Systems. Curran Associates 9694--9705."},{"key":"e_1_3_2_1_20_1","volume-title":"UniFormer: Unified Transformer for Efficient Spatial-Temporal Representation Learning. In International Conference on Learning Representations. OpenReview.net.","author":"Li Kunchang","year":"2022","unstructured":"Kunchang Li, Yali Wang, Peng Gao, Guanglu Song, Yu Liu, Hongsheng Li, and Yu Qiao. 2022b. UniFormer: Unified Transformer for Efficient Spatial-Temporal Representation Learning. In International Conference on Learning Representations. OpenReview.net."},{"key":"e_1_3_2_1_21_1","volume-title":"Invariant Grounding for Video Question Answering. In Conference on Computer Vision and Pattern Recognition. IEEE, 2918--2927","author":"Li Yicong","year":"2022","unstructured":"Yicong Li, Xiang Wang, Junbin Xiao, Wei Ji, and Tat-Seng Chua. 2022c. Invariant Grounding for Video Question Answering. In Conference on Computer Vision and Pattern Recognition. IEEE, 2918--2927."},{"key":"e_1_3_2_1_22_1","volume-title":"TSM: Temporal Shift Module for Efficient Video Understanding. In International Conference on Computer Vision. IEEE, 7082--7092","author":"Lin Ji","year":"2019","unstructured":"Ji Lin, Chuang Gan, and Song Han. 2019. TSM: Temporal Shift Module for Efficient Video Understanding. In International Conference on Computer Vision. IEEE, 7082--7092."},{"key":"e_1_3_2_1_23_1","volume-title":"SwinBERT: End-to-End Transformers with Sparse Attention for Video Captioning. In Conference on Computer Vision and Pattern Recognition. IEEE, 17928--17937","author":"Lin Kevin","year":"2022","unstructured":"Kevin Lin, Linjie Li, Chung-Ching Lin, Faisal Ahmed, Zhe Gan, Zicheng Liu, Yumao Lu, and Lijuan Wang. 2022. SwinBERT: End-to-End Transformers with Sparse Attention for Video Captioning. In Conference on Computer Vision and Pattern Recognition. IEEE, 17928--17937."},{"key":"e_1_3_2_1_24_1","volume-title":"Revisiting Temporal Modeling for CLIP-based Image-to-Video Knowledge Transferring. In Conference on Computer Vision and Pattern Recognition. IEEE, 6422--6431","author":"Liu Ruyang","unstructured":"Ruyang Liu, Jingjia Huang, Ge Li, Jiashi Feng, Xinglong Wu, and Thomas H. Li. 2023. Revisiting Temporal Modeling for CLIP-based Image-to-Video Knowledge Transferring. In Conference on Computer Vision and Pattern Recognition. IEEE, 6422--6431."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19781-9_19"},{"key":"e_1_3_2_1_26_1","unstructured":"Huaishao Luo Lei Ji Botian Shi Haoyang Huang Nan Duan Tianrui Li Jason Li Taroon Bharti and Ming Zhou. 2020. UniVL: A Uni?ed Video and Language Pre-Training Model for Multimodal Understanding and Generation. showeprint[arXiv]2002.06353"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2022.07.028"},{"key":"e_1_3_2_1_28_1","volume-title":"Expanding Language-Image Pretrained Models for General Video Recognition. In European Conference on Computer Vision","volume":"13664","author":"Ni Bolin","year":"2022","unstructured":"Bolin Ni, Houwen Peng, Minghao Chen, Songyang Zhang, Gaofeng Meng, Jianlong Fu, Shiming Xiang, and Haibin Ling. 2022. Expanding Language-Image Pretrained Models for General Video Recognition. In European Conference on Computer Vision, Vol. 13664. Springer, 1--18."},{"key":"e_1_3_2_1_29_1","volume-title":"Search-oriented Micro-video Captioning. In International Conference on Multimedia. ACM, 3234--3243","author":"Nie Liqiang","year":"2022","unstructured":"Liqiang Nie, Leigang Qu, Dai Meng, Min Zhang, Qi Tian, and Alberto Del Bimbo. 2022. Search-oriented Micro-video Captioning. In International Conference on Multimedia. ACM, 3234--3243."},{"key":"e_1_3_2_1_30_1","volume-title":"Dynamic Modality Interaction Modeling for Image-Text Retrieval. In SIGIR Conference on Research and Development in Information Retrieval. ACM, 1104--1113","author":"Qu Leigang","year":"2021","unstructured":"Leigang Qu, Meng Liu, Jianlong Wu, Zan Gao, and Liqiang Nie. 2021. Dynamic Modality Interaction Modeling for Image-Text Retrieval. In SIGIR Conference on Research and Development in Information Retrieval. ACM, 1104--1113."},{"key":"e_1_3_2_1_31_1","volume-title":"Learning Transferable Visual Models From Natural Language Supervision. In International Conference on Machine Learning. PMLR, 8748--8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. In International Conference on Machine Learning. PMLR, 8748--8763."},{"key":"e_1_3_2_1_32_1","volume-title":"End-to-end Generative Pretraining for Multimodal Video Captioning. In Conference on Computer Vision and Pattern Recognition. IEEE, 17938--17947","author":"Seo Paul Hongsuck","year":"2022","unstructured":"Paul Hongsuck Seo, Arsha Nagrani, Anurag Arnab, and Cordelia Schmid. 2022. End-to-end Generative Pretraining for Multimodal Video Captioning. In Conference on Computer Vision and Pattern Recognition. IEEE, 17938--17947."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3479207"},{"key":"e_1_3_2_1_34_1","unstructured":"Junke Wang Dongdong Chen Zuxuan Wu Chong Luo Luowei Zhou Yucheng Zhao Yujia Xie Ce Liu Yu-Gang Jiang and Lu Yuan. 2022a. OmniVL:One Foundation Model for Image-Language and Video-Language Tasks. In Advances in Neural Information Processing Systems. Curran Associates."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2021.3087038"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548098"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2019.2923608"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2023.3265261"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01031"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00965"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i3.20184"},{"key":"e_1_3_2_1_42_1","unstructured":"Junbin Xiao Pan Zhou Angela Yao Yicong Li Richang Hong Shuicheng Yan and Tat-Seng Chua. 2023. Contrastive Video Question Answering via Video Graph Transformer. showeprint[arXiv]2302.13668"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123427"},{"key":"e_1_3_2_1_44_1","unstructured":"Haiyang Xu Qinghao Ye Ming Yan Yaya Shi Jiabo Ye Yuanhong Xu Chenliang Li Bin Bi Qi Qian Wei Wang Guohai Xu Ji Zhang Songfang Huang Fei Huang and Jingren Zhou. 2023. mPLUG-2: A Modularized Multi-modal Foundation Model Across Text Image and Video. showeprint[arXiv]2302.00402"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.571"},{"key":"e_1_3_2_1_46_1","volume-title":"International Conference on Learning Representations. OpenReview.net.","author":"Xue Hongwei","year":"2023","unstructured":"Hongwei Xue, Yuchong Sun, Bei Liu, Jianlong Fu, Ruihua Song, Houqiang Li, and Jiebo Luo. 2023. CLIP-ViP: Adapting Pre-trained Image-Text Model to Video-Language Representation Alignment. In International Conference on Learning Representations. OpenReview.net."},{"key":"e_1_3_2_1_47_1","unstructured":"Shen Yan Tao Zhu Zirui Wang Yuan Cao Mi Zhang Soham Ghosh Yonghui Wu and Jiahui Yu. 2023. VideoCoCa: Video-Text Modeling with Zero-Shot Transfer from Contrastive Captioners. [arXiv]2212.04979"},{"key":"e_1_3_2_1_48_1","volume-title":"CLIP Meets Video Captioning: Concept-Aware Representation Learning Does Matter. In Chinese Conference of Pattern Recognition and Computer Vision. Springer, 368--381","author":"Yang Bang","year":"2022","unstructured":"Bang Yang, Tong Zhang, and Yuexian Zou. 2022. CLIP Meets Video Captioning: Concept-Aware Representation Learning Does Matter. In Chinese Conference of Pattern Recognition and Computer Vision. Springer, 368--381."},{"key":"e_1_3_2_1_49_1","volume-title":"Hierarchical Modular Network for Video Captioning. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2022","author":"Ye Hanhua","year":"2022","unstructured":"Hanhua Ye, Guorong Li, Yuankai Qi, Shuhui Wang, Qingming Huang, and Ming-Hsuan Yang. 2022a. Hierarchical Modular Network for Video Captioning. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2022, New Orleans, LA, USA, June 18-24, 2022. IEEE, 17918--17927."},{"key":"e_1_3_2_1_50_1","unstructured":"Qinghao Ye Guohai Xu Ming Yan Haiyang Xu Qi Qian Ji Zhang and Fei Huang. 2022b. HiTeA: Hierarchical Temporal-Aware Video-Language Pre-training. showeprint[arXiv]2212.14546"},{"key":"e_1_3_2_1_51_1","volume-title":"Jize Cao, Ali Farhadi, and Yejin Choi.","author":"Zellers Rowan","year":"2021","unstructured":"Rowan Zellers, Ximing Lu, Jack Hessel, Youngjae Yu, Jae Sung Park, Jize Cao, Ali Farhadi, and Yejin Choi. 2021. MERLOT: Multimodal Neural Script Knowledge Models. In Advances in Neural Information Processing Systems. Curran Associates, 23634--23651."},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1145\/3477495.3531950"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00909"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"crossref","unstructured":"Weihong Zhong Mao Zheng Duyu Tang Xuan Luo Heng Gong Xiaocheng Feng and Bing Qin. 2023. STOA-VLP: Spatial-Temporal Modeling of Object and Action for Video-Language Pre-training. [arXiv]2302.09736","DOI":"10.1609\/aaai.v37i3.25483"}],"event":{"name":"MM '23: The 31st ACM International Conference on Multimedia","location":"Ottawa ON Canada","acronym":"MM '23","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 31st ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612152","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3581783.3612152","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T00:04:22Z","timestamp":1755821062000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612152"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,26]]},"references-count":54,"alternative-id":["10.1145\/3581783.3612152","10.1145\/3581783"],"URL":"https:\/\/doi.org\/10.1145\/3581783.3612152","relation":{},"subject":[],"published":{"date-parts":[[2023,10,26]]},"assertion":[{"value":"2023-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}