{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:50:12Z","timestamp":1765309812012,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":35,"publisher":"ACM","funder":[{"name":"Institute of Information Engineering","award":["E5Z0051106"],"award-info":[{"award-number":["E5Z0051106"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755275","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T07:26:38Z","timestamp":1761377198000},"page":"4001-4009","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["T2VParser: Adaptive Decomposition Tokens for Partial Alignment in Text to Video Retrieval"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-4037-819X","authenticated-orcid":false,"given":"Yili","family":"Li","sequence":"first","affiliation":[{"name":"Institute of Information Engineering, Chinese Academy of Sciences, Beijing, China and School of Cyber Security, University of Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3190-6521","authenticated-orcid":false,"given":"Gang","family":"Xiong","sequence":"additional","affiliation":[{"name":"Institute of Information Engineering, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3533-4874","authenticated-orcid":false,"given":"Gaopeng","family":"Gou","sequence":"additional","affiliation":[{"name":"Institute of Information Engineering, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3658-8099","authenticated-orcid":false,"given":"Xiangyan","family":"Qu","sequence":"additional","affiliation":[{"name":"Institute of Information Engineering, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1598-0764","authenticated-orcid":false,"given":"Jiamin","family":"Zhuang","sequence":"additional","affiliation":[{"name":"Institute of Information Engineering, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3892-4909","authenticated-orcid":false,"given":"Zhen","family":"Li","sequence":"additional","affiliation":[{"name":"Institute of Information Engineering, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4653-1686","authenticated-orcid":false,"given":"Junzheng","family":"Shi","sequence":"additional","affiliation":[{"name":"Institute of Information Engineering, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.618"},{"key":"e_1_3_2_1_2_1","first-page":"1708","volume-title":"Frozen in Time: A Joint Video and Image Encoder for End-to-End Retrieval. In IEEE\/CVF International Conference on Computer Vision","author":"Bain Max","year":"2021","unstructured":"Max Bain, Arsha Nagrani, G\u00fcl Varol, and Andrew Zisserman. 2021. Frozen in Time: A Joint Video and Image Encoder for End-to-End Retrieval. In IEEE\/CVF International Conference on Computer Vision, 2021. 1708-1718."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.5555\/2002472.2002497"},{"key":"e_1_3_2_1_5_1","first-page":"10635","volume-title":"Fine-Grained Video-Text Retrieval With Hierarchical Graph Reasoning. In 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Chen Shizhe","year":"2020","unstructured":"Shizhe Chen, Yida Zhao, Qin Jin, and Qi Wu. 2020. Fine-Grained Video-Text Retrieval With Hierarchical Graph Reasoning. In 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2020. 10635-10644."},{"key":"e_1_3_2_1_6_1","volume-title":"Improving Video-Text Retrieval by Multi-Stream Corpus Alignment and Dual Softmax Loss. arXiv preprint arXiv:2109.04290","author":"Cheng Xing","year":"2021","unstructured":"Xing Cheng, Hezheng Lin, Xiangyu Wu, Fan Yang, and Dong Shen. 2021. Improving Video-Text Retrieval by Multi-Stream Corpus Alignment and Dual Softmax Loss. arXiv preprint arXiv:2109.04290 (2021)."},{"key":"e_1_3_2_1_7_1","volume-title":"A Strong, Economical, and Efficient Mixture-of-Experts Language Model. arXiv preprint arXiv:2405.04434","author":"Aixin Liu AI","year":"2024","unstructured":"DeepSeek-AI, Aixin Liu, Bei Feng, Bin Wang, Bingxuan Wang, Bo Liu, Chenggang Zhao, Chengqi Deng, Chong Ruan, Damai Dai, Daya Guo, Dejian Yang, Deli Chen, Dongjie Ji, Erhang Li, Fangyun Lin, Fuli Luo, Guangbo Hao, Guanting Chen, Guowei Li, Hao Zhang, Hanwei Xu, Hao Yang, Haowei Zhang, Honghui Ding, Huajian Xin, Huazuo Gao, Hui Li, Hui Qu, J. L. Cai, Jian Liang, Jianzhong Guo, Jiaqi Ni, Jiashi Li, Jin Chen, Jingyang Yuan, Junjie Qiu, Junxiao Song, Kai Dong, Kaige Gao, Kang Guan, Lean Wang, Lecong Zhang, Lei Xu, Leyi Xia, Liang Zhao, Liyue Zhang, Meng Li, Miaojun Wang, Mingchuan Zhang, Minghua Zhang, Minghui Tang, Mingming Li, Ning Tian, Panpan Huang, Peiyi Wang, Peng Zhang, Qihao Zhu, Qinyu Chen, Qiushi Du, R. J. Chen, R. L. Jin, Ruiqi Ge, Ruizhe Pan, Runxin Xu, Ruyi Chen, S. S. Li, Shanghao Lu, Shangyan Zhou, Shanhuang Chen, Shaoqing Wu, Shengfeng Ye, Shirong Ma, Shiyu Wang, Shuang Zhou, Shuiping Yu, Shunfeng Zhou, Size Zheng, Tao Wang, Tian Pei, Tian Yuan, Tianyu Sun, W. L. Xiao, Wangding Zeng, Wei An, Wen Liu, Wenfeng Liang, Wenjun Gao, Wentao Zhang, X. Q. Li, Xiangyue Jin, Xianzu Wang, Xiao Bi, Xiaodong Liu, Xiaohan Wang, Xiaojin Shen, Xiaokang Chen, Xiaosha Chen, Xiaotao Nie, and Xiaowen Sun. 2024. DeepSeek-V2: A Strong, Economical, and Efficient Mixture-of-Experts Language Model. arXiv preprint arXiv:2405.04434 (2024)."},{"key":"e_1_3_2_1_8_1","volume-title":"Partially Relevant Video Retrieval. In MM '22: The 30th ACM International Conference on Multimedia. 246-257","author":"Dong Jianfeng","year":"2022","unstructured":"Jianfeng Dong, Xianke Chen, Minsong Zhang, Xun Yang, Shujie Chen, Xirong Li, and Xun Wang. 2022. Partially Relevant Video Retrieval. In MM '22: The 30th ACM International Conference on Multimedia. 246-257."},{"key":"e_1_3_2_1_9_1","unstructured":"Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey and et al. 2024. The Llama 3 Herd of Models. arXiv preprint arXiv:2407.21783 (2024)."},{"volume-title":"Computer Vision - ECCV 2020 - 16th European Conference. 214-229.","author":"Gabeur Valentin","key":"e_1_3_2_1_10_1","unstructured":"Valentin Gabeur, Chen Sun, Karteek Alahari, and Cordelia Schmid. 2020. Multi-modal Transformer for Video Retrieval. In Computer Vision - ECCV 2020 - 16th European Conference. 214-229."},{"key":"e_1_3_2_1_11_1","unstructured":"Team GLM Aohan Zeng Bin Xu and et al. 2024. ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools. arXiv:2406.12793"},{"key":"e_1_3_2_1_12_1","first-page":"4996","volume-title":"X-Pool: Cross-Modal Language-Video Attention for Text-Video Retrieval. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Gorti Satya Krishna","year":"2022","unstructured":"Satya Krishna Gorti, No\u00ebl Vouitsis, Junwei Ma, Keyvan Golestan, Maksims Volkovs, Animesh Garg, and Guangwei Yu. 2022. X-Pool: Cross-Modal Language-Video Attention for Text-Video Retrieval. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022. 4996-5005."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.02242"},{"key":"e_1_3_2_1_14_1","first-page":"706","volume-title":"Dense-Captioning Events in Videos. In IEEE International Conference on Computer Vision","author":"Krishna Ranjay","year":"2017","unstructured":"Ranjay Krishna, Kenji Hata, Frederic Ren, Li Fei-Fei, and Juan Carlos Niebles. 2017. Dense-Captioning Events in Videos. In IEEE International Conference on Computer Vision 2017. 706-715."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00725"},{"key":"e_1_3_2_1_16_1","volume-title":"International Conference on Machine Learning","volume":"19742","author":"Li Junnan","year":"2023","unstructured":"Junnan Li, Dongxu Li, Silvio Savarese, and Steven C. H. Hoi. 2023. BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models. In International Conference on Machine Learning, 2023 (Proceedings of Machine Learning Research, Vol. 202). 19730-19742."},{"key":"e_1_3_2_1_17_1","volume-title":"The Twelfth International Conference on Learning Representations","author":"Li Tianhong","year":"2024","unstructured":"Tianhong Li, Sangnie Bhardwaj, Yonglong Tian, Han Zhang, Jarred Barber, Dina Katabi, Guillaume Lajoie, Huiwen Chang, and Dilip Krishnan. 2024a. Leveraging Unpaired Data for Vision-Language Generative Models via Cycle Consistency. In The Twelfth International Conference on Learning Representations, 2024."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3680673"},{"key":"e_1_3_2_1_19_1","volume-title":"Text-Adaptive Multiple Visual Prototype Matching for Video-Text Retrieval. In Advances in Neural Information Processing Systems 35: Annual Conference on Neural Information Processing Systems","author":"Lin Chengzhi","year":"2022","unstructured":"Chengzhi Lin, Ancong Wu, Junwei Liang, Jun Zhang, Wenhang Ge, Wei-Shi Zheng, and Chunhua Shen. 2022. Text-Adaptive Multiple Visual Prototype Matching for Video-Text Retrieval. In Advances in Neural Information Processing Systems 35: Annual Conference on Neural Information Processing Systems 2022."},{"key":"e_1_3_2_1_20_1","volume-title":"Mug-STAN: Adapting Image-Language Pretrained Models for General Video Understanding. arXiv preprint arXiv:2311.15075","author":"Liu Ruyang","year":"2023","unstructured":"Ruyang Liu, Jingjia Huang, Wei Gao, Thomas H Li, and Ge Li. 2023a. Mug-STAN: Adapting Image-Language Pretrained Models for General Video Understanding. arXiv preprint arXiv:2311.15075 (2023)."},{"key":"e_1_3_2_1_21_1","first-page":"6555","volume-title":"Revisiting Temporal Modeling for CLIP-Based Image-to-Video Knowledge Transferring. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Liu Ruyang","year":"2023","unstructured":"Ruyang Liu, Jingjia Huang, Ge Li, Jiashi Feng, Xinglong Wu, and Thomas H. Li. 2023b. Revisiting Temporal Modeling for CLIP-Based Image-to-Video Knowledge Transferring. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023. 6555-6564."},{"volume-title":"Revisiting Temporal Modeling for CLIP-Based Image-to-Video Knowledge Transferring. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 6555-6564","author":"Liu Ruyang","key":"e_1_3_2_1_22_1","unstructured":"Ruyang Liu, Jingjia Huang, Ge Li, Jiashi Feng, Xinglong Wu, and Thomas H. Li. 2023c. Revisiting Temporal Modeling for CLIP-Based Image-to-Video Knowledge Transferring. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 6555-6564."},{"key":"e_1_3_2_1_23_1","volume-title":"30th British Machine Vision Conference","author":"Liu Yang","year":"2019","unstructured":"Yang Liu, Samuel Albanie, Arsha Nagrani, and Andrew Zisserman. 2019. Use What You Have: Video retrieval using representations from collaborative experts. In 30th British Machine Vision Conference 2019. 279."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2022.07.028"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3680829"},{"key":"e_1_3_2_1_26_1","first-page":"8748","volume-title":"Proceedings of the 38th International Conference on Machine Learning","volume":"139","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. In Proceedings of the 38th International Conference on Machine Learning, 2021, Vol. 139. 8748-8763."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01834"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3571735"},{"key":"e_1_3_2_1_29_1","first-page":"233","volume-title":"Multi-query Video Retrieval. In Computer Vision-17th European Conference","volume":"13674","author":"Wang Zeyu","year":"2022","unstructured":"Zeyu Wang, Yu Wu, Karthik Narasimhan, and Olga Russakovsky. 2022. Multi-query Video Retrieval. In Computer Vision-17th European Conference, 2022, Vol. 13674. 233-249."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475515"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01031"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.571"},{"key":"e_1_3_2_1_33_1","volume-title":"The Eleventh International Conference on Learning Representations","author":"Xue Hongwei","year":"2023","unstructured":"Hongwei Xue, Yuchong Sun, Bei Liu, Jianlong Fu, Ruihua Song, Houqiang Li, and Jiebo Luo. 2023. CLIP-ViP: Adapting Pre-trained Image-Text Model to Video-Language Alignment. In The Eleventh International Conference on Learning Representations, 2023."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"crossref","unstructured":"Bowen Zhang Hexiang Hu and Fei Sha. 2018. Cross-modal and hierarchical modeling of video and text. In 2018 european conference on computer vision. 374-390.","DOI":"10.1007\/978-3-030-01261-8_23"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3680839"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Dublin Ireland","acronym":"MM '25"},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755275","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:45:15Z","timestamp":1765309515000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755275"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":35,"alternative-id":["10.1145\/3746027.3755275","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755275","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}