{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,21]],"date-time":"2026-02-21T18:31:00Z","timestamp":1771698660570,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":57,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,26]],"date-time":"2023-10-26T00:00:00Z","timestamp":1698278400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,26]]},"DOI":"10.1145\/3581783.3612006","type":"proceedings-article","created":{"date-parts":[[2023,10,27]],"date-time":"2023-10-27T07:27:30Z","timestamp":1698391650000},"page":"4626-4636","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":14,"title":["Dual-Modal Attention-Enhanced Text-Video Retrieval with Triplet Partial Margin Contrastive Learning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-7616-8926","authenticated-orcid":false,"given":"Chen","family":"Jiang","sequence":"first","affiliation":[{"name":"Artificial Intelligence Innovation and Incubation Institute, Fudan University &amp; Ant Group, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-2361-5721","authenticated-orcid":false,"given":"Hong","family":"Liu","sequence":"additional","affiliation":[{"name":"Ant Group, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-9752-799X","authenticated-orcid":false,"given":"Xuzheng","family":"Yu","sequence":"additional","affiliation":[{"name":"Ant Group, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-1353-7541","authenticated-orcid":false,"given":"Qing","family":"Wang","sequence":"additional","affiliation":[{"name":"Ant Group, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2502-9101","authenticated-orcid":false,"given":"Yuan","family":"Cheng","sequence":"additional","affiliation":[{"name":"Artificial Intelligence Innovation and Incubation Institute, Fudan University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-1163-513X","authenticated-orcid":false,"given":"Jia","family":"Xu","sequence":"additional","affiliation":[{"name":"Ant Group, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9478-8107","authenticated-orcid":false,"given":"Zhongyi","family":"Liu","sequence":"additional","affiliation":[{"name":"Ant Group, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8638-6594","authenticated-orcid":false,"given":"Qingpei","family":"Guo","sequence":"additional","affiliation":[{"name":"Ant Group, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6401-6111","authenticated-orcid":false,"given":"Wei","family":"Chu","sequence":"additional","affiliation":[{"name":"Ant Group, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1691-6817","authenticated-orcid":false,"given":"Ming","family":"Yang","sequence":"additional","affiliation":[{"name":"Ant Group, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-9377-5755","authenticated-orcid":false,"given":"Yuan","family":"Qi","sequence":"additional","affiliation":[{"name":"Artificial Intelligence Innovation and Incubation Institute, Fudan University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1016\/S0306-4573(02)00021-3"},{"key":"e_1_3_2_1_2_1","volume-title":"Frozen in Time: A Joint Video and Image Encoder for End-to-End Retrieval. In 2021 IEEE\/CVF International Conference on Computer Vision (ICCV). 1708--1718","author":"Bain Max","year":"2021","unstructured":"Max Bain, Arsha Nagrani, G\u00fcl Varol, and Andrew Zisserman. 2021. Frozen in Time: A Joint Video and Image Encoder for End-to-End Retrieval. In 2021 IEEE\/CVF International Conference on Computer Vision (ICCV). 1708--1718."},{"key":"e_1_3_2_1_3_1","volume-title":"Cross Modal Retrieval with Querybank Normalisation. In 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 5184--5195","author":"Bogolin Simion-Vlad","year":"2021","unstructured":"Simion-Vlad Bogolin, Ioana Croitoru, Hailin Jin, Yang Liu, and Samuel Albanie. 2021. Cross Modal Retrieval with Querybank Normalisation. In 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 5184--5195."},{"key":"e_1_3_2_1_4_1","volume-title":"Collecting Highly Parallel Data for Paraphrase Evaluation. In Annual Meeting of the Association for Computational Linguistics.","author":"David","unstructured":"David L. Chen and William B. Dolan. 2011. Collecting Highly Parallel Data for Paraphrase Evaluation. In Annual Meeting of the Association for Computational Linguistics."},{"key":"e_1_3_2_1_5_1","volume-title":"Fine-Grained Video-Text Retrieval With Hierarchical Graph Reasoning. In 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 10635--10644","author":"Chen Shizhe","year":"2020","unstructured":"Shizhe Chen, Yida Zhao, Qin Jin, and Qi Wu. 2020. Fine-Grained Video-Text Retrieval With Hierarchical Graph Reasoning. In 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 10635--10644."},{"key":"e_1_3_2_1_6_1","volume-title":"Improving Video-Text Retrieval by Multi-Stream Corpus Alignment and Dual Softmax Loss. ArXiv","author":"Cheng Xingyi","year":"2021","unstructured":"Xingyi Cheng, Hezheng Lin, Xiangyu Wu, F. Yang, and Dong Shen. 2021. Improving Video-Text Retrieval by Multi-Stream Corpus Alignment and Dual Softmax Loss. ArXiv, Vol. abs\/2109.04290 (2021)."},{"key":"e_1_3_2_1_7_1","first-page":"4065","article-title":"Dual Encoding for Video Retrieval by Text","volume":"44","author":"Dong Jianfeng","year":"2021","unstructured":"Jianfeng Dong, Xirong Li, Chaoxi Xu, Xun Yang, Gang Yang, Xun Wang, and Meng Wang. 2021. Dual Encoding for Video Retrieval by Text. IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI), Vol. 44 (2021), 4065--4080.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI)"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/3512527.3531395"},{"key":"e_1_3_2_1_9_1","volume-title":"CLIP2Video: Mastering Video-Text Retrieval via Image CLIP. ArXiv","author":"Fang Han","year":"2021","unstructured":"Han Fang, Pengfei Xiong, Luhui Xu, and Yu Chen. 2021. CLIP2Video: Mastering Video-Text Retrieval via Image CLIP. ArXiv, Vol. abs\/2106.11097 (2021)."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547785"},{"key":"e_1_3_2_1_11_1","volume-title":"Multi-modal Transformer for Video Retrieval. In European Conference on Computer Vision (ECCV).","author":"Gabeur Valentin","year":"2020","unstructured":"Valentin Gabeur, Chen Sun, Alahari Karteek, and Cordelia Schmid. 2020. Multi-modal Transformer for Video Retrieval. In European Conference on Computer Vision (ECCV)."},{"key":"e_1_3_2_1_12_1","volume-title":"SST-VLM: Sparse Sampling-Twice Inspired Video-Language Model. In Asian Conference on Computer Vision (ACCV).","author":"Gao Yizhao","year":"2022","unstructured":"Yizhao Gao and Zhiwu Lu. 2022. SST-VLM: Sparse Sampling-Twice Inspired Video-Language Model. In Asian Conference on Computer Vision (ACCV)."},{"key":"e_1_3_2_1_13_1","volume-title":"CLIP2TV: An Empirical Study on Transformer-based Methods for Video-Text Retrieval. ArXiv","author":"Gao Zijian","year":"2021","unstructured":"Zijian Gao, Jingyu Liu, Sheng Chen, Dedan Chang, Hao Zhang, and Jinwei Yuan. 2021. CLIP2TV: An Empirical Study on Transformer-based Methods for Video-Text Retrieval. ArXiv, Vol. abs\/2111.05610 (2021)."},{"key":"e_1_3_2_1_14_1","volume-title":"X-Pool: Cross-Modal Language-Video Attention for Text-Video Retrieval. In 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 4996--5005","author":"Gorti Satya Krishna","year":"2022","unstructured":"Satya Krishna Gorti, Noel Vouitsis, Junwei Ma, Keyvan Golestan, Maksims Volkovs, Animesh Garg, and Guangwei Yu. 2022. X-Pool: Cross-Modal Language-Video Attention for Text-Video Retrieval. In 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 4996--5005."},{"key":"e_1_3_2_1_15_1","volume-title":"International Conference on Artificial Intelligence and Statistics.","author":"Gutmann Michael U","year":"2010","unstructured":"Michael U Gutmann and Aapo Hyv\u00e4rinen. 2010. Noise-contrastive estimation: A new estimation principle for unnormalized statistical models. In International Conference on Artificial Intelligence and Statistics."},{"key":"e_1_3_2_1_16_1","volume-title":"Momentum Contrast for Unsupervised Visual Representation Learning. In 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 9726--9735","author":"He Kaiming","unstructured":"Kaiming He, Haoqi Fan, Yuxin Wu, Saining Xie, and Ross B. Girshick. 2019. Momentum Contrast for Unsupervised Visual Representation Learning. In 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 9726--9735."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298698"},{"key":"e_1_3_2_1_18_1","volume-title":"Localizing Moments in Video with Natural Language. In 2017 IEEE International Conference on Computer Vision (ICCV). 5804--5813","author":"Hendricks Lisa Anne","unstructured":"Lisa Anne Hendricks, Oliver Wang, Eli Shechtman, Josef Sivic, Trevor Darrell, and Bryan C. Russell. 2017. Localizing Moments in Video with Natural Language. In 2017 IEEE International Conference on Computer Vision (ICCV). 5804--5813."},{"key":"e_1_3_2_1_19_1","unstructured":"Matthew Honnibal and Ines Montani. 2017. spaCy 2: Natural language understanding with Bloom embeddings convolutional neural networks and incremental parsing. (2017)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1108\/00220410410560573"},{"key":"e_1_3_2_1_21_1","volume-title":"Noe Pion, Philippe Weinzaepfel, and Diane Larlus.","author":"Kalantidis Yannis","year":"2020","unstructured":"Yannis Kalantidis, Mert Bulent Sariyildiz, Noe Pion, Philippe Weinzaepfel, and Diane Larlus. 2020. Hard Negative Mixing for Contrastive Learning. In Advances in Neural Information Processing Systems (NeurIPS). Curran Associates, Inc."},{"key":"e_1_3_2_1_22_1","volume-title":"Kingma and Jimmy Ba","author":"Diederik","year":"2014","unstructured":"Diederik P. Kingma and Jimmy Ba. 2014. Adam: A Method for Stochastic Optimization. ArXiv, Vol. abs\/1412.6980 (2014)."},{"key":"e_1_3_2_1_23_1","volume-title":"International Conference on Learning Representations (ICLR).","author":"Kolesnikov Alexander","year":"2021","unstructured":"Alexander Kolesnikov, Alexey Dosovitskiy, Dirk Weissenborn, Georg Heigold, Jakob Uszkoreit, Lucas Beyer, Matthias Minderer, Mostafa Dehghani, Neil Houlsby, Sylvain Gelly, Thomas Unterthiner, and Xiaohua Zhai. 2021. An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_24_1","volume-title":"Align and Prompt: Video-and-Language Pre-training with Entity Prompts. In 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 4943--4953","author":"Li Dongxu","unstructured":"Dongxu Li, Junnan Li, Hongdong Li, Juan Carlos Niebles, and Steven C. H. Hoi. 2021. Align and Prompt: Video-and-Language Pre-training with Entity Prompts. In 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 4943--4953."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.161"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3512527.3531429"},{"key":"e_1_3_2_1_27_1","volume-title":"A strong and robust baseline for text-image matching. arXiv preprint arXiv:1906.01205","author":"Liu Fangyu","year":"2019","unstructured":"Fangyu Liu and Rongtian Ye. 2019. A strong and robust baseline for text-image matching. arXiv preprint arXiv:1906.01205 (2019)."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00634"},{"key":"e_1_3_2_1_29_1","volume-title":"HiT: Hierarchical Transformer with Momentum Contrast for Video-Text Retrieval. In 2021 IEEE\/CVF International Conference on Computer Vision (ICCV). 11895--11905","author":"Liu Song","year":"2021","unstructured":"Song Liu, Haoqi Fan, Shengsheng Qian, Yiru Chen, Wenkui Ding, and Zhongyuan Wang. 2021a. HiT: Hierarchical Transformer with Momentum Contrast for Video-Text Retrieval. In 2021 IEEE\/CVF International Conference on Computer Vision (ICCV). 11895--11905."},{"key":"e_1_3_2_1_30_1","volume-title":"Contrastive Multimodal Fusion with TupleInfoNCE. In 2021 IEEE\/CVF International Conference on Computer Vision (ICCV). 734--743","author":"Liu Yunze","year":"2021","unstructured":"Yunze Liu, Qingnan Fan, Shanghang Zhang, Hao Dong, Thomas A. Funkhouser, and Li Yi. 2021b. Contrastive Multimodal Fusion with TupleInfoNCE. In 2021 IEEE\/CVF International Conference on Computer Vision (ICCV). 734--743."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19781-9_19"},{"key":"e_1_3_2_1_32_1","volume-title":"SGDR: Stochastic Gradient Descent with Warm Restarts. In 5th International Conference on Learning Representations (ICLR).","author":"Loshchilov Ilya","year":"2017","unstructured":"Ilya Loshchilov and Frank Hutter. 2017. SGDR: Stochastic Gradient Descent with Warm Restarts. In 5th International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_1_33_1","volume-title":"LGDN: Language-Guided Denoising Network for Video-Language Modeling. In Advances in Neural Information Processing Systems (NeurIPS).","author":"Lu Haoyu","year":"2022","unstructured":"Haoyu Lu, Mingyu Ding, Nanyi Fei, Yuqi Huo, and Zhiwu Lu. 2022. LGDN: Language-Guided Denoising Network for Video-Language Modeling. In Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_2_1_34_1","volume-title":"UniViLM: A Unified Video and Language Pre-Training Model for Multimodal Understanding and Generation. ArXiv","author":"Luo Huaishao","year":"2020","unstructured":"Huaishao Luo, Lei Ji, Botian Shi, Haoyang Huang, Nan Duan, Tianrui Li, Xilin Chen, and Ming Zhou. 2020. UniViLM: A Unified Video and Language Pre-Training Model for Multimodal Understanding and Generation. ArXiv, Vol. abs\/2002.06353 (2020)."},{"key":"e_1_3_2_1_35_1","volume-title":"CLIP4Clip: An Empirical Study of CLIP for End to End Video Clip Retrieval and Captioning. Neurocomput","author":"Luo Huaishao","year":"2022","unstructured":"Huaishao Luo, Lei Ji, Ming Zhong, Yang Chen, Wen Lei, Nan Duan, and Tianrui Li. 2022. CLIP4Clip: An Empirical Study of CLIP for End to End Video Clip Retrieval and Captioning. Neurocomput., Vol. 508, C (oct 2022), 293--304."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547910"},{"key":"e_1_3_2_1_37_1","volume-title":"Bermano","author":"Mokady Ron","year":"2021","unstructured":"Ron Mokady, Amir Hertz, and Amit H. Bermano. 2021. ClipCap: CLIP Prefix for Image Captioning. ArXiv, Vol. abs\/2111.09734 (2021)."},{"key":"e_1_3_2_1_38_1","volume-title":"StyleCLIP: Text-Driven Manipulation of StyleGAN Imagery. In 2021 IEEE\/CVF International Conference on Computer Vision (ICCV). 2065--2074","author":"Patashnik Or","year":"2021","unstructured":"Or Patashnik, Zongze Wu, Eli Shechtman, Daniel Cohen-Or, and Dani Lischinski. 2021. StyleCLIP: Text-Driven Manipulation of StyleGAN Imagery. In 2021 IEEE\/CVF International Conference on Computer Vision (ICCV). 2065--2074."},{"key":"e_1_3_2_1_39_1","volume-title":"Learning Transferable Visual Models From Natural Language Supervision. In International Conference on Machine Learning (ICML).","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. In International Conference on Machine Learning (ICML)."},{"key":"e_1_3_2_1_40_1","article-title":"Hubs in space: Popular nearest neighbors in high-dimensional data","volume":"11","author":"Radovanovic Milos","year":"2010","unstructured":"Milos Radovanovic, Alexandros Nanopoulos, and Mirjana Ivanovic. 2010. Hubs in space: Popular nearest neighbors in high-dimensional data. Journal of Machine Learning Research, Vol. 11, sept (2010), 2487--2531.","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_2_1_41_1","volume-title":"Hierarchical Text-Conditional Image Generation with CLIP Latents. ArXiv","author":"Ramesh Aditya","year":"2022","unstructured":"Aditya Ramesh, Prafulla Dhariwal, Alex Nichol, Casey Chu, and Mark Chen. 2022. Hierarchical Text-Conditional Image Generation with CLIP Latents. ArXiv, Vol. abs\/2204.06125 (2022)."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298682"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P16-1162"},{"key":"e_1_3_2_1_44_1","volume-title":"VideoBERT: A Joint Model for Video and Language Representation Learning. In 2019 IEEE\/CVF International Conference on Computer Vision (ICCV). 7463--7472","author":"Sun Chen","year":"2019","unstructured":"Chen Sun, Austin Myers, Carl Vondrick, Kevin P. Murphy, and Cordelia Schmid. 2019. VideoBERT: A Joint Model for Video and Language Representation Learning. In 2019 IEEE\/CVF International Conference on Computer Vision (ICCV). 7463--7472."},{"key":"e_1_3_2_1_45_1","volume-title":"Representation Learning with Contrastive Predictive Coding. CoRR","author":"van den Oord A\u00e4ron","year":"2018","unstructured":"A\u00e4ron van den Oord, Yazhe Li, and Oriol Vinyals. 2018. Representation Learning with Contrastive Predictive Coding. CoRR (2018)."},{"key":"e_1_3_2_1_46_1","volume-title":"ActionCLIP: A New Paradigm for Video Action Recognition. ArXiv","author":"Wang Mengmeng","year":"2021","unstructured":"Mengmeng Wang, Jiazheng Xing, and Yong Liu. 2021. ActionCLIP: A New Paradigm for Video Action Recognition. ArXiv, Vol. abs\/2109.08472 (2021)."},{"key":"e_1_3_2_1_47_1","volume-title":"Disentangled Representation Learning for Text-Video Retrieval. ArXiv","author":"Wang Qiang","year":"2022","unstructured":"Qiang Wang, Yanhao Zhang, Yun Zheng, Pan Pan, and Xian-Sheng Hua. 2022b. Disentangled Representation Learning for Text-Video Retrieval. ArXiv, Vol. abs\/2203.07111 (2022)."},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547968"},{"key":"e_1_3_2_1_49_1","volume-title":"Fine-Grained Action Retrieval Through Multiple Parts-of-Speech Embeddings. In 2019 IEEE\/CVF International Conference on Computer Vision (ICCV). 450--459","author":"Wray Michael","year":"2019","unstructured":"Michael Wray, Diane Larlus, Gabriela Csurka, and Dima Damen. 2019. Fine-Grained Action Retrieval Through Multiple Parts-of-Speech Embeddings. In 2019 IEEE\/CVF International Conference on Computer Vision (ICCV). 450--459."},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475515"},{"key":"e_1_3_2_1_51_1","volume-title":"MSR-VTT: A Large Video Description Dataset for Bridging Video and Language. In 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR). 5288--5296","author":"Xu Jun","year":"2016","unstructured":"Jun Xu, Tao Mei, Ting Yao, and Yong Rui. 2016. MSR-VTT: A Large Video Description Dataset for Bridging Video and Language. In 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR). 5288--5296."},{"key":"e_1_3_2_1_52_1","volume-title":"TACo: Token-aware Cascade Contrastive Learning for Video-Text Alignment. In 2021 IEEE\/CVF International Conference on Computer Vision (ICCV). 11542--11552","author":"Yang Jianwei","year":"2021","unstructured":"Jianwei Yang, Yonatan Bisk, and Jianfeng Gao. 2021. TACo: Token-aware Cascade Contrastive Learning for Video-Text Alignment. In 2021 IEEE\/CVF International Conference on Computer Vision (ICCV). 11542--11552."},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"crossref","unstructured":"Ye Yuan Wuyang Chen Yang Yang and Zhangyang Wang. 2020. In Defense of the Triplet Loss Again: Learning Robust Person Re-Identification with Fast Approximated Triplet Loss and Label Distillation. In 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshops (CVPRW). 1454--1463.","DOI":"10.1109\/CVPRW50498.2020.00185"},{"key":"e_1_3_2_1_54_1","volume-title":"Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics. 4892--4903","author":"Yuhao Zhang","year":"2022","unstructured":"Zhang Yuhao, Zhu Hongji, Wang Yongliang, Xu Nan, Li Xiaobo, and Zhao Binqiang. 2022. A Contrastive Framework for Learning Sentence Representations from Pairwise and Triple-wise Perspective in Angular Space. In Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics. 4892--4903."},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1145\/3477495.3531950"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-022-01653-1"},{"key":"e_1_3_2_1_57_1","volume-title":"ActBERT: Learning Global-Local Video-Text Representations. In 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 8743--8752","author":"Zhu Linchao","year":"2020","unstructured":"Linchao Zhu and Yi Yang. 2020. ActBERT: Learning Global-Local Video-Text Representations. In 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 8743--8752."}],"event":{"name":"MM '23: The 31st ACM International Conference on Multimedia","location":"Ottawa ON Canada","acronym":"MM '23","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 31st ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612006","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3581783.3612006","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T23:57:39Z","timestamp":1755820659000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3612006"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,26]]},"references-count":57,"alternative-id":["10.1145\/3581783.3612006","10.1145\/3581783"],"URL":"https:\/\/doi.org\/10.1145\/3581783.3612006","relation":{},"subject":[],"published":{"date-parts":[[2023,10,26]]},"assertion":[{"value":"2023-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}