{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:16:29Z","timestamp":1765307789915,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":83,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3758179","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T05:44:48Z","timestamp":1761371088000},"page":"12509-12518","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Turing Patterns for Multimedia: Reaction-Diffusion Multi-Modal Fusion for Language-Guided Video Moment Retrieval"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3231-5771","authenticated-orcid":false,"given":"Xiang","family":"Fang","sequence":"first","affiliation":[{"name":"Nanyang Technological University, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4480-3107","authenticated-orcid":false,"given":"Wanlong","family":"Fang","sequence":"additional","affiliation":[{"name":"Nanyang Technological University, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8106-9768","authenticated-orcid":false,"given":"Wei","family":"Ji","sequence":"additional","affiliation":[{"name":"Nanjing University, Suzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6097-7807","authenticated-orcid":false,"given":"Tat-Seng","family":"Chua","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Imperceptible Beam-Sensitive Adversarial Attacks for LiDAR-based Object Detection in Autonomous Driving. In IEEE International Conference on Multimedia & Expo 2025 (ICME","author":"Cai Fuyao","year":"2025","unstructured":"Fuyao Cai, Daizong Liu, Xiang Fang, Jixiang Yu, Keke Tang, and Pan Zhou. 2025a. Imperceptible Beam-Sensitive Adversarial Attacks for LiDAR-based Object Detection in Autonomous Driving. In IEEE International Conference on Multimedia & Expo 2025 (ICME 2025)."},{"key":"e_1_3_2_1_2_1","unstructured":"Xiaowen Cai Daizong Liu Xiaoye Qu Xiang Fang Jianfeng Dong Keke Tang Pan Zhou Lichao Sun and Wei Hu. 2025b. Towards Building Model\/Prompt-Transferable Attackers against Large Vision-Language Models. In Advances in Neural Information Processing Systems."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"e_1_3_2_1_4_1","first-page":"213","volume-title":"End-to-End Object Detection with Transformers. European Conference on Computer Vision (ECCV)","author":"Carion Nicolas","year":"2020","unstructured":"Nicolas Carion, Filippo Massa, Gabriel Synnaeve, Nicolas Usunier, Alexander Kirillov, and Sergey Zagoruyko. 2020b. End-to-End Object Detection with Transformers. European Conference on Computer Vision (ECCV) (2020), 213-229."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"crossref","unstructured":"Jingyuan Chen Xinpeng Chen Lin Ma Zequn Jie and Tat-Seng Chua. 2018. Temporally grounding natural sentence in video. In EMNLP.","DOI":"10.18653\/v1\/D18-1015"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1810.04805"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"crossref","unstructured":"Xiang Fang Arvind Easwaran and Blaise Genest. 2024a. Uncertainty-Guided Appearance-Motion Association Network for Out-of-Distribution Action Detection. In MIPR.","DOI":"10.1109\/MIPR62202.2024.00034"},{"key":"e_1_3_2_1_8_1","volume-title":"Adaptive Multi-prompt Contrastive Network for Few-shot Out-of-distribution Detection. arXiv preprint arXiv:2506.17633","author":"Fang Xiang","year":"2025","unstructured":"Xiang Fang, Arvind Easwaran, and Blaise Genest. 2025a. Adaptive Multi-prompt Contrastive Network for Few-shot Out-of-distribution Detection. arXiv preprint arXiv:2506.17633 (2025)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2024.126031"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3680947"},{"key":"e_1_3_2_1_11_1","unstructured":"Xiang Fang Wanlong Fang and Changshuo Wang. 2025c. Hierarchical Semantic-Augmented Navigation: Optimal Transport and Graph-Driven Reasoning for Vision-Language Navigation. In Advances in Neural Information Processing Systems."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i3.32298"},{"key":"e_1_3_2_1_13_1","volume-title":"Double self-weighted multi-view clustering via adaptive view fusion. arXiv preprint arXiv:2011.10396","author":"Fang Xiang","year":"2020","unstructured":"Xiang Fang and Yuchong Hu. 2020. Double self-weighted multi-view clustering via adaptive view fusion. arXiv preprint arXiv:2011.10396 (2020)."},{"key":"e_1_3_2_1_14_1","volume-title":"Animc: A soft approach for autoweighted noisy and incomplete multiview clustering","author":"Fang Xiang","year":"2021","unstructured":"Xiang Fang, Yuchong Hu, Pan Zhou, and Dapeng Wu. 2021a. Animc: A soft approach for autoweighted noisy and incomplete multiview clustering. IEEE TAI (2021)."},{"key":"e_1_3_2_1_15_1","volume-title":"Unbalanced incomplete multi-view clustering via the scheme of view evolution: Weak views are meat","author":"Fang Xiang","year":"2021","unstructured":"Xiang Fang, Yuchong Hu, Pan Zhou, and Dapeng Oliver Wu. 2021b. Unbalanced incomplete multi-view clustering via the scheme of view evolution: Weak views are meat; strong views do eat. IEEE TETCI (2021)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/TAI.2021.3052425"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"crossref","unstructured":"Xiang Fang Daizong Liu Wanlong Fang Pan Zhou Yu Cheng Keke Tang and Kai Zou. 2023a. Annotations Are Not All You Need: A Cross-modal Knowledge Transfer Network for Unsupervised Temporal Sentence Grounding. In Findings of EMNLP.","DOI":"10.18653\/v1\/2023.findings-emnlp.583"},{"key":"e_1_3_2_1_18_1","first-page":"1735","volume-title":"Better Performance: Efficient Cross-Modal Clip Trimming for Video Moment Retrieval Using Language. In Proceedings of the AAAI Conference on Artificial Intelligence","volume":"38","author":"Fang Xiang","year":"2024","unstructured":"Xiang Fang, Daizong Liu, Wanlong Fang, Pan Zhou, Zichuan Xu, Wenzheng Xu, Junyang Chen, and Renfu Li. 2024c. Fewer Steps, Better Performance: Efficient Cross-Modal Clip Trimming for Video Moment Retrieval Using Language. In Proceedings of the AAAI Conference on Artificial Intelligence, Vol. 38. 1735-1743."},{"key":"e_1_3_2_1_19_1","volume-title":"Multi-Modal Cross-Domain Alignment Network for Video Moment Retrieval. TMM","author":"Fang Xiang","year":"2022","unstructured":"Xiang Fang, Daizong Liu, Pan Zhou, and Yuchong Hu. 2022. Multi-Modal Cross-Domain Alignment Network for Video Moment Retrieval. TMM (2022)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"crossref","unstructured":"Xiang Fang Daizong Liu Pan Zhou and Guoshun Nan. 2023b. You Can Ground Earlier than See: An Effective and Efficient Pipeline for Temporal Sentence Grounding in Compressed Videos. In CVPR.","DOI":"10.1109\/CVPR52729.2023.00242"},{"key":"e_1_3_2_1_21_1","volume-title":"Hierarchical local-global transformer for temporal sentence grounding. TMM","author":"Fang Xiang","year":"2023","unstructured":"Xiang Fang, Daizong Liu, Pan Zhou, Zichuan Xu, and Ruixuan Li. 2023c. Hierarchical local-global transformer for temporal sentence grounding. TMM (2023)."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"crossref","unstructured":"Xiang Fang Zeyu Xiong Wanlong Fang Xiaoye Qu Chen Chen Jianfeng Dong Keke Tang Pan Zhou Yu Cheng and Daizong Liu. 2024d. Rethinking Weakly-supervised Video Temporal Grounding From a Game Perspective. In ECCV.","DOI":"10.1007\/978-3-031-72995-9_17"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00730"},{"key":"e_1_3_2_1_24_1","volume-title":"Proceedings of the International Conference on Machine Learning.","author":"Fei Hao","year":"2024","unstructured":"Hao Fei, Shengqiong Wu, Wei Ji, Hanwang Zhang, Meishan Zhang, Mong-Li Lee, and Wynne Hsu. 2024b. Video-of-thought: Step-by-step video reasoning from perception to cognition. In Proceedings of the International Conference on Machine Learning."},{"key":"e_1_3_2_1_25_1","volume-title":"Editing. Proceedings of the Advances in neural information processing systems.","author":"Fei Hao","year":"2024","unstructured":"Hao Fei, Shengqiong Wu, Hanwang Zhang, Tat-Seng Chua, and Shuicheng Yan. 2024c. VITRON: A Unified Pixel-level Vision LLM for Understanding, Generating, Segmenting, Editing. Proceedings of the Advances in neural information processing systems."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2024.3393452"},{"key":"e_1_3_2_1_27_1","volume-title":"Proceedings of the International Conference on Machine Learning.","author":"Fei Hao","year":"2025","unstructured":"Hao Fei, Yuan Zhou, Juncheng Li, Xiangtai Li, Qingshan Xu, Bobo Li, Shengqiong Wu, Yaoting Wang, Junbao Zhou, Jiahao Meng, et al., 2025. On path to multimodal generalist: General-level and general-bench. In Proceedings of the International Conference on Machine Learning."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00630"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.563"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.563"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3578517"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1016\/0009-2509(84)87007-0"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2024.3354731"},{"key":"e_1_3_2_1_34_1","first-page":"4904","volume-title":"Scaling Up Visual and Vision-Language Representation Learning With Noisy Text Supervision. International Conference on Machine Learning","author":"Jia Chao","year":"2022","unstructured":"Chao Jia, Yinfei Yang, Ye Xia, Yi-Ting Chen, Zarana Parekh, Hieu Pham, Quoc V. Le, Yun-Hsuan Sung, Zhen Li, and Tom Duerig. 2022. Scaling Up Visual and Vision-Language Representation Learning With Noisy Text Supervision. International Conference on Machine Learning (2022), 4904-4916."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"crossref","unstructured":"Ranjay Krishna Kenji Hata Frederic Ren Li Fei-Fei and Juan Carlos Niebles. 2017. Dense-captioning events in videos. In ICCV.","DOI":"10.1109\/ICCV.2017.83"},{"key":"e_1_3_2_1_36_1","volume-title":"Exploring Disentangled Appearance-Motion Contexts for Temporal Activity Localization. In 2025 International Joint Conference on Neural Networks (IJCNN","author":"Lei Huashuo","year":"2025","unstructured":"Huashuo Lei, Xiaowen Cai, Daizong Liu, Xiang Fang, Xiaoye Qu, Jianfeng Dong, Jixiang Yu, and Keyan Jin. 2025. Exploring Disentangled Appearance-Motion Contexts for Temporal Activity Localization. In 2025 International Joint Conference on Neural Networks (IJCNN 2025)."},{"key":"e_1_3_2_1_37_1","first-page":"11846","article-title":"Detecting Moments and Highlights in Videos via Natural Language Queries","volume":"34","author":"Lei Jie","year":"2021","unstructured":"Jie Lei, Tamara Wang, Yiwu Zhou, Xihui Fan, Tong Lin, and Yang Yang. 2021a. Detecting Moments and Highlights in Videos via Natural Language Queries. Advances in Neural Information Processing Systems (NeurIPS), Vol. 34 (2021), 11846-11858.","journal-title":"Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"e_1_3_2_1_38_1","first-page":"20354","article-title":"QVHighlights: Detecting Moments and Highlights in Videos via Natural Language Queries","volume":"34","author":"Lei Jie","year":"2021","unstructured":"Jie Lei, Licheng Yu, Tamara L. Berg, and Mohit Bansal. 2021b. QVHighlights: Detecting Moments and Highlights in Videos via Natural Language Queries. Advances in Neural Information Processing Systems, Vol. 34 (2021), 20354-20366.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_39_1","first-page":"2798","volume-title":"UniVTG: Towards Unified Video-Language Temporal Grounding. IEEE International Conference on Computer Vision (ICCV)","author":"Lin Kevin","year":"2023","unstructured":"Kevin Lin, Xuefeng Zhao, and Yang Yang. 2023. UniVTG: Towards Unified Video-Language Temporal Grounding. IEEE International Conference on Computer Vision (ICCV) (2023), 2798-2808."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2023.3238514"},{"key":"e_1_3_2_1_41_1","unstructured":"Daizong Liu Xiang Fang Xiaoye Qu Jianfeng Dong He Yan Yang Yang Pan Zhou and Yu Cheng. 2024a. Unsupervised Domain Adaptative Temporal Sentence Localization with Mutual Information Maximization. In AAAI."},{"key":"e_1_3_2_1_42_1","unstructured":"Daizong Liu Xiang Fang Pan Zhou Xing Di Weining Lu and Yu Cheng. 2023b. Hypotheses tree building for one-shot temporal sentence localization. In AAAI."},{"key":"e_1_3_2_1_43_1","unstructured":"Daizong Liu Xiaoye Qu Xiang Fang Jianfeng Dong Pan Zhou Guoshun Nan Keke Tang Wanlong Fang and Yu Cheng. 2024b. Towards Robust Temporal Activity Localization Learning with Noisy Labels. In COLING."},{"key":"e_1_3_2_1_44_1","first-page":"52127","article-title":"Pandora's box: Towards building universal attackers against real-world large vision-language models","volume":"37","author":"Liu Daizong","year":"2024","unstructured":"Daizong Liu, Mingyu Yang, Xiaoye Qu, Pan Zhou, Xiang Fang, Keke Tang, Yao Wan, and Lichao Sun. 2024d. Pandora's box: Towards building universal attackers against real-world large vision-language models. Advances in Neural Information Processing Systems, Vol. 37 (2024), 52127-52158.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_45_1","volume-title":"Conditional Video Diffusion Network for Fine-grained Temporal Sentence Grounding","author":"Liu Daizong","year":"2023","unstructured":"Daizong Liu, Jiahao Zhu, Xiang Fang, Zeyu Xiong, Huan Wang, Renfu Li, and Pan Zhou. 2023c. Conditional Video Diffusion Network for Fine-grained Temporal Sentence Grounding. IEEE TMM (2023)."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2403.04567"},{"key":"e_1_3_2_1_47_1","volume-title":"ELIOT: Zero-Shot Video-Text Retrieval through Relevance-Boosted Captioning and Structural Information Extraction. In Proceedings of the 2025 Conference of the Nations of the Americas","author":"Liu Xuye","year":"2025","unstructured":"Xuye Liu, Yimu Wang, and Jian Zhao. 2025. ELIOT: Zero-Shot Video-Text Retrieval through Relevance-Boosted Captioning and Structural Information Extraction. In Proceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 4: Student Research Workshop), Abteen Ebrahimi, Samar Haider, Emmy Liu, Sammar Haider, Maria Leonor Pacheco, and Shira Wein (Eds.). Association for Computational Linguistics, Albuquerque, USA, 381-391. https:\/\/aclanthology.org\/2025.naacl-srw.37\/"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00318"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2025.3562083"},{"key":"e_1_3_2_1_50_1","first-page":"12345","volume-title":"QD-DETR: Query-Dependent Detection Transformer for Temporal Sentence Grounding. IEEE Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Moon Seongho","year":"2023","unstructured":"Seongho Moon, Sangmin Lee, Youngtaek Kim, and Juho Cho. 2023. QD-DETR: Query-Dependent Detection Transformer for Temporal Sentence Grounding. IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2023), 12345-12354."},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1007\/b98868"},{"key":"e_1_3_2_1_52_1","first-page":"8748","volume-title":"Learning Transferable Visual Models From Natural Language Supervision. International Conference on Machine Learning (ICML)","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. International Conference on Machine Learning (ICML) (2021), 8748-8763."},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00474"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"crossref","unstructured":"Keke Tang Wenyu Zhao Weilong Peng Xiang Fang Xiaodong Cui Peican Zhu and Zhihong Tian. 2024. Reparameterization Head for Efficient Multi-Input Networks. In ICASSP.","DOI":"10.1109\/ICASSP48485.2024.10447574"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1098\/rstb.1952.0012"},{"key":"e_1_3_2_1_56_1","first-page":"5998","article-title":"Attention is All You Need","volume":"30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszoreit, Llion Jones, Aidan N. Gomez, Lukasz Kaiser, and Illia Polosukhin. 2017. Attention is All You Need. Advances in Neural Information Processing Systems, Vol. 30 (2017), 5998-6008.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_57_1","volume-title":"DyPolySeg: Taylor Series-Inspired Dynamic Polynomial Fitting Network for Few-shot Point Cloud Semantic Segmentation. In Forty-second International Conference on Machine Learning.","author":"Wang Changshuo","year":"2025","unstructured":"Changshuo Wang, Xiang Fang, and Prayag Tiwari. 2025b. DyPolySeg: Taylor Series-Inspired Dynamic Polynomial Fitting Network for Few-shot Point Cloud Semantic Segmentation. In Forty-second International Conference on Machine Learning."},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.02066"},{"key":"e_1_3_2_1_59_1","unstructured":"Changshuo Wang Shuting He Xiang Fang Zhijian Hu Jiahong Huang Yixian Shen and Prayag Tiwari. 2025d. Reasoning Beyond Points: A Visual Introspective Approach for Few-Shot 3D Segmentation. In Advances in Neural Information Processing Systems."},{"key":"e_1_3_2_1_60_1","volume-title":"Seeing the Overlooked: Bio-Visual Inspired Weak Saliency Feedback Transformer for Person Re-identification. In ACM International Conference on Multimedia.","author":"Wang Changshuo","year":"2025","unstructured":"Changshuo Wang, Shuting He, Xiang Fang, Fangzhe Nan, and Prayag Tiwari. 2025 e. Seeing the Overlooked: Bio-Visual Inspired Weak Saliency Feedback Transformer for Person Re-identification. In ACM International Conference on Multimedia."},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i7.32810"},{"key":"e_1_3_2_1_62_1","volume-title":"2025 g. Prototype-Driven Structure Synergy Network for Remote Sensing Images Segmentation. arXiv preprint arXiv:2508.04022","author":"Wang Junyi","year":"2025","unstructured":"Junyi Wang, Jinjiang Li, Guodong Fan, Yakun Ju, Xiang Fang, and Alex C Kot. 2025 g. Prototype-Driven Structure Synergy Network for Remote Sensing Images Segmentation. arXiv preprint arXiv:2508.04022 (2025)."},{"key":"e_1_3_2_1_63_1","first-page":"35","volume-title":"Proceedings of the Great Lakes Symposium on VLSI","author":"Wang Siyi","year":"2025","unstructured":"Siyi Wang, Suman Dutta, Wei Jie Bryan Lee, Jerrie Feng, Xiang Fang, and Anupam Chattopadhyay. 2025a. Reducing T-Depth and T-Count in Quantum Multiplication Using Compressor Primitives. In Proceedings of the Great Lakes Symposium on VLSI 2025. 35-40."},{"key":"e_1_3_2_1_64_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2206.11215"},{"key":"e_1_3_2_1_65_1","volume-title":"Balance Act: Mitigating Hubness in Cross-Modal Retrieval with Query and Gallery Banks. In EMNLP.","author":"Wang Yimu","year":"2023","unstructured":"Yimu Wang, Xiangru Jian, and Bo Xue. 2023a. Balance Act: Mitigating Hubness in Cross-Modal Retrieval with Query and Gallery Banks. In EMNLP."},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2312.12886"},{"key":"e_1_3_2_1_67_1","unstructured":"Yimu Wang Evelien Riddell Adrian Chow Sean Sedwards and Krzysztof Czarnecki. 2025 h. Mitigating the Modality Gap: Few-Shot Out-of-Distribution Detection with Multi-modal Prototypes and Image Bias Estimation. arXiv:2502.00662 [cs.CV] https:\/\/arxiv.org\/abs\/2502.00662"},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"crossref","unstructured":"Yimu Wang and Peng Shi. 2023. Video-Text Retrieval by Supervised Sparse Multi-Grained Learning. In Findings EMNLP.","DOI":"10.18653\/v1\/2023.findings-emnlp.46"},{"key":"e_1_3_2_1_69_1","doi-asserted-by":"crossref","unstructured":"Yimu Wang Bo Xue Quan Cheng Yuhui Chen and Lijun Zhang. 2021. Deep Unified Cross-Modality Hashing by Pairwise Data Alignment. In IJCAI.","DOI":"10.24963\/ijcai.2021\/156"},{"key":"e_1_3_2_1_70_1","volume-title":"2025 i","author":"Wang Yimu","year":"2025","unstructured":"Yimu Wang, Shuai Yuan, Bo Xue, Xiangru Jian, Wei Pang, Mushi Wang, and Ning Yu. 2025 i. DREAM: Improving Video-Text Retrieval Through Relevance-Based Augmentation Using Large Foundation Models. In Proceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers), Luis Chiruzzo, Alan Ritter, and Lu Wang (Eds.). Association for Computational Linguistics, Albuquerque, New Mexico, 3037-3056. https:\/\/aclanthology.org\/2025.naacl-long.156\/"},{"key":"e_1_3_2_1_71_1","volume-title":"Proceedings of the International Conference on Machine Learning. 53366-53397","author":"Wu Shengqiong","year":"2024","unstructured":"Shengqiong Wu, Hao Fei, Leigang Qu, Wei Ji, and Tat-Seng Chua. 2024. NExT-GPT: Any-to-Any Multimodal LLM. In Proceedings of the International Conference on Machine Learning. 53366-53397."},{"volume-title":"Rethinking Video Sentence Grounding from a Tracking Perspective with Memory Network and Masked Attention","author":"Xiong Zeyu","key":"e_1_3_2_1_72_1","unstructured":"Zeyu Xiong, Daizong Liu, Xiang Fang, Xiaoye Qu, Jianfeng Dong, Jiahao Zhu, Keke Tang, and Pan Zhou. 2024. Rethinking Video Sentence Grounding from a Tracking Perspective with Memory Network and Masked Attention. In IEEE TMM."},{"key":"e_1_3_2_1_73_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2110.02334"},{"key":"e_1_3_2_1_74_1","unstructured":"Hai Yan Haijian Ma Xiaowen Cai Daizong Liu Zenghui Yuan Xiaoye Qu Jianfeng Dong Runwei Guan Xiang Fang Hongyang He Yulai Xie and Pan Zhou. 2025. Fit the Distribution: Cross-Image\/Prompt Adversarial Attacks on Multimodal Large Language Models. In Advances in Neural Information Processing Systems."},{"key":"e_1_3_2_1_75_1","first-page":"16442","article-title":"TubeDETR","author":"Yang Antoine","year":"2022","unstructured":"Antoine Yang, Antoine Miech, Josef Sivic, Ivan Laptev, and Cordelia Schmid. 2022. TubeDETR: Spatio-Temporal Video Grounding with Transformers. In CVPR. 16442-16453.","journal-title":"Spatio-Temporal Video Grounding with Transformers. In CVPR."},{"key":"e_1_3_2_1_76_1","volume-title":"EOOD: Entropy-based Out-of-distribution Detection. arXiv preprint arXiv:2504.03342","author":"Yang Guide","year":"2025","unstructured":"Guide Yang, Chao Hou, Weilong Peng, Xiang Fang, Yongwei Nie, Peican Zhu, and Keke Tang. 2025. EOOD: Entropy-based Out-of-distribution Detection. arXiv preprint arXiv:2504.03342 (2025)."},{"key":"e_1_3_2_1_77_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00811"},{"key":"e_1_3_2_1_78_1","first-page":"6832","volume-title":"VSLNet: End-to-End Learning for Video-Text Temporal Grounding. IEEE Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Zhang Hao","year":"2020","unstructured":"Hao Zhang, Aixin Gao, Bing Peng, Yu Sun, Zheng Yang, and Ram Nevatia. 2020a. VSLNet: End-to-End Learning for Video-Text Temporal Grounding. IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2020), 6832-6841."},{"key":"e_1_3_2_1_79_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2020.2973762"},{"key":"e_1_3_2_1_80_1","doi-asserted-by":"crossref","unstructured":"Songyang Zhang Houwen Peng Jianlong Fu and Jiebo Luo. 2020c. Learning 2d temporal adjacent networks for moment localization with natural language. In AAAI.","DOI":"10.1609\/aaai.v34i07.6984"},{"key":"e_1_3_2_1_81_1","doi-asserted-by":"crossref","unstructured":"X Zhang H Lei D Liu X Qu X Fang R Guan and K Jin. 2025a. Manipulating the Bounding Box: Multimodal Controlled Backdoor Attacks on 3D Visual Grounding Models. IJCNN.","DOI":"10.1109\/IJCNN64981.2025.11229253"},{"key":"e_1_3_2_1_82_1","doi-asserted-by":"crossref","unstructured":"X Zhang H Lei D Liu X Qu X Fang R Guan and K Jin. 2025b. MonoAttack: A Strong Attack Framework with Depth-Migration and Attribute-Tampering for Monocular 3D Object Detection. IJCNN.","DOI":"10.1109\/IJCNN64981.2025.11228111"},{"key":"e_1_3_2_1_83_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548324"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Dublin Ireland","acronym":"MM '25"},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3758179","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:15:06Z","timestamp":1765307706000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3758179"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":83,"alternative-id":["10.1145\/3746027.3758179","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3758179","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}