{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T18:06:41Z","timestamp":1784138801973,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":84,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,19]],"date-time":"2026-07-19T00:00:00Z","timestamp":1784419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/legalcode"}],"funder":[{"name":"National Natural Science Foundation of China","award":["No. U24A20328"],"award-info":[{"award-number":["No. U24A20328"]}]},{"name":"National Natural Science Foundation of China","award":["No. 62476071"],"award-info":[{"award-number":["No. 62476071"]}]},{"name":"National Natural Science Foundation of China","award":["No. 62236003"],"award-info":[{"award-number":["No. 62236003"]}]},{"name":"National Natural Science Foundation of China","award":["No. 62276155"],"award-info":[{"award-number":["No. 62276155"]}]},{"name":"Guangdong Basic and Applied Basic Research Foundation","award":["No. 2025A1515011732"],"award-info":[{"award-number":["No. 2025A1515011732"]}]},{"name":"Key R&#x5c;&#x5c;&amp;D Program of Shandong Province &#x28;Major scientific and technological innovation projects&#x29;","award":["No. 2025CXGC020101"],"award-info":[{"award-number":["No. 2025CXGC020101"]}]},{"name":"China Postdoctoral Science Foundation","award":["No. 2025M784372"],"award-info":[{"award-number":["No. 2025M784372"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,20]]},"DOI":"10.1145\/3805712.3809704","type":"proceedings-article","created":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:06:26Z","timestamp":1784135186000},"page":"1800-1811","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["StructAlign: Structured Cross-Modal Alignment for Continual Text-to-Video Retrieval"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8945-1200","authenticated-orcid":false,"given":"Shaokun","family":"Wang","sequence":"first","affiliation":[{"name":"Harbin Institute of Technology, Shenzhen, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5658-5509","authenticated-orcid":false,"given":"Weili","family":"Guan","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology, Shenzhen, Shenzhen, China and Shenzhen Loop Area Institute, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-7804-3739","authenticated-orcid":false,"given":"Jizhou","family":"Han","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong University, Xi'an, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0247-5221","authenticated-orcid":false,"given":"Jianlong","family":"Wu","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology, Shenzhen, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5653-8286","authenticated-orcid":false,"given":"Yupeng","family":"Hu","sequence":"additional","affiliation":[{"name":"Shandong University, Jinan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1476-0273","authenticated-orcid":false,"given":"Liqiang","family":"Nie","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology, Shenzhen, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,19]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Class-Incremental Learning with Cross-Space Clustering and Controlled Transfer. In European Conference on Computer Vision. 105-122","author":"Ashok A.","unstructured":"A. Ashok, K. J. Joseph, and V. N. Balasubramanian. 2022. Class-Incremental Learning with Cross-Space Clustering and Controlled Transfer. In European Conference on Computer Vision. 105-122."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00175"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548184"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298698"},{"key":"e_1_3_2_1_5_1","first-page":"16532","article-title":"Online Fast Adaptation and Knowledge Accumulation (OSAKA): A New Approach to Continual Learning","volume":"33","author":"Caccia Massimo","year":"2020","unstructured":"Massimo Caccia, Pau Rodriguez, Oleksiy Ostapenko, Fabrice Normandin, Min Lin, Alexandre Lacoste, Yoshua Bengio, and Jan-Willem van de Meent. 2020. Online Fast Adaptation and Knowledge Accumulation (OSAKA): A New Approach to Continual Learning. Advances in Neural Information Processing Systems, Vol. 33 (2020), 16532-16545.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1613\/jair.1.17940"},{"key":"e_1_3_2_1_7_1","volume-title":"End-to-End Incremental Learning. In European Conference on Computer Vision. 233-248","author":"Castro Francisco M.","year":"2018","unstructured":"Francisco M. Castro, Manuel J. Mar\u00edn-Jim\u00e9nez, Nicolas Guil, Cordelia Schmid, and Karteek Alahari. 2018. End-to-End Incremental Learning. In European Conference on Computer Vision. 233-248."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01065"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10446924"},{"key":"e_1_3_2_1_10_1","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision. 11583-11593","author":"Croitoru Ioana","year":"2021","unstructured":"Ioana Croitoru, Simion-Vlad Bogolin, Marius Leordeanu, Hailin Jin, Andrew Zisserman, Samuel Albanie, and Yang Liu. 2021. TeachText: Cross-Modal Generalized Distillation for Text-Video Retrieval. In Proceedings of the IEEE\/CVF International Conference on Computer Vision. 11583-11593."},{"key":"e_1_3_2_1_11_1","volume-title":"Don't stop learning: Towards continual learning for the CLIP model. arXiv preprint arXiv:2207.09248","author":"Ding Yuxuan","year":"2022","unstructured":"Yuxuan Ding, Lingqiao Liu, Chunna Tian, Jingyuan Yang, and Haoxuan Ding. 2022. Don't stop learning: Towards continual learning for the CLIP model. arXiv preprint arXiv:2207.09248 (2022)."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548365"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01262"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2022.3227416"},{"key":"e_1_3_2_1_15_1","volume-title":"Multi-Modal Transformer for Video Retrieval. In European Conference on Computer Vision. 214-229","author":"Gabeur Valentin","year":"2020","unstructured":"Valentin Gabeur, Chen Sun, Karteek Alahari, and Cordelia Schmid. 2020. Multi-Modal Transformer for Video Retrieval. In European Conference on Computer Vision. 214-229."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00495"},{"key":"e_1_3_2_1_17_1","first-page":"4565","volume-title":"GOAL: Geometrically Optimal Alignment for Continual Generalized Category Discovery. In Proceedings of the AAAI Conference on Artificial Intelligence","volume":"40","author":"Han Jizhou","year":"2026","unstructured":"Jizhou Han, Chenhao Ding, SongLin Dong, Yuhang He, Shaokun Wang, Qiang Wang, and Yihong Gong. 2026 a. GOAL: Geometrically Optimal Alignment for Continual Generalized Category Discovery. In Proceedings of the AAAI Conference on Artificial Intelligence, Vol. 40. 4565-4573."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2025.3637903"},{"key":"e_1_3_2_1_19_1","volume-title":"Consistent Supervised-Unsupervised Alignment for Generalized Category Discovery. In The Thirty-ninth Annual Conference on Neural Information Processing Systems.","author":"Han Jizhou","year":"2025","unstructured":"Jizhou Han, Shaokun Wang, Yuhang He, Chenhao Ding, Qiang Wang, Xinyuan Gao, SongLin Dong, and Yihong Gong. 2025. Consistent Supervised-Unsupervised Alignment for Generalized Category Discovery. In The Thirty-ninth Annual Conference on Neural Information Processing Systems."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02241"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00092"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2022.3196092"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00395"},{"key":"e_1_3_2_1_24_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 21043-21052","author":"Huang Chao","year":"2023","unstructured":"Chao Huang, Lingxi Xie, Yuhang Yang, Wenxuan Wang, Bin Lin, and Deng Cai. 2023. Neural Collapse Inspired Federated Learning with Non-iid Data. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 21043-21052."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00730"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01088"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.02271"},{"key":"e_1_3_2_1_28_1","volume-title":"Proceedings of the ACM International Conference on Multimedia. 4747-4758","author":"Lao Mingrui","unstructured":"Mingrui Lao, Nan Pu, Yu Liu, Zhun Zhong, Erwin M. Bakker, Nicu Sebe, and Michael S. Lew. 2023. Multi-Domain Lifelong Visual Question Answering via Self-Critical Distillation. In Proceedings of the ACM International Conference on Multimedia. 4747-4758."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.02844"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681024"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2017.2773081"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2022.3219605"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3726302.3729984"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3209978.3210003"},{"key":"e_1_3_2_1_35_1","first-page":"1","article-title":"Pre-train, Prompt, and Predict","volume":"55","author":"Liu Peng","year":"2023","unstructured":"Peng Liu, Weizhen Yuan, Jing Fu, Weizhu Xiong, Xiang Jiang, Hiroaki Hayashi, and Graham Neubig. 2023. Pre-train, Prompt, and Predict: A Systematic Survey of Prompting Methods in Natural Language Processing. Comput. Surveys, Vol. 55, 9 (2023), 1-35.","journal-title":"Comput. Surveys"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00257"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19781-9_19"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2022.07.028"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547910"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00810"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.2015509117"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3746027.3754846"},{"key":"e_1_3_2_1_43_1","volume-title":"International Conference on Machine Learning. 8748-8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al., 2021. Learning transferable visual models from natural language supervision. In International Conference on Machine Learning. 8748-8763."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01834"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01146"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3611940"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2021.3090595"},{"key":"e_1_3_2_1_48_1","volume-title":"Learning Endogenous Attention for Incremental Object Detection. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 30354-30364","author":"Song Xiang","year":"2025","unstructured":"Xiang Song, Yuhang He, Jingyuan Li, Qiang Wang, and Yihong Gong. 2025. Learning Endogenous Attention for Incremental Object Detection. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 30354-30364."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCDS.2020.2973280"},{"key":"e_1_3_2_1_50_1","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision. 1706-1716","author":"Tang Y. M.","unstructured":"Y. M. Tang, Y. X. Peng, and W. S. Zheng. 2023. When Prompt-Based Incremental Learning Does Not Meet Strong Pretraining. In Proceedings of the IEEE\/CVF International Conference on Computer Vision. 1706-1716."},{"key":"e_1_3_2_1_51_1","volume-title":"Topology-Preserving Class-Incremental Learning. In European Conference on Computer Vision. 254-270","author":"Tao Xiaoyu","year":"2020","unstructured":"Xiaoyu Tao, Xinyuan Chang, Xiaopeng Hong, Songlin Dong, Xing Wei, and Yihong Gong. 2020a. Topology-Preserving Class-Incremental Learning. In European Conference on Computer Vision. 254-270."},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i04.6060"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01622"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3680774"},{"key":"e_1_3_2_1_55_1","volume-title":"European Conference on Computer Vision. 144-162","author":"Wang Qiang","year":"2024","unstructured":"Qiang Wang, Yuhang He, Songlin Dong, Xinyuan Gao, Shaokun Wang, and Yihong Gong. 2024a. Non-exemplar domain incremental learning via cross-domain concept integration. In European Conference on Computer Vision. 144-162."},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i20.35418"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00456"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2023.3262739"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3611926"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00504"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.52202\/068431-0411"},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00264"},{"key":"e_1_3_2_1_63_1","volume-title":"DualPrompt: Complementary Prompting for Rehearsal-Free Continual Learning. In European Conference on Computer Vision. 631-648","author":"Wang Zifeng","year":"2022","unstructured":"Zifeng Wang, Zizhao Zhang, Sayna Ebrahimi, Ruoxi Sun, Han Zhang, Chen-Yu Lee, Xiaoqi Ren, Guolong Su, Vincent Perot, Jennifer Dy, et al., 2022b. DualPrompt: Complementary Prompting for Rehearsal-Free Continual Learning. In European Conference on Computer Vision. 631-648."},{"key":"e_1_3_2_1_64_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00024"},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00116"},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00046"},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.571"},{"key":"e_1_3_2_1_68_1","volume-title":"The International Conference on Learning Representations.","author":"Xue Hongwei","year":"2023","unstructured":"Hongwei Xue, Yuchong Sun, Bei Liu, Jianlong Fu, Ruihua Song, Houqiang Li, and Jiebo Luo. 2023. CLIP-ViP: Adapting Pre-trained Image-Text Model to Video-Language Alignment. In The International Conference on Learning Representations."},{"key":"e_1_3_2_1_69_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00303"},{"key":"e_1_3_2_1_70_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681264"},{"key":"e_1_3_2_1_71_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2022.3175849"},{"key":"e_1_3_2_1_72_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01136"},{"key":"e_1_3_2_1_73_1","volume-title":"Recent Advances of Multimodal Continual Learning: A Comprehensive Survey. arXiv preprint arXiv:2410.05352","author":"Yu Dianzhi","year":"2024","unstructured":"Dianzhi Yu, Xinni Zhang, Yankai Chen, Aiwei Liu, Yifei Zhang, Philip S. Yu, and Irwin King. 2024a. Recent Advances of Multimodal Continual Learning: A Comprehensive Survey. arXiv preprint arXiv:2410.05352 (2024)."},{"key":"e_1_3_2_1_74_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02191"},{"key":"e_1_3_2_1_75_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01234-2_29"},{"key":"e_1_3_2_1_76_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.02054"},{"key":"e_1_3_2_1_77_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3680839"},{"key":"e_1_3_2_1_78_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01831"},{"key":"e_1_3_2_1_79_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3133897"},{"key":"e_1_3_2_1_80_1","doi-asserted-by":"publisher","DOI":"10.1145\/3477495.3531950"},{"key":"e_1_3_2_1_81_1","doi-asserted-by":"publisher","DOI":"10.1145\/3726302.3729936"},{"key":"e_1_3_2_1_82_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01752"},{"key":"e_1_3_2_1_83_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01873"},{"key":"e_1_3_2_1_84_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02223"}],"event":{"name":"SIGIR '26: The 49th International ACM SIGIR Conference on Research and Development in Information Retrieval","location":"Melbourne VIC Australia","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval"]},"container-title":["Proceedings of the 49th International ACM SIGIR Conference on Research and Development in Information Retrieval"],"original-title":[],"deposited":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T17:15:55Z","timestamp":1784135755000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805712.3809704"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,19]]},"references-count":84,"alternative-id":["10.1145\/3805712.3809704","10.1145\/3805712"],"URL":"https:\/\/doi.org\/10.1145\/3805712.3809704","relation":{},"subject":[],"published":{"date-parts":[[2026,7,19]]},"assertion":[{"value":"2026-07-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}