{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T13:17:47Z","timestamp":1788268667937,"version":"build-2803163510"},"publisher-location":"New York, NY, USA","reference-count":90,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,4,22]],"date-time":"2025-04-22T00:00:00Z","timestamp":1745280000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"name":"National Social Science Foundation Major Project in Art","award":["2024ZDE054"],"award-info":[{"award-number":["2024ZDE054"]}]},{"name":"University of Macau Start-up Research Grant","award":["SRG2024-00002-FST"],"award-info":[{"award-number":["SRG2024-00002-FST"]}]},{"name":"Multi-Year Research Grant","award":["MYRG-GRG2024-00077-FST-UMDF"],"award-info":[{"award-number":["MYRG-GRG2024-00077-FST-UMDF"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,4,22]]},"DOI":"10.1145\/3696410.3714788","type":"proceedings-article","created":{"date-parts":[[2025,4,22]],"date-time":"2025-04-22T18:57:28Z","timestamp":1745348248000},"page":"2341-2351","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":23,"title":["From Data Deluge to Data Curation: A Filtering-WoRA Paradigm for Efficient Text-based Person Search"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-3996-4011","authenticated-orcid":false,"given":"Jintao","family":"Sun","sequence":"first","affiliation":[{"name":"School of Computer Science and Technology, Beijing Institute of Technology, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3026-6347","authenticated-orcid":false,"given":"Hao","family":"Fei","sequence":"additional","affiliation":[{"name":"School of Computing, National University of Singapore, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-4185-4833","authenticated-orcid":false,"given":"Gangyi","family":"Ding","sequence":"additional","affiliation":[{"name":"School of Computer Science and Technology, Beijing Institute of Technology, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2434-9050","authenticated-orcid":false,"given":"Zhedong","family":"Zheng","sequence":"additional","affiliation":[{"name":"Faculty of Science and Technology, and Institute of Collaborative Innovation, University of Macau, Macau, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,4,22]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV45572.2020.9093640"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01022"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr.2018.00636"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2023\/62"},{"key":"e_1_3_2_1_5_1","volume-title":"An Empirical Study of CLIP for Text-based Person Search. arXiv preprint arXiv:2308.10045","author":"Cao Min","year":"2023","unstructured":"Min Cao, Yang Bai, Ziyin Zeng, Mang Ye, and Min Zhang. 2023. An Empirical Study of CLIP for Text-based Person Search. arXiv preprint arXiv:2308.10045 (2023)."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","unstructured":"Nicolas Carion Francisco Massa Gabriel Synnaeve Nicolas Usunier Alexander Kirillov and Sergey Zagoruyko. 2020. End-to-End Object Detection with Transformers. 213--229. https:\/\/doi.org\/10.1007\/978--3-030--58452--813","DOI":"10.1007\/978--3-030--58452--813"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1007\/978--3-030-01270-0_4"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV.2018.00208"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2022.04.081"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1007\/978--3-030--58577--8_7"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/cvprw50498.2020.00359"},{"key":"e_1_3_2_1_12_1","volume-title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. arxiv","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. arxiv: 1810.04805 [cs.CL]"},{"key":"e_1_3_2_1_13_1","volume-title":"Semantically Self-Aligned Network for Text-to-Image Part-aware Person Re-identification. arXiv: Computer Vision and Pattern Recognition,arXiv: Computer Vision and Pattern Recognition (Jul","author":"Ding Zefeng","year":"2021","unstructured":"Zefeng Ding, Changxing Ding, Zhiyin Shao, and Dacheng Tao. 2021. Semantically Self-Aligned Network for Text-to-Image Part-aware Person Re-identification. arXiv: Computer Vision and Pattern Recognition,arXiv: Computer Vision and Pattern Recognition (Jul 2021)."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i4.20370"},{"key":"e_1_3_2_1_15_1","volume-title":"Large-Scale Adversarial Training for Vision-and-Language Representation Learning. Neural Information Processing Systems,Neural Information Processing Systems (Jun","author":"Gan Zhe","year":"2020","unstructured":"Zhe Gan, Yen-Chun Chen, Pingqing Fu, Chen Zhu, Yu Cheng, and Jingjing Liu. 2020. Large-Scale Adversarial Training for Vision-and-Language Representation Learning. Neural Information Processing Systems,Neural Information Processing Systems (Jun 2020)."},{"key":"e_1_3_2_1_16_1","unstructured":"Chenyang Gao Guanyu Cai Xinyang Jiang Feng Zheng Jun Zhang Yifei Gong Pai Peng Xiaowei Guo and Xing Sun. 2021. Contextual Non-Local Alignment over Full-Scale Representation for Text-Based Person Search. arxiv: 2101.03036 [cs.CV]"},{"key":"e_1_3_2_1_17_1","volume-title":"Evaluating Appearance Models for Recognition, Reacquisition, and Tracking. (Jan","author":"Gray A.","year":"2007","unstructured":"DouglasA. Gray, Shane Brennan, and Hai Tao. 2007. Evaluating Appearance Models for Recognition, Reacquisition, and Tracking. (Jan 2007)."},{"key":"e_1_3_2_1_18_1","volume-title":"Text-Based Person Search with Limited Data. In British Machine Vision Conference. https:\/\/api.semanticscholar.org\/CorpusID:239050116","author":"Han Xiaoping","year":"2021","unstructured":"Xiaoping Han, Sen He, Li Zhang, and Tao Xiang. 2021. Text-Based Person Search with Limited Data. In British Machine Vision Conference. https:\/\/api.semanticscholar.org\/CorpusID:239050116"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2023.3337653"},{"key":"e_1_3_2_1_20_1","volume-title":"Parameter-Efficient Transfer Learning for NLP. arxiv","author":"Houlsby Neil","year":"1902","unstructured":"Neil Houlsby, Andrei Giurgiu, Stanislaw Jastrzebski, Bruna Morrone, Quentin de Laroussilhe, Andrea Gesmundo, Mona Attariyan, and Sylvain Gelly. 2019. Parameter-Efficient Transfer Learning for NLP. arxiv: 1902.00751 [cs.LG]"},{"key":"e_1_3_2_1_21_1","volume-title":"LoRA: Low-Rank Adaptation of Large Language Models. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=nZeVKeeFYf9","author":"Hu Edward J","year":"2022","unstructured":"Edward J Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, Lu Wang, and Weizhu Chen. 2022. LoRA: Low-Rank Adaptation of Large Language Models. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=nZeVKeeFYf9"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3589334.3645324"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","unstructured":"Zhicheng Huang Zhaoyang Zeng Yupan Huang Bei Liu Dongmei Fu and Jianlong Fu. 2021. Seeing Out of tHe bOx: End-to-End Pre-training for Vision-Language Representation Learning. 12971--12980. https:\/\/doi.org\/10.1109\/CVPR46437.2021.01278","DOI":"10.1109\/CVPR46437.2021.01278"},{"key":"e_1_3_2_1_24_1","volume-title":"Pixel-BERT: Aligning Image Pixels with Text by Deep Multi-Modal Transformers. arXiv: Computer Vision and Pattern Recognition,arXiv: Computer Vision and Pattern Recognition (Apr","author":"Huang Zhicheng","year":"2020","unstructured":"Zhicheng Huang, Zhaoyang Zeng, Bei Liu, Dongmei Fu, and Jianlong Fu. 2020. Pixel-BERT: Aligning Image Pixels with Text by Deep Multi-Modal Transformers. arXiv: Computer Vision and Pattern Recognition,arXiv: Computer Vision and Pattern Recognition (Apr 2020)."},{"key":"e_1_3_2_1_25_1","volume-title":"Cross-Modal Implicit Relation Reasoning and Aligning for Text-to-Image Person Retrieval. In IEEE International Conference on Computer Vision and Pattern Recognition (CVPR).","author":"Jiang Ding","year":"2023","unstructured":"Ding Jiang and Mang Ye. 2023. Cross-Modal Implicit Relation Reasoning and Aligning for Text-to-Image Person Retrieval. In IEEE International Conference on Computer Vision and Pattern Recognition (CVPR)."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr42600.2020.01028"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3589334.3645318"},{"key":"e_1_3_2_1_28_1","volume-title":"International conference on machine learning. PMLR, 5583--5594","author":"Kim Wonjae","year":"2021","unstructured":"Wonjae Kim, Bokyung Son, and Ildoo Kim. 2021. Vilt: Vision-and-language transformer without convolution or region supervision. In International conference on machine learning. PMLR, 5583--5594."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-016-0981--7"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01225-0_13"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6795"},{"key":"e_1_3_2_1_32_1","volume-title":"Proceedings of the 40th International Conference on Machine Learning (","author":"Li Junnan","year":"2023","unstructured":"Junnan Li, Dongxu Li, Silvio Savarese, and Steven Hoi. 2023. BLIP-2: bootstrapping language-image pre-training with frozen image encoders and large language models. In Proceedings of the 40th International Conference on Machine Learning (, Honolulu, Hawaii, USA,) (ICML'23). JMLR.org, Article 814, 13 pages."},{"key":"e_1_3_2_1_33_1","volume-title":"Proceedings of the 39th International Conference on Machine Learning (Proceedings of Machine Learning Research","volume":"12900","author":"Li Junnan","year":"2022","unstructured":"Junnan Li, Dongxu Li, Caiming Xiong, and Steven Hoi. 2022. BLIP: Bootstrapping Language-Image Pre-training for Unified Vision-Language Understanding and Generation. In Proceedings of the 39th International Conference on Machine Learning (Proceedings of Machine Learning Research, Vol. 162), Kamalika Chaudhuri, Stefanie Jegelka, Le Song, Csaba Szepesvari, Gang Niu, and Sivan Sabato (Eds.). PMLR, 12888--12900. https:\/\/proceedings.mlr.press\/v162\/li22n.html"},{"key":"e_1_3_2_1_34_1","volume-title":"Align before fuse: Vision and language representation learning with momentum distillation. Advances in neural information processing systems","author":"Li Junnan","year":"2021","unstructured":"Junnan Li, Ramprasaath Selvaraju, Akhilesh Gotmare, Shafiq Joty, Caiming Xiong, and Steven Chu Hong Hoi. 2021. Align before fuse: Vision and language representation learning with momentum distillation. Advances in neural information processing systems, Vol. 34 (2021), 9694--9705."},{"key":"e_1_3_2_1_35_1","volume-title":"VisualBERT: A Simple and Performant Baseline for Vision and Language","author":"Li LiunianHarold","year":"2019","unstructured":"LiunianHarold Li, Mark Yatskar, Dong Yin, Cho-Jui Hsieh, and Kai-Wei Chang. 2019. VisualBERT: A Simple and Performant Baseline for Vision and Language. Cornell University - arXiv,Cornell University - arXiv (Aug 2019)."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.209"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr.2017.551"},{"key":"e_1_3_2_1_38_1","volume-title":"Computer Vision--ACCV 2012: 11th Asian Conference on Computer Vision, Daejeon, Korea, November 5--9","author":"Li Wei","year":"2012","unstructured":"Wei Li, Rui Zhao, and Xiaogang Wang. 2013. Human reidentification with transferred metric learning. In Computer Vision--ACCV 2012: 11th Asian Conference on Computer Vision, Daejeon, Korea, November 5--9, 2012, Revised Selected Papers, Part I 11. Springer, 31--44."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr.2014.27"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1007\/978--3-030--58577--8_8"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","unstructured":"Tsung-Yi Lin Michael Maire Serge Belongie James Hays Pietro Perona Deva Ramanan Piotr Doll\u00e1r and C. Lawrence Zitnick. 2014. Microsoft COCO: Common Objects in Context. 740--755. https:\/\/doi.org\/10.1007\/978--3--319--10602--1_48","DOI":"10.1007\/978--3--319--10602--1_48"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3343031.3350991"},{"key":"e_1_3_2_1_43_1","volume-title":"Proceedings of the 41st International Conference on Machine Learning","author":"Liu Shih-Yang","year":"2024","unstructured":"Shih-Yang Liu, Chien-Yi Wang, Hongxu Yin, Pavlo Molchanov, Yu-Chiang Frank Wang, Kwang-Ting Cheng, and Min-Hung Chen. 2024. DoRA: weight-decomposed low-rank adaptation. In Proceedings of the 41st International Conference on Machine Learning (Vienna, Austria) (ICML'24). JMLR.org, Article 1299, 22 pages."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/iccv48922.2021.00986"},{"key":"e_1_3_2_1_46_1","volume-title":"Decoupled Weight Decay Regularization. In International Conference on Learning Representations. https:\/\/api.semanticscholar.org\/CorpusID:53592270","author":"Loshchilov Ilya","year":"2017","unstructured":"Ilya Loshchilov and Frank Hutter. 2017. Decoupled Weight Decay Regularization. In International Conference on Learning Representations. https:\/\/api.semanticscholar.org\/CorpusID:53592270"},{"key":"e_1_3_2_1_47_1","volume-title":"ViLBERT: Pretraining Task-Agnostic Visiolinguistic Representations for Vision-and-Language Tasks. Neural Information Processing Systems,Neural Information Processing Systems (Aug","author":"Lu Jing","year":"2019","unstructured":"Jing Lu, Dhruv Batra, Devi Parikh, and Stefan Lee. 2019. ViLBERT: Pretraining Task-Agnostic Visiolinguistic Representations for Vision-and-Language Tasks. Neural Information Processing Systems,Neural Information Processing Systems (Aug 2019)."},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2020.2984883"},{"key":"e_1_3_2_1_49_1","volume-title":"BEiT v2: Masked Image Modeling with Vector-Quantized Visual Tokenizers. ArXiv","author":"Peng Zhiliang","year":"2022","unstructured":"Zhiliang Peng, Li Dong, Hangbo Bao, Qixiang Ye, and Furu Wei. 2022. BEiT v2: Masked Image Modeling with Vector-Quantized Visual Tokenizers. ArXiv, Vol. abs\/2208.06366 (2022). https:\/\/api.semanticscholar.org\/CorpusID:251554649"},{"key":"e_1_3_2_1_50_1","volume-title":"SDXL: Improving Latent Diffusion Models for High-Resolution Image Synthesis. arxiv: 2307.01952 [cs.CV]","author":"Podell Dustin","year":"2023","unstructured":"Dustin Podell, Zion English, Kyle Lacey, Andreas Blattmann, Tim Dockhorn, Jonas M\u00fcller, Joe Penna, and Robin Rombach. 2023. SDXL: Improving Latent Diffusion Models for High-Resolution Image Synthesis. arxiv: 2307.01952 [cs.CV]"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00673"},{"key":"e_1_3_2_1_52_1","volume-title":"International conference on machine learning. PMLR, 8748--8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al. 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PMLR, 8748--8763."},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.13"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2016.2577031"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2021.3063681"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548028"},{"key":"e_1_3_2_1_58_1","volume-title":"European Conference on Computer Vision. Springer, 624--641","author":"Shu Xiujun","year":"2022","unstructured":"Xiujun Shu, Wei Wen, Haoqian Wu, Keyu Chen, Yiran Song, Ruizhi Qiao, Bo Ren, and Xiao Wang. 2022. See finer, see more: Implicit modality alignment for text-based person retrieval. In European Conference on Computer Vision. Springer, 624--641."},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1109\/AIMLA59606.2024.10531327"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1514"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2021\/148"},{"key":"e_1_3_2_1_62_1","volume-title":"OFA: Unifying Architectures, Tasks, and Modalities Through a Simple Sequence-to-Sequence Learning Framework. CoRR","author":"Wang Peng","year":"2022","unstructured":"Peng Wang, An Yang, Rui Men, Junyang Lin, Shuai Bai, Zhikang Li, Jianxin Ma, Chang Zhou, Jingren Zhou, and Hongxia Yang. 2022a. OFA: Unifying Architectures, Tasks, and Modalities Through a Simple Sequence-to-Sequence Learning Framework. CoRR, Vol. abs\/2202.03052 (2022)."},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCYB.2021.3130047"},{"key":"e_1_3_2_1_64_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2023.3256092"},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58610-2_24"},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548057"},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548166"},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"publisher","DOI":"10.1117\/1.JEI.29.4.043028"},{"key":"e_1_3_2_1_69_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1670"},{"key":"e_1_3_2_1_70_1","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr.2018.00016"},{"key":"e_1_3_2_1_71_1","unstructured":"Xindi Wu Byron Zhang Zhiwei Deng and Olga Russakovsky. 2024. Vision-Language Dataset Distillation. arxiv: 2308.07545 [cs.CV]"},{"key":"e_1_3_2_1_72_1","volume-title":"End-to-End Deep Learning for Person Search. arXiv: Computer Vision and Pattern Recognition,arXiv: Computer Vision and Pattern Recognition (Apr","author":"Xiao Tong","year":"2016","unstructured":"Tong Xiao, Shuang Li, Bochao Wang, Lin Li, and Xiaogang Wang. 2016. End-to-End Deep Learning for Person Search. arXiv: Computer Vision and Pattern Recognition,arXiv: Computer Vision and Pattern Recognition (Apr 2016)."},{"key":"e_1_3_2_1_73_1","doi-asserted-by":"publisher","unstructured":"Haiyang Xu Ming Yan Chenliang Li Bin Bi Songfang Huang Wenming Xiao and Fei Huang. 2021. E2E-VLP: End-to-End Vision-Language Pre-training Enhanced by Visual Learning. In Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers). https:\/\/doi.org\/10.18653\/v1\/2021.acl-long.42","DOI":"10.18653\/v1"},{"key":"e_1_3_2_1_74_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2023.3327924"},{"key":"e_1_3_2_1_75_1","volume-title":"Beyond Walking: A Large-Scale Image-Text Benchmark for Text-based Person Anomaly Search. arxiv: 2411.17776 [cs.CV] https:\/\/arxiv.org\/abs\/2411.17776","author":"Yang Shuyu","year":"2024","unstructured":"Shuyu Yang, Yaxiong Wang, Li Zhu, and Zhedong Zheng. 2024. Beyond Walking: A Large-Scale Image-Text Benchmark for Text-based Person Anomaly Search. arxiv: 2411.17776 [cs.CV] https:\/\/arxiv.org\/abs\/2411.17776"},{"key":"e_1_3_2_1_76_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3611709"},{"key":"e_1_3_2_1_77_1","volume-title":"Multi-Grained Vision Language Pre-Training: Aligning Texts with Visual Concepts. In International Conference on Machine Learning. https:\/\/api.semanticscholar.org\/CorpusID:244129883","author":"Zeng Yan","year":"2021","unstructured":"Yan Zeng, Xinsong Zhang, and Hang Li. 2021. Multi-Grained Vision Language Pre-Training: Aligning Texts with Visual Concepts. In International Conference on Machine Learning. https:\/\/api.semanticscholar.org\/CorpusID:244129883"},{"key":"e_1_3_2_1_78_1","volume-title":"X2-VLM: All-In-One Pre-trained Model For Vision-Language Tasks. arXiv:2211.12402","author":"Zeng Yan","year":"2022","unstructured":"Yan Zeng, Xinsong Zhang, Hang Li, Jiawei Wang, Jipeng Zhang, and Wangchunshu Zhou. 2022. X2-VLM: All-In-One Pre-trained Model For Vision-Language Tasks. arXiv:2211.12402 (2022)."},{"key":"e_1_3_2_1_79_1","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr46437.2021.00553"},{"key":"e_1_3_2_1_80_1","doi-asserted-by":"publisher","DOI":"10.1145\/3543507.3583301"},{"key":"e_1_3_2_1_81_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01246-5_42"},{"key":"e_1_3_2_1_82_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413864"},{"key":"e_1_3_2_1_83_1","volume-title":"Person Re-identification Meets Image Search. Tpami (Feb","author":"Zheng Liang","year":"2015","unstructured":"Liang Zheng, Liyue Shen, Lei Tian, Shengjin Wang, Jiahao Bu, and Qi Tian. 2015. Person Re-identification Meets Image Search. Tpami (Feb 2015)."},{"key":"e_1_3_2_1_84_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00224"},{"key":"e_1_3_2_1_85_1","doi-asserted-by":"publisher","DOI":"10.2307\/jj.12124947.4"},{"key":"e_1_3_2_1_86_1","doi-asserted-by":"publisher","DOI":"10.1145\/3383184"},{"key":"e_1_3_2_1_87_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.405"},{"key":"e_1_3_2_1_88_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.7000"},{"key":"e_1_3_2_1_89_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475369"},{"key":"e_1_3_2_1_90_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01630"}],"event":{"name":"WWW '25: The ACM Web Conference 2025","location":"Sydney NSW Australia","acronym":"WWW '25","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web"]},"container-title":["Proceedings of the ACM on Web Conference 2025"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3696410.3714788","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3696410.3714788","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T21:18:41Z","timestamp":1750281521000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3696410.3714788"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,4,22]]},"references-count":90,"alternative-id":["10.1145\/3696410.3714788","10.1145\/3696410"],"URL":"https:\/\/doi.org\/10.1145\/3696410.3714788","relation":{},"subject":[],"published":{"date-parts":[[2025,4,22]]},"assertion":[{"value":"2025-04-22","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}