{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,28]],"date-time":"2026-07-28T23:54:35Z","timestamp":1785282875125,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":63,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,10,10]],"date-time":"2022-10-10T00:00:00Z","timestamp":1665360000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"the Special Fund of Hubei Luojia Laboratory award number(s)","award":["220100015"],"award-info":[{"award-number":["220100015"]}]},{"name":"the Bingtuan Science and Technology Program award number(s)","award":["No. 2019BC008"],"award-info":[{"award-number":["No. 2019BC008"]}]},{"name":"the Key Research and Development Program of Hubei Province award number(s)","award":["2021BAA187"],"award-info":[{"award-number":["2021BAA187"]}]},{"name":"the National Natural Science Foundation of China award number(s)","award":["62176188"],"award-info":[{"award-number":["62176188"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,10,10]]},"DOI":"10.1145\/3503161.3547799","type":"proceedings-article","created":{"date-parts":[[2022,10,10]],"date-time":"2022-10-10T15:43:01Z","timestamp":1665416581000},"page":"2565-2574","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":42,"title":["Rotation Invariant Transformer for Recognizing Object in UAVs"],"prefix":"10.1145","author":[{"given":"Shuoyi","family":"Chen","sequence":"first","affiliation":[{"name":"Wuhan University, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mang","family":"Ye","sequence":"additional","affiliation":[{"name":"Wuhan University &amp; Hubei Luojia Laboratory, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bo","family":"Du","sequence":"additional","affiliation":[{"name":"Wuhan University &amp; Hubei Luojia Laboratory, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2022,10,10]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"Why do deep convolutional networks generalize so poorly to small image transformations? arXiv preprint arXiv:1805.12177","author":"Azulay Aharon","year":"2018","unstructured":"Aharon Azulay and Yair Weiss . 2018. Why do deep convolutional networks generalize so poorly to small image transformations? arXiv preprint arXiv:1805.12177 ( 2018 ). Aharon Azulay and Yair Weiss. 2018. Why do deep convolutional networks generalize so poorly to small image transformations? arXiv preprint arXiv:1805.12177 (2018)."},{"key":"e_1_3_2_2_2_1","first-page":"2352","article-title":"Structure-Aware Positional Transformer for Visible-Infrared Person Re-Identification","volume":"31","author":"Chen Cuiqun","year":"2022","unstructured":"Cuiqun Chen , Mang Ye , Meibin Qi , Jingjing Wu , Jianguo Jiang , and Chia-Wen Lin . 2022 . Structure-Aware Positional Transformer for Visible-Infrared Person Re-Identification . IEEE TIP 31 (2022), 2352 -- 2364 . Cuiqun Chen, Mang Ye, Meibin Qi, Jingjing Wu, Jianguo Jiang, and Chia-Wen Lin. 2022. Structure-Aware Positional Transformer for Visible-Infrared Person Re-Identification. IEEE TIP 31 (2022), 2352--2364.","journal-title":"IEEE TIP"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"crossref","unstructured":"Guangyi Chen Chunze Lin Liangliang Ren Jiwen Lu and Jie Zhou. 2019. Selfcritical attention learning for person re-identification. In ICCV. 9637--9646.  Guangyi Chen Chunze Lin Liangliang Ren Jiwen Lu and Jie Zhou. 2019. Selfcritical attention learning for person re-identification. In ICCV. 9637--9646.","DOI":"10.1109\/ICCV.2019.00973"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"crossref","unstructured":"Xuesong Chen Canmiao Fu Yong Zhao Feng Zheng Jingkuan Song Rongrong Ji and Yi Yang. 2020. Salience-guided cascaded suppression network for person re-identification. In CVPR. 3300--3310.  Xuesong Chen Canmiao Fu Yong Zhao Feng Zheng Jingkuan Song Rongrong Ji and Yi Yang. 2020. Salience-guided cascaded suppression network for person re-identification. In CVPR. 3300--3310.","DOI":"10.1109\/CVPR42600.2020.00336"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"crossref","unstructured":"Yehansen Chen Lin Wan Zhihang Li Qianyan Jing and Zongyuan Sun. 2021. Neural Feature Search for RGB-Infrared Person Re-Identification. In CVPR. 587-- 597.  Yehansen Chen Lin Wan Zhihang Li Qianyan Jing and Zongyuan Sun. 2021. Neural Feature Search for RGB-Infrared Person Re-Identification. In CVPR. 587-- 597.","DOI":"10.1109\/CVPR46437.2021.00065"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2017.2666805"},{"key":"e_1_3_2_2_7_1","volume-title":"Rifd-cnn: Rotation-invariant and fisher discriminative convolutional neural networks for object detection. In CVPR. 2884--2893.","author":"Cheng Gong","year":"2016","unstructured":"Gong Cheng , Peicheng Zhou , and Junwei Han . 2016 . Rifd-cnn: Rotation-invariant and fisher discriminative convolutional neural networks for object detection. In CVPR. 2884--2893. Gong Cheng, Peicheng Zhou, and Junwei Han. 2016. Rifd-cnn: Rotation-invariant and fisher discriminative convolutional neural networks for object detection. In CVPR. 2884--2893."},{"key":"e_1_3_2_2_8_1","volume-title":"Relationnet: Bridging visual representations for object detection via transformer decoder. arXiv preprint arXiv:2010.15831","author":"Chi Cheng","year":"2020","unstructured":"Cheng Chi , Fangyun Wei , and Han Hu . 2020 . Relationnet: Bridging visual representations for object detection via transformer decoder. arXiv preprint arXiv:2010.15831 (2020). Cheng Chi, Fangyun Wei, and Han Hu. 2020. Relationnet: Bridging visual representations for object detection via transformer decoder. arXiv preprint arXiv:2010.15831 (2020)."},{"key":"e_1_3_2_2_9_1","volume-title":"Jeffrey De Fauw, and Koray Kavukcuoglu","author":"Dieleman Sander","year":"2016","unstructured":"Sander Dieleman , Jeffrey De Fauw, and Koray Kavukcuoglu . 2016 . Exploiting cyclic symmetry in convolutional neural networks. In ICML. 1889--1898. Sander Dieleman, Jeffrey De Fauw, and Koray Kavukcuoglu. 2016. Exploiting cyclic symmetry in convolutional neural networks. In ICML. 1889--1898."},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"crossref","unstructured":"Jian Ding Nan Xue Yang Long Gui-Song Xia and Qikai Lu. 2019. Learning RoI transformer for oriented object detection in aerial images. In CVPR. 2849--2858.  Jian Ding Nan Xue Yang Long Gui-Song Xia and Qikai Lu. 2019. Learning RoI transformer for oriented object detection in aerial images. In CVPR. 2849--2858.","DOI":"10.1109\/CVPR.2019.00296"},{"key":"e_1_3_2_2_11_1","unstructured":"Alexey Dosovitskiy Lucas Beyer Alexander Kolesnikov Dirk Weissenborn Xiaohua Zhai Thomas Unterthiner Mostafa Dehghani Matthias Minderer Georg Heigold Sylvain Gelly etal 2020. An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020).  Alexey Dosovitskiy Lucas Beyer Alexander Kolesnikov Dirk Weissenborn Xiaohua Zhai Thomas Unterthiner Mostafa Dehghani Matthias Minderer Georg Heigold Sylvain Gelly et al. 2020. An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)."},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"crossref","unstructured":"Lijie Fan Tianhong Li Rongyao Fang Rumen Hristov Yuan Yuan and Dina Katabi. 2020. Learning longterm representations for person re-identification using radio signals. In CVPR. 10699--10709.  Lijie Fan Tianhong Li Rongyao Fang Rumen Hristov Yuan Yuan and Dina Katabi. 2020. Learning longterm representations for person re-identification using radio signals. In CVPR. 10699--10709.","DOI":"10.1109\/CVPR42600.2020.01071"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1186\/s13634-019-0647-z"},{"key":"e_1_3_2_2_14_1","volume-title":"Redet: A rotationequivariant detector for aerial object detection. In CVPR. 2786--2795.","author":"Han Jiaming","year":"2021","unstructured":"Jiaming Han , Jian Ding , Nan Xue , and Gui-Song Xia . 2021 . Redet: A rotationequivariant detector for aerial object detection. In CVPR. 2786--2795. Jiaming Han, Jian Ding, Nan Xue, and Gui-Song Xia. 2021. Redet: A rotationequivariant detector for aerial object detection. In CVPR. 2786--2795."},{"key":"e_1_3_2_2_15_1","volume-title":"Transreid: Transformer-based object re-identification. In ICCV. 15013--15022.","author":"He Shuting","year":"2021","unstructured":"Shuting He , Hao Luo , Pichao Wang , Fan Wang , Hao Li , and Wei Jiang . 2021 . Transreid: Transformer-based object re-identification. In ICCV. 15013--15022. Shuting He, Hao Luo, Pichao Wang, Fan Wang, Hao Li, and Wei Jiang. 2021. Transreid: Transformer-based object re-identification. In ICCV. 15013--15022."},{"key":"e_1_3_2_2_16_1","unstructured":"Alexander Hermans Lucas Beyer and Bastian Leibe. 2017. In Defense of the Triplet Loss for Person Re-Identification. arXiv preprint arXiv:1703.07737 (2017).  Alexander Hermans Lucas Beyer and Bastian Leibe. 2017. In Defense of the Triplet Loss for Person Re-Identification. arXiv preprint arXiv:1703.07737 (2017)."},{"key":"e_1_3_2_2_17_1","unstructured":"Max Jaderberg Karen Simonyan Andrew Zisserman etal 2015. Spatial transformer networks. Advances in neural information processing systems 28 (2015) 2017--2025.  Max Jaderberg Karen Simonyan Andrew Zisserman et al. 2015. Spatial transformer networks. Advances in neural information processing systems 28 (2015) 2017--2025."},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"crossref","unstructured":"Xin Jin Cuiling Lan Wenjun Zeng Zhibo Chen and Li Zhang. 2020. Style normalization and restitution for generalizable person re-identification. In CVPR. 3143--3152.  Xin Jin Cuiling Lan Wenjun Zeng Zhibo Chen and Li Zhang. 2020. Style normalization and restitution for generalizable person re-identification. In CVPR. 3143--3152.","DOI":"10.1109\/CVPR42600.2020.00321"},{"key":"e_1_3_2_2_19_1","volume-title":"The P-DESTRE: a fully annotated dataset for pedestrian detection, tracking, reidentification and search from aerial devices. arXiv preprint arXiv:2004.02782","author":"Kumar SV","year":"2020","unstructured":"SV Kumar , Ehsan Yaghoubi , Abhijit Das , BS Harish , and Hugo Proen\u00e7a . 2020. The P-DESTRE: a fully annotated dataset for pedestrian detection, tracking, reidentification and search from aerial devices. arXiv preprint arXiv:2004.02782 ( 2020 ). SV Kumar, Ehsan Yaghoubi, Abhijit Das, BS Harish, and Hugo Proen\u00e7a. 2020. The P-DESTRE: a fully annotated dataset for pedestrian detection, tracking, reidentification and search from aerial devices. arXiv preprint arXiv:2004.02782 (2020)."},{"key":"e_1_3_2_2_20_1","volume-title":"Weperson: Learning a generalized reidentification model from all-weather virtual data. In ACM MM. 3115--3123.","author":"Li He","year":"2021","unstructured":"He Li , Mang Ye , and Bo Du . 2021 . Weperson: Learning a generalized reidentification model from all-weather virtual data. In ACM MM. 3115--3123. He Li, Mang Ye, and Bo Du. 2021. Weperson: Learning a generalized reidentification model from all-weather virtual data. In ACM MM. 3115--3123."},{"key":"e_1_3_2_2_21_1","unstructured":"Tianjiao Li Jun Liu Wei Zhang Yun Ni Wenqian Wang and Zhiheng Li. 2021. UAV-Human: A Large Benchmark for Human Behavior Understanding with Unmanned Aerial Vehicles. In CVPR. 16266--16275.  Tianjiao Li Jun Liu Wei Zhang Yun Ni Wenqian Wang and Zhiheng Li. 2021. UAV-Human: A Large Benchmark for Human Behavior Understanding with Unmanned Aerial Vehicles. In CVPR. 16266--16275."},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"crossref","unstructured":"Wei Li Xiatian Zhu and Shaogang Gong. 2018. Harmonious attention network for person re-identification. In CVPR. 2285--2294.  Wei Li Xiatian Zhu and Shaogang Gong. 2018. Harmonious attention network for person re-identification. In CVPR. 2285--2294.","DOI":"10.1109\/CVPR.2018.00243"},{"key":"e_1_3_2_2_23_1","volume-title":"Swin transformer: Hierarchical vision transformer using shifted windows. arXiv preprint arXiv:2103.14030","author":"Liu Ze","year":"2021","unstructured":"Ze Liu , Yutong Lin , Yue Cao , Han Hu , Yixuan Wei , Zheng Zhang , Stephen Lin , and Baining Guo . 2021. Swin transformer: Hierarchical vision transformer using shifted windows. arXiv preprint arXiv:2103.14030 ( 2021 ). Ze Liu, Yutong Lin, Yue Cao, Han Hu, Yixuan Wei, Zheng Zhang, Stephen Lin, and Baining Guo. 2021. Swin transformer: Hierarchical vision transformer using shifted windows. arXiv preprint arXiv:2103.14030 (2021)."},{"key":"e_1_3_2_2_24_1","first-page":"2597","article-title":"A strong baseline and batch normalization neck for deep person re-identification","volume":"22","author":"Luo Hao","year":"2019","unstructured":"Hao Luo , Wei Jiang , Youzhi Gu , Fuxu Liu , Xingyu Liao , Shenqi Lai , and Jianyang Gu . 2019 . A strong baseline and batch normalization neck for deep person re-identification . IEEE TMM 22 , 10 (2019), 2597 -- 2609 . Hao Luo, Wei Jiang, Youzhi Gu, Fuxu Liu, Xingyu Liao, Shenqi Lai, and Jianyang Gu. 2019. A strong baseline and batch normalization neck for deep person re-identification. IEEE TMM 22, 10 (2019), 2597--2609.","journal-title":"IEEE TMM"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"crossref","unstructured":"Dechao Meng Liang Li Xuejing Liu Yadong Li Shijie Yang Zheng-Jun Zha Xingyu Gao Shuhui Wang and Qingming Huang. 2020. Parsing-based viewaware embedding network for vehicle re-identification. In CVPR. 7103--7112.  Dechao Meng Liang Li Xuejing Liu Yadong Li Shijie Yang Zheng-Jun Zha Xingyu Gao Shuhui Wang and Qingming Huang. 2020. Parsing-based viewaware embedding network for vehicle re-identification. In CVPR. 7103--7112.","DOI":"10.1109\/CVPR42600.2020.00713"},{"key":"e_1_3_2_2_26_1","volume-title":"Fahad Shahbaz Khan, and Ming-Hsuan Yang","author":"Naseer Muzammal","year":"2021","unstructured":"Muzammal Naseer , Kanchana Ranasinghe , Salman Khan , Munawar Hayat , Fahad Shahbaz Khan, and Ming-Hsuan Yang . 2021 . Intriguing Properties of Vision Transformers . arXiv preprint arXiv:2105.10497 (2021). Muzammal Naseer, Kanchana Ranasinghe, Salman Khan, Munawar Hayat, Fahad Shahbaz Khan, and Ming-Hsuan Yang. 2021. Intriguing Properties of Vision Transformers. arXiv preprint arXiv:2105.10497 (2021)."},{"key":"e_1_3_2_2_27_1","volume-title":"The Multi-Modal Video Reasoning and Analyzing Competition. In ICCV Workshop. 806--813","author":"Peng Haoran","year":"2021","unstructured":"Haoran Peng , He Huang , Li Xu , Tianjiao Li , Jun Liu , Hossein Rahmani , Qiuhong Ke , Zhicheng Guo , Cong Wu , Rongchang Li , 2021 . The Multi-Modal Video Reasoning and Analyzing Competition. In ICCV Workshop. 806--813 . Haoran Peng, He Huang, Li Xu, Tianjiao Li, Jun Liu, Hossein Rahmani, Qiuhong Ke, Zhicheng Guo, Cong Wu, Rongchang Li, et al. 2021. The Multi-Modal Video Reasoning and Analyzing Competition. In ICCV Workshop. 806--813."},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"crossref","unstructured":"Nan Pu Wei Chen Yu Liu Erwin M Bakker and Michael S Lew. 2021. Lifelong Person Re-Identification via Adaptive Knowledge Accumulation. In CVPR. 7901-- 7910.  Nan Pu Wei Chen Yu Liu Erwin M Bakker and Michael S Lew. 2021. Lifelong Person Re-Identification via Adaptive Knowledge Accumulation. In CVPR. 7901-- 7910.","DOI":"10.1109\/CVPR46437.2021.00781"},{"key":"e_1_3_2_2_29_1","volume-title":"Person Re- Identification with a Locally Aware Transformer. arXiv preprint arXiv:2106.03720","author":"Sharma Charu","year":"2021","unstructured":"Charu Sharma , Siddhant R Kapil , and David Chapman . 2021. Person Re- Identification with a Locally Aware Transformer. arXiv preprint arXiv:2106.03720 ( 2021 ). Charu Sharma, Siddhant R Kapil, and David Chapman. 2021. Person Re- Identification with a Locally Aware Transformer. arXiv preprint arXiv:2106.03720 (2021)."},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"crossref","unstructured":"Yantao Shen Tong Xiao Hongsheng Li Shuai Yi and Xiaogang Wang. 2018. End-to-end deep kronecker-product matching for person re-identification. In CVPR. 6886--6895.  Yantao Shen Tong Xiao Hongsheng Li Shuai Yi and Xiaogang Wang. 2018. End-to-end deep kronecker-product matching for person re-identification. In CVPR. 6886--6895.","DOI":"10.1109\/CVPR.2018.00720"},{"key":"e_1_3_2_2_31_1","unstructured":"Jianlou Si Honggang Zhang Chun-Guang Li Jason Kuen Xiangfei Kong Alex C Kot and Gang Wang. 2018. Dual attention matching network for context-aware feature sequence based person re-identification. In CVPR. 5363--5372.  Jianlou Si Honggang Zhang Chun-Guang Li Jason Kuen Xiangfei Kong Alex C Kot and Gang Wang. 2018. Dual attention matching network for context-aware feature sequence based person re-identification. In CVPR. 5363--5372."},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"crossref","unstructured":"Yifan Sun Liang Zheng Yi Yang Qi Tian and ShengjinWang. 2018. Beyond part models: Person retrieval with refined part pooling (and a strong convolutional baseline). In ECCV. 480--496.  Yifan Sun Liang Zheng Yi Yang Qi Tian and ShengjinWang. 2018. Beyond part models: Person retrieval with refined part pooling (and a strong convolutional baseline). In ECCV. 480--496.","DOI":"10.1007\/978-3-030-01225-0_30"},{"key":"e_1_3_2_2_33_1","unstructured":"Hugo Touvron Matthieu Cord Matthijs Douze Francisco Massa Alexandre Sablayrolles and Herv\u00e9 J\u00e9gou. 2021. Training data-efficient image transformers & distillation through attention. In ICML. 10347--10357.  Hugo Touvron Matthieu Cord Matthijs Douze Francisco Massa Alexandre Sablayrolles and Herv\u00e9 J\u00e9gou. 2021. Training data-efficient image transformers & distillation through attention. In ICML. 10347--10357."},{"key":"e_1_3_2_2_34_1","volume-title":"Mancs: A multi-task attentional network with curriculum sampling for person re-identification. In ECCV. 365--381.","author":"Wang Cheng","year":"2018","unstructured":"Cheng Wang , Qian Zhang , Chang Huang , Wenyu Liu , and Xinggang Wang . 2018 . Mancs: A multi-task attentional network with curriculum sampling for person re-identification. In ECCV. 365--381. Cheng Wang, Qian Zhang, Chang Huang, Wenyu Liu, and Xinggang Wang. 2018. Mancs: A multi-task attentional network with curriculum sampling for person re-identification. In ECCV. 365--381."},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"crossref","unstructured":"Guan'an Wang Shuo Yang Huanyu Liu Zhicheng Wang Yang Yang Shuliang Wang Gang Yu Erjin Zhou and Jian Sun. 2020. High-order information matters: Learning relation and topology for occluded person re-identification. In CVPR. 6449--6458.  Guan'an Wang Shuo Yang Huanyu Liu Zhicheng Wang Yang Yang Shuliang Wang Gang Yu Erjin Zhou and Jian Sun. 2020. High-order information matters: Learning relation and topology for occluded person re-identification. In CVPR. 6449--6458.","DOI":"10.1109\/CVPR42600.2020.00648"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"crossref","unstructured":"GuanshuoWang Yufeng Yuan Xiong Chen Jiwei Li and Xi Zhou. 2018. Learning discriminative features with multiple granularities for person re-identification. In ACM MM. 274--282.  GuanshuoWang Yufeng Yuan Xiong Chen Jiwei Li and Xi Zhou. 2018. Learning discriminative features with multiple granularities for person re-identification. In ACM MM. 274--282.","DOI":"10.1145\/3240508.3240552"},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"crossref","unstructured":"Peng Wang Bingliang Jiao Lu Yang Yifei Yang Shizhou Zhang Wei Wei and Yanning Zhang. 2019. Vehicle re-identification in aerial imagery: Dataset and approach. In ICCV. 460--469.  Peng Wang Bingliang Jiao Lu Yang Yifei Yang Shizhou Zhang Wei Wei and Yanning Zhang. 2019. Vehicle re-identification in aerial imagery: Dataset and approach. In ICCV. 460--469.","DOI":"10.1109\/ICCV.2019.00055"},{"key":"e_1_3_2_2_38_1","volume-title":"Pose-guided Feature Disentangling for Occluded Person Re-identification Based on Transformer. AAAI","author":"Wang Tao","year":"2021","unstructured":"Tao Wang , Hong Liu , Pinhao Song , Tianyu Guo , and Wei Shi . 2021. Pose-guided Feature Disentangling for Occluded Person Re-identification Based on Transformer. AAAI ( 2021 ). Tao Wang, Hong Liu, Pinhao Song, Tianyu Guo, and Wei Shi. 2021. Pose-guided Feature Disentangling for Occluded Person Re-identification Based on Transformer. AAAI (2021)."},{"key":"e_1_3_2_2_39_1","volume-title":"Pyramid vision transformer: A versatile backbone for dense prediction without convolutions. arXiv preprint arXiv:2102.12122","author":"Xie Enze","year":"2021","unstructured":"WenhaiWang, Enze Xie , Xiang Li , Deng-Ping Fan , Kaitao Song , Ding Liang , Tong Lu , Ping Luo , and Ling Shao . 2021. Pyramid vision transformer: A versatile backbone for dense prediction without convolutions. arXiv preprint arXiv:2102.12122 ( 2021 ). WenhaiWang, Enze Xie, Xiang Li, Deng-Ping Fan, Kaitao Song, Ding Liang, Tong Lu, Ping Luo, and Ling Shao. 2021. Pyramid vision transformer: A versatile backbone for dense prediction without convolutions. arXiv preprint arXiv:2102.12122 (2021)."},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"crossref","unstructured":"XiaolongWang Ross Girshick Abhinav Gupta and Kaiming He. 2018. Non-local neural networks. In CVPR. 7794--7803.  XiaolongWang Ross Girshick Abhinav Gupta and Kaiming He. 2018. Non-local neural networks. In CVPR. 7794--7803.","DOI":"10.1109\/CVPR.2018.00813"},{"key":"e_1_3_2_2_41_1","unstructured":"Longhui Wei Shiliang Zhang Wen Gao and Qi Tian. 2018. Person transfer gan to bridge domain gap for person re-identification. In CVPR. 79--88.  Longhui Wei Shiliang Zhang Wen Gao and Qi Tian. 2018. Person transfer gan to bridge domain gap for person re-identification. In CVPR. 79--88."},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"crossref","unstructured":"Tong Xiao Hongsheng Li Wanli Ouyang and Xiaogang Wang. 2016. Learning deep feature representations with domain guided dropout for person reidentification. In CVPR. 1249--1258.  Tong Xiao Hongsheng Li Wanli Ouyang and Xiaogang Wang. 2016. Learning deep feature representations with domain guided dropout for person reidentification. In CVPR. 1249--1258.","DOI":"10.1109\/CVPR.2016.140"},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2018.08.015"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-022-01593-w"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i4.16426"},{"key":"e_1_3_2_2_46_1","unstructured":"Xue Yang Junchi Yan Qi Ming Wentao Wang Xiaopeng Zhang and Qi Tian. 2021. Rethinking rotated object detection with gaussian wasserstein distance loss. In ICML. 11830--11841.  Xue Yang Junchi Yan Qi Ming Wentao Wang Xiaopeng Zhang and Qi Tian. 2021. Rethinking rotated object detection with gaussian wasserstein distance loss. In ICML. 11830--11841."},{"key":"e_1_3_2_2_47_1","first-page":"18381","article-title":"Learning high-precision bounding box for rotated object detection via kullback-leibler divergence","volume":"34","author":"Yang Xue","year":"2021","unstructured":"Xue Yang , Xiaojiang Yang , Jirui Yang , Qi Ming ,WentaoWang, Qi Tian , and Junchi Yan . 2021 . Learning high-precision bounding box for rotated object detection via kullback-leibler divergence . NeurIPS 34 (2021), 18381 -- 18394 . Xue Yang, Xiaojiang Yang, Jirui Yang, Qi Ming,WentaoWang, Qi Tian, and Junchi Yan. 2021. Learning high-precision bounding box for rotated object detection via kullback-leibler divergence. NeurIPS 34 (2021), 18381--18394.","journal-title":"NeurIPS"},{"key":"e_1_3_2_2_48_1","first-page":"379","article-title":"Collaborative refining for person re-identification with label noise","volume":"31","author":"Ye Mang","year":"2021","unstructured":"Mang Ye , He Li , Bo Du , Jianbing Shen , Ling Shao , and Steven CH Hoi . 2021 . Collaborative refining for person re-identification with label noise . IEEE TIP 31 (2021), 379 -- 391 . Mang Ye, He Li, Bo Du, Jianbing Shen, Ling Shao, and Steven CH Hoi. 2021. Collaborative refining for person re-identification with label noise. IEEE TIP 31 (2021), 379--391.","journal-title":"IEEE TIP"},{"key":"e_1_3_2_2_49_1","unstructured":"Mang Ye Weijian Ruan Bo Du and Mike Zheng Shou. 2021. Channel augmented joint learning for visible-infrared recognition. In ICCV. 13567--13576.  Mang Ye Weijian Ruan Bo Du and Mike Zheng Shou. 2021. Channel augmented joint learning for visible-infrared recognition. In ICCV. 13567--13576."},{"key":"e_1_3_2_2_50_1","volume-title":"Hoi","author":"Ye Mang","year":"2021","unstructured":"Mang Ye , Jianbing Shen , Gaojie Lin , Tao Xiang , Ling Shao , and Steven C. H . Hoi . 2021 . Deep learning for person re-identification: A survey and outlook. IEEE TPAMI ( 2021), 1--1. Mang Ye, Jianbing Shen, Gaojie Lin, Tao Xiang, Ling Shao, and Steven C. H. Hoi. 2021. Deep learning for person re-identification: A survey and outlook. IEEE TPAMI (2021), 1--1."},{"key":"e_1_3_2_2_51_1","volume-title":"Augmentation invariant and instance spreading feature for softmax embedding","author":"Ye Mang","year":"2020","unstructured":"Mang Ye , Jianbing Shen , Xu Zhang , Pong C Yuen , and Shih-Fu Chang . 2020. Augmentation invariant and instance spreading feature for softmax embedding . IEEE TPAMI ( 2020 ). Mang Ye, Jianbing Shen, Xu Zhang, Pong C Yuen, and Shih-Fu Chang. 2020. Augmentation invariant and instance spreading feature for softmax embedding. IEEE TPAMI (2020)."},{"key":"e_1_3_2_2_52_1","volume-title":"Jiashi Feng, and Shuicheng Yan.","author":"Yuan Li","year":"2021","unstructured":"Li Yuan , Yunpeng Chen , Tao Wang , Weihao Yu , Yujun Shi , Zihang Jiang , Francis EH Tay , Jiashi Feng, and Shuicheng Yan. 2021 . Tokens-to-token vit: Training vision transformers from scratch on imagenet. arXiv preprint arXiv:2101.11986 (2021). Li Yuan, Yunpeng Chen, Tao Wang, Weihao Yu, Yujun Shi, Zihang Jiang, Francis EH Tay, Jiashi Feng, and Shuicheng Yan. 2021. Tokens-to-token vit: Training vision transformers from scratch on imagenet. arXiv preprint arXiv:2101.11986 (2021)."},{"key":"e_1_3_2_2_53_1","volume-title":"HAT: Hierarchical Aggregation Transformers for Person Re-identification. ACM MM","author":"Zhang Guowen","year":"2021","unstructured":"Guowen Zhang , Pingping Zhang , Jinqing Qi , and Huchuan Lu . 2021 . HAT: Hierarchical Aggregation Transformers for Person Re-identification. ACM MM (2021), 516--525. Guowen Zhang, Pingping Zhang, Jinqing Qi, and Huchuan Lu. 2021. HAT: Hierarchical Aggregation Transformers for Person Re-identification. ACM MM (2021), 516--525."},{"key":"e_1_3_2_2_54_1","first-page":"281","article-title":"Person re-identification in aerial imagery","volume":"23","author":"Zhang Shizhou","year":"2020","unstructured":"Shizhou Zhang , Qi Zhang , Yifei Yang , Xing Wei , Peng Wang , Bingliang Jiao , and Yanning Zhang . 2020 . Person re-identification in aerial imagery . IEEE TMM 23 (2020), 281 -- 291 . Shizhou Zhang, Qi Zhang, Yifei Yang, Xing Wei, Peng Wang, Bingliang Jiao, and Yanning Zhang. 2020. Person re-identification in aerial imagery. IEEE TMM 23 (2020), 281--291.","journal-title":"IEEE TMM"},{"key":"e_1_3_2_2_55_1","doi-asserted-by":"crossref","unstructured":"Zhizheng Zhang Cuiling Lan Wenjun Zeng Xin Jin and Zhibo Chen. 2020. Relation-aware global attention for person re-identification. In CVPR. 3186--3195.  Zhizheng Zhang Cuiling Lan Wenjun Zeng Xin Jin and Zhibo Chen. 2020. Relation-aware global attention for person re-identification. In CVPR. 3186--3195.","DOI":"10.1109\/CVPR42600.2020.00325"},{"key":"e_1_3_2_2_56_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW54120.2021.00314"},{"key":"e_1_3_2_2_57_1","doi-asserted-by":"crossref","unstructured":"Liang Zheng Liyue Shen Lu Tian Shengjin Wang Jingdong Wang and Qi Tian. 2015. Scalable person re-identification: A benchmark. In ICCV. 1116--1124.  Liang Zheng Liyue Shen Lu Tian Shengjin Wang Jingdong Wang and Qi Tian. 2015. Scalable person re-identification: A benchmark. In ICCV. 1116--1124.","DOI":"10.1109\/ICCV.2015.133"},{"key":"e_1_3_2_2_58_1","volume-title":"Person reidentification: Past, present and future. arXiv preprint arXiv:1610.02984","author":"Zheng Liang","year":"2016","unstructured":"Liang Zheng , Yi Yang , and Alexander G Hauptmann . 2016. Person reidentification: Past, present and future. arXiv preprint arXiv:1610.02984 ( 2016 ). Liang Zheng, Yi Yang, and Alexander G Hauptmann. 2016. Person reidentification: Past, present and future. arXiv preprint arXiv:1610.02984 (2016)."},{"key":"e_1_3_2_2_59_1","doi-asserted-by":"crossref","unstructured":"Liang Zheng Hengheng Zhang Shaoyan Sun Manmohan Chandraker Yi Yang and Qi Tian. 2017. Person re-identification in the wild. In CVPR. 1367--1376.  Liang Zheng Hengheng Zhang Shaoyan Sun Manmohan Chandraker Yi Yang and Qi Tian. 2017. Person re-identification in the wild. In CVPR. 1367--1376.","DOI":"10.1109\/CVPR.2017.357"},{"key":"e_1_3_2_2_60_1","unstructured":"Zhedong Zheng Xiaodong Yang Zhiding Yu Liang Zheng Yi Yang and Jan Kautz. 2019. Joint discriminative and generative learning for person re-identification. In CVPR. 2138--2147.  Zhedong Zheng Xiaodong Yang Zhiding Yu Liang Zheng Yi Yang and Jan Kautz. 2019. Joint discriminative and generative learning for person re-identification. In CVPR. 2138--2147."},{"key":"e_1_3_2_2_61_1","doi-asserted-by":"crossref","unstructured":"Kaiyang Zhou Yongxin Yang Andrea Cavallaro and Tao Xiang. 2019. Omni-scale feature learning for person re-identification. In ICCV. 3702--3712.  Kaiyang Zhou Yongxin Yang Andrea Cavallaro and Tao Xiang. 2019. Omni-scale feature learning for person re-identification. In ICCV. 3702--3712.","DOI":"10.1109\/ICCV.2019.00380"},{"key":"e_1_3_2_2_62_1","volume-title":"AAformer: Auto-Aligned Transformer for Person Re-Identification. arXiv preprint arXiv:2104.00921","author":"Zhu Kuan","year":"2021","unstructured":"Kuan Zhu , Haiyun Guo , Shiliang Zhang , Yaowei Wang , Gaopan Huang , Honglin Qiao , Jing Liu , Jinqiao Wang , and Ming Tang . 2021. AAformer: Auto-Aligned Transformer for Person Re-Identification. arXiv preprint arXiv:2104.00921 ( 2021 ). Kuan Zhu, Haiyun Guo, Shiliang Zhang, Yaowei Wang, Gaopan Huang, Honglin Qiao, Jing Liu, Jinqiao Wang, and Ming Tang. 2021. AAformer: Auto-Aligned Transformer for Person Re-Identification. arXiv preprint arXiv:2104.00921 (2021)."},{"key":"e_1_3_2_2_63_1","volume-title":"Deformable detr: Deformable transformers for end-to-end object detection. arXiv preprint arXiv:2010.04159","author":"Zhu Xizhou","year":"2020","unstructured":"Xizhou Zhu , Weijie Su , Lewei Lu , Bin Li , Xiaogang Wang , and Jifeng Dai . 2020. Deformable detr: Deformable transformers for end-to-end object detection. arXiv preprint arXiv:2010.04159 ( 2020 ). Xizhou Zhu, Weijie Su, Lewei Lu, Bin Li, Xiaogang Wang, and Jifeng Dai. 2020. Deformable detr: Deformable transformers for end-to-end object detection. arXiv preprint arXiv:2010.04159 (2020)."}],"event":{"name":"MM '22: The 30th ACM International Conference on Multimedia","location":"Lisboa Portugal","acronym":"MM '22","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 30th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3503161.3547799","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3503161.3547799","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T19:02:34Z","timestamp":1750186954000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3503161.3547799"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,10,10]]},"references-count":63,"alternative-id":["10.1145\/3503161.3547799","10.1145\/3503161"],"URL":"https:\/\/doi.org\/10.1145\/3503161.3547799","relation":{},"subject":[],"published":{"date-parts":[[2022,10,10]]},"assertion":[{"value":"2022-10-10","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}