{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,15]],"date-time":"2026-07-15T18:31:32Z","timestamp":1784140292522,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":90,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3754806","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T06:54:17Z","timestamp":1761375257000},"page":"2409-2418","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":15,"title":["UniRGB-IR: A Unified Framework for Visible-Infrared Semantic Tasks via Adapter Tuning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-7463-7328","authenticated-orcid":false,"given":"Maoxun","family":"Yuan","sequence":"first","affiliation":[{"name":"Institute of Artificial Intelligence, Beihang University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-9807-910X","authenticated-orcid":false,"given":"Bo","family":"Cui","sequence":"additional","affiliation":[{"name":"School of Computer Science and Engineering, Beihang University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2487-3944","authenticated-orcid":false,"given":"Tianyi","family":"Zhao","sequence":"additional","affiliation":[{"name":"Institute of Artificial Intelligence, Beihang University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-1567-7665","authenticated-orcid":false,"given":"Jiayi","family":"Wang","sequence":"additional","affiliation":[{"name":"CTTL-Terminal, China Academy of Information and Communications Technology, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1489-2575","authenticated-orcid":false,"given":"Shan","family":"Fu","sequence":"additional","affiliation":[{"name":"CTTL-Terminals, China Academy of Information and Communications Technology, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7084-9101","authenticated-orcid":false,"given":"Xue","family":"Yang","sequence":"additional","affiliation":[{"name":"School of Automation and Intelligent Sensing, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0778-8377","authenticated-orcid":false,"given":"Xingxing","family":"Wei","sequence":"additional","affiliation":[{"name":"Institute of Artificial Intelligence, Beihang University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Jamie Ryan Kiros, and Geoffrey E Hinton","author":"Ba Jimmy Lei","year":"2016","unstructured":"Jimmy Lei Ba, Jamie Ryan Kiros, and Geoffrey E Hinton. 2016. Layer normalization. arXiv preprint arXiv:1607.06450 (2016)."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.cja.2020.09.022"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00644"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW59228.2023.00046"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2019.2891104"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2018.08.007"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20077-9_9"},{"key":"e_1_3_2_1_8_1","volume-title":"Vision transformer adapter for dense predictions. arXiv preprint arXiv:2205.08534","author":"Chen Zhe","year":"2022","unstructured":"Zhe Chen, Yuchen Duan, Wenhai Wang, Junjun He, Tong Lu, Jifeng Dai, and Yu Qiao. 2022a. Vision transformer adapter for dense predictions. arXiv preprint arXiv:2205.08534 (2022)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_2_1_10_1","unstructured":"Alexey Dosovitskiy Lucas Beyer Alexander Kolesnikov Dirk Weissenborn Xiaohua Zhai Thomas Unterthiner Mostafa Dehghani Matthias Minderer Georg Heigold Sylvain Gelly et al. 2020. An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00675"},{"key":"e_1_3_2_1_12_1","first-page":"5541","volume-title":"TPAMI","volume":"44","author":"Fu Keren","year":"2021","unstructured":"Keren Fu, Deng-Ping Fan, Ge-Peng Ji, Qijun Zhao, Jianbing Shen, and Ce Zhu. 2021. Siamese network for RGB-D salient object detection and beyond. TPAMI, Vol. 44, 9 (2021), 5541-5559."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2018.11.017"},{"key":"e_1_3_2_1_14_1","volume-title":"UniTR: A unified transformer-based framework for co-object and multi-modal saliency detection","author":"Guo Ruohao","year":"2024","unstructured":"Ruohao Guo, Xianghua Ying, Yanyu Qi, and Liao Qu. 2024. UniTR: A unified transformer-based framework for co-object and multi-modal saliency detection. IEEE transactions on multimedia (2024)."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2017.8206396"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2019.2937688"},{"key":"e_1_3_2_1_18_1","volume-title":"International conference on machine learning. PMLR, 2790-2799","author":"Houlsby Neil","year":"2019","unstructured":"Neil Houlsby, Andrei Giurgiu, Stanislaw Jastrzebski, Bruna Morrone, Quentin De Laroussilhe, Andrea Gesmundo, Mona Attariyan, and Sylvain Gelly. 2019. Parameter-efficient transfer learning for NLP. In International conference on machine learning. PMLR, 2790-2799."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00745"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICIP.2019.8803025"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00069"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298706"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW54120.2021.00389"},{"key":"e_1_3_2_1_24_1","volume-title":"A survey of the recent architectures of deep convolutional neural networks. Artificial intelligence review","author":"Khan Asifullah","year":"2020","unstructured":"Asifullah Khan, Anabia Sohail, Umme Zahoora, and Aqsa Saeed Qureshi. 2020. A survey of the recent architectures of deep convolutional neural networks. Artificial intelligence review, Vol. 53 (2020), 5455-5516."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2017.36"},{"key":"e_1_3_2_1_26_1","volume-title":"British Machine Vision Conference (BMVC).","author":"Li Chengyang","year":"2018","unstructured":"Chengyang Li, Dan Song, Ruofeng Tong, and Min Tang. 2018. Multispectral Pedestrian Detection via Simultaneous Detection and Segmentation. In British Machine Vision Conference (BMVC)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2018.08.005"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00297"},{"key":"e_1_3_2_1_29_1","volume-title":"Residual spatial fusion network for rgb-thermal semantic segmentation. arXiv preprint arXiv:2306.10364","author":"Li Ping","year":"2023","unstructured":"Ping Li, Junjie Chen, Binbin Lin, and Xianghua Xu. 2023a. Residual spatial fusion network for rgb-thermal semantic segmentation. arXiv preprint arXiv:2306.10364 (2023)."},{"key":"e_1_3_2_1_30_1","volume-title":"Confidence-aware fusion using dempster-shafer theory for multispectral pedestrian detection","author":"Li Qing","year":"2022","unstructured":"Qing Li, Changqing Zhang, Qinghua Hu, Huazhu Fu, and Pengfei Zhu. 2022b. Confidence-aware fusion using dempster-shafer theory for multispectral pedestrian detection. IEEE Transactions on Multimedia (2022)."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20077-9_17"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2022.12.036"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.324"},{"key":"e_1_3_2_1_34_1","first-page":"740","volume-title":"Zurich","author":"Lin Tsung-Yi","year":"2014","unstructured":"Tsung-Yi Lin, Michael Maire, Serge Belongie, James Hays, Pietro Perona, Deva Ramanan, Piotr Doll\u00e1r, and C Lawrence Zitnick. 2014. Microsoft coco: Common objects in context. In Computer Vision-ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6-12, 2014, Proceedings, Part V 13. Springer, 740-755."},{"key":"e_1_3_2_1_35_1","volume-title":"Multispectral deep neural networks for pedestrian detection. arXiv preprint arXiv:1611.02644","author":"Liu Jingjing","year":"2016","unstructured":"Jingjing Liu, Shaoting Zhang, Shu Wang, and Dimitris N Metaxas. 2016b. Multispectral deep neural networks for pedestrian detection. arXiv preprint arXiv:1611.02644 (2016)."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00479"},{"key":"e_1_3_2_1_37_1","first-page":"13756","article-title":"Learning selective self-mutual attention for RGB-D saliency detection","author":"Liu Nian","year":"2020","unstructured":"Nian Liu, Ni Zhang, and Junwei Han. 2020. Learning selective self-mutual attention for RGB-D saliency detection. In CVPR. 13756-13765.","journal-title":"CVPR."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46448-0_2"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2021.3127149"},{"key":"e_1_3_2_1_41_1","volume-title":"Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101","author":"Loshchilov Ilya","year":"2017","unstructured":"Ilya Loshchilov and Frank Hutter. 2017. Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101 (2017)."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2023.3234702"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01059"},{"key":"e_1_3_2_1_44_1","volume-title":"Faster r-cnn: Towards real-time object detection with region proposal networks. Advances in neural information processing systems","author":"Ren Shaoqing","year":"2015","unstructured":"Shaoqing Ren, Kaiming He, Ross Girshick, and Jian Sun. 2015. Faster r-cnn: Towards real-time object detection with region proposal networks. Advances in neural information processing systems, Vol. 28 (2015)."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2023.109913"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW59228.2023.00673"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA40945.2020.9196831"},{"key":"e_1_3_2_1_48_1","volume-title":"International Conference on Machine Learning. PMLR, 5986-5995","author":"Stickland Asa Cooper","year":"2019","unstructured":"Asa Cooper Stickland and Iain Murray. 2019. Bert and pals: Projected attention layers for efficient adaptation in multi-task learning. In International Conference on Machine Learning. PMLR, 5986-5995."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2019.2904733"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00516"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2021.3087412"},{"key":"e_1_3_2_1_52_1","volume-title":"RGBT salient object detection: A large-scale dataset and benchmark. TMM","author":"Tu Zhengzheng","year":"2022","unstructured":"Zhengzheng Tu, Yan Ma, Zhun Li, Chenglong Li, Jieming Xu, and Yongtao Liu. 2022. RGBT salient object detection: A large-scale dataset and benchmark. TMM (2022)."},{"key":"e_1_3_2_1_53_1","first-page":"141","article-title":"M3S-NIR: Multi-modal multi-scale noise-insensitive ranking for RGB-T saliency detection","author":"Tu Zhengzheng","year":"2019","unstructured":"Zhengzheng Tu, Tian Xia, Chenglong Li, Yijuan Lu, and Jin Tang. 2019a. M3S-NIR: Multi-modal multi-scale noise-insensitive ranking for RGB-T saliency detection. In MIPR. IEEE, 141-146.","journal-title":"MIPR. IEEE"},{"key":"e_1_3_2_1_54_1","first-page":"160","volume-title":"TMM","volume":"22","author":"Tu Zhengzheng","year":"2019","unstructured":"Zhengzheng Tu, Tian Xia, Chenglong Li, Xiaoxiao Wang, Yan Ma, and Jin Tang. 2019b. RGB-T image saliency detection via collaborative graph learning. TMM, Vol. 22, 1 (2019), 160-173."},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00182"},{"key":"e_1_3_2_1_56_1","volume-title":"Attention is all you need. Advances in neural information processing systems","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, Lukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems, Vol. 30 (2017)."},{"key":"e_1_3_2_1_57_1","volume-title":"RGB-T saliency detection benchmark: Dataset, baselines, analysis and a novel approach","author":"Wang Guizhao","unstructured":"Guizhao Wang, Chenglong Li, Yunpeng Ma, Aihua Zheng, Jin Tang, and Bin Luo. 2018. RGB-T saliency detection benchmark: Dataset, baselines, analysis and a novel approach. In IGTA. Springer, 359-369."},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00061"},{"key":"e_1_3_2_1_59_1","volume-title":"Sgfnet: semantic-guided fusion network for rgb-thermal semantic segmentation","author":"Wang Yike","year":"2023","unstructured":"Yike Wang, Gongyang Li, and Zhi Liu. 2023. Sgfnet: semantic-guided fusion network for rgb-thermal semantic segmentation. IEEE Transactions on Circuits and Systems for Video Technology (2023)."},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"crossref","unstructured":"Zhaokai Wang Xizhou Zhu Xue Yang Gen Luo Hao Li Changyao Tian Wenhan Dou Junqi Ge Lewei Lu Yu Qiao et al. 2025. Parameter-Inverted Image Pyramid Networks for Visual Perception and Multimodal Understanding. arXiv preprint arXiv:2501.07783 (2025).","DOI":"10.1109\/TPAMI.2025.3593283"},{"key":"e_1_3_2_1_61_1","volume-title":"Unified Adversarial Patch for Visible-Infrared Cross-modal Attacks in the Physical World","author":"Wei Xingxing","year":"2023","unstructured":"Xingxing Wei, Yao Huang, Yitong Sun, and Jie Yu. 2023. Unified Adversarial Patch for Visible-Infrared Cross-modal Attacks in the Physical World. IEEE Transactions on Pattern Analysis and Machine Intelligence (2023)."},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i3.20176"},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2022.108881"},{"key":"e_1_3_2_1_64_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2022.3228497"},{"key":"e_1_3_2_1_65_1","volume-title":"ViT-CoMer: Vision Transformer with Convolutional Multi-scale Feature Interaction for Dense Predictions. arXiv preprint arXiv:2403.07392","author":"Xia Chunlong","year":"2024","unstructured":"Chunlong Xia, Xinliang Wang, Feng Lv, Xin Hao, and Yifeng Shi. 2024. ViT-CoMer: Vision Transformer with Convolutional Multi-scale Feature Interaction for Dense Predictions. arXiv preprint arXiv:2403.07392 (2024)."},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.451"},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"publisher","DOI":"10.1109\/JSTARS.2022.3179612"},{"key":"e_1_3_2_1_68_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition.","author":"Yin Dongshuo","year":"2025","unstructured":"Dongshuo Yin, Leiyi Hu, Bin Li, Youqun Zhang, and Xue Yang. 2025. 5%&gt; 100%: Breaking performance shackles of full fine-tuning on visual recognition tasks. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition."},{"key":"e_1_3_2_1_69_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01926"},{"key":"e_1_3_2_1_70_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2024.102246"},{"key":"e_1_3_2_1_71_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20077-9_30"},{"key":"e_1_3_2_1_72_1","volume-title":"C2Former: Calibrated and Complementary Transformer for RGB-Infrared Object Detection","author":"Yuan Maoxun","year":"2024","unstructured":"Maoxun Yuan and Xingxing Wei. 2024. C2Former: Calibrated and Complementary Transformer for RGB-Infrared Object Detection. IEEE Transactions on Geoscience and Remote Sensing (2024), 1-1."},{"key":"e_1_3_2_1_73_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICIP40778.2020.9191080"},{"key":"e_1_3_2_1_74_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV48630.2021.00012"},{"key":"e_1_3_2_1_75_1","volume-title":"CMX: Cross-modal fusion for RGB-X semantic segmentation with transformers","author":"Zhang Jiaming","year":"2023","unstructured":"Jiaming Zhang, Huayao Liu, Kailun Yang, Xinxin Hu, Ruiping Liu, and Rainer Stiefelhagen. 2023a. CMX: Cross-modal fusion for RGB-X semantic segmentation with transformers. IEEE Transactions on Intelligent Transportation Systems (2023)."},{"key":"e_1_3_2_1_76_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2018.09.015"},{"key":"e_1_3_2_1_77_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2021.3105143"},{"key":"e_1_3_2_1_78_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00523"},{"key":"e_1_3_2_1_79_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2019.2959253"},{"key":"e_1_3_2_1_80_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00708"},{"key":"e_1_3_2_1_81_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2022.3233089"},{"key":"e_1_3_2_1_82_1","volume-title":"A feature divide-and-conquer network for rgb-t semantic segmentation","author":"Zhao Shenlu","year":"2022","unstructured":"Shenlu Zhao and Qiang Zhang. 2022. A feature divide-and-conquer network for rgb-t semantic segmentation. IEEE Transactions on Circuits and Systems for Video Technology (2022)."},{"key":"e_1_3_2_1_83_1","volume-title":"Removal and Selection: Improving RGB-Infrared Object Detection via Coarse-to-Fine Fusion. arXiv preprint arXiv:2401.10731","author":"Zhao Tianyi","year":"2024","unstructured":"Tianyi Zhao, Maoxun Yuan, and Xingxing Wei. 2024b. Removal and Selection: Improving RGB-Infrared Object Detection via Coarse-to-Fine Fusion. arXiv preprint arXiv:2401.10731 (2024)."},{"key":"e_1_3_2_1_84_1","first-page":"6881","article-title":"Rethinking semantic segmentation from a sequence-to-sequence perspective with transformers","author":"Zheng Sixiao","year":"2021","unstructured":"Sixiao Zheng, Jiachen Lu, Hengshuang Zhao, Xiatian Zhu, Zekun Luo, Yabiao Wang, Yanwei Fu, Jianfeng Feng, Tao Xiang, Philip HS Torr, et al., 2021. Rethinking semantic segmentation from a sequence-to-sequence perspective with transformers. In CVPR. 6881-6890.","journal-title":"CVPR."},{"key":"e_1_3_2_1_85_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58523-5_46"},{"key":"e_1_3_2_1_86_1","volume-title":"Knowledge distillation segformer-based network for RGB-T semantic segmentation","author":"Zhou Wujie","year":"2024","unstructured":"Wujie Zhou, Tingting Gong, and Weiqing Yan. 2024. Knowledge distillation segformer-based network for RGB-T semantic segmentation. IEEE Transactions on Systems, Man, and Cybernetics: Systems (2024)."},{"key":"e_1_3_2_1_87_1","volume-title":"WaveNet: Wavelet network with knowledge distillation for RGB-T salient object detection","author":"Zhou Wujie","year":"2023","unstructured":"Wujie Zhou, Fan Sun, Qiuping Jiang, Runmin Cong, and Jenq-Neng Hwang. 2023a. WaveNet: Wavelet network with knowledge distillation for RGB-T salient object detection. IEEE Transactions on Image Processing (2023)."},{"key":"e_1_3_2_1_88_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2023.3242775"},{"key":"e_1_3_2_1_89_1","volume-title":"Deformable detr: Deformable transformers for end-to-end object detection. arXiv preprint arXiv:2010.04159","author":"Zhu Xizhou","year":"2020","unstructured":"Xizhou Zhu, Weijie Su, Lewei Lu, Bin Li, Xiaogang Wang, and Jifeng Dai. 2020. Deformable detr: Deformable transformers for end-to-end object detection. arXiv preprint arXiv:2010.04159 (2020)."},{"key":"e_1_3_2_1_90_1","first-page":"132267","article-title":"Parameter-inverted image pyramid networks","volume":"37","author":"Zhu Xizhou","year":"2024","unstructured":"Xizhou Zhu, Xue Yang, Zhaokai Wang, Hao Li, Wenhan Dou, Junqi Ge, Lewei Lu, Yu Qiao, and Jifeng Dai. 2024. Parameter-inverted image pyramid networks. Advances in Neural Information Processing Systems, Vol. 37 (2024), 132267-132288.","journal-title":"Advances in Neural Information Processing Systems"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3754806","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:47:08Z","timestamp":1765309628000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3754806"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":90,"alternative-id":["10.1145\/3746027.3754806","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3754806","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}