{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T15:23:07Z","timestamp":1783610587181,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":50,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,26]],"date-time":"2023-10-26T00:00:00Z","timestamp":1698278400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2022ZD0160404"],"award-info":[{"award-number":["2022ZD0160404"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62206031, 62271346"],"award-info":[{"award-number":["62206031, 62271346"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100002858","name":"China Postdoctoral Science Foundation","doi-asserted-by":"publisher","award":["2021M700613,2022M720581,2023T160762"],"award-info":[{"award-number":["2021M700613,2022M720581,2023T160762"]}],"id":[{"id":"10.13039\/501100002858","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,26]]},"DOI":"10.1145\/3581783.3613444","type":"proceedings-article","created":{"date-parts":[[2023,10,27]],"date-time":"2023-10-27T07:26:54Z","timestamp":1698391614000},"page":"3787-3795","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":23,"title":["Attentive Alignment Network for Multispectral Pedestrian Detection"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-0581-0327","authenticated-orcid":false,"given":"Nuo","family":"Chen","sequence":"first","affiliation":[{"name":"Chongqing University, Chongqing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6978-8834","authenticated-orcid":false,"given":"Jin","family":"Xie","sequence":"additional","affiliation":[{"name":"Chongqing University, Chongqing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9872-9286","authenticated-orcid":false,"given":"Jing","family":"Nie","sequence":"additional","affiliation":[{"name":"Chongqing University, Chongqing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5160-6841","authenticated-orcid":false,"given":"Jiale","family":"Cao","sequence":"additional","affiliation":[{"name":"Tianjin University, Tianjin, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7824-0985","authenticated-orcid":false,"given":"Zhuang","family":"Shao","sequence":"additional","affiliation":[{"name":"University of Warwick, Coventry, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6670-3727","authenticated-orcid":false,"given":"Yanwei","family":"Pang","sequence":"additional","affiliation":[{"name":"Tianjin University &amp; Shanghai Artificial Intelligence Laboratory, Tianjin, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2023,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00644"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3076733"},{"key":"e_1_3_2_1_3_1","volume-title":"R-fcn: Object detection via region-based fully convolutional networks. In Adv. Neural Inform. Process. Syst. 379--387.","author":"Dai Jifeng","year":"2016","unstructured":"Jifeng Dai, Yi Li, Kaiming He, and Jian Sun. 2016. R-fcn: Object detection via region-based fully convolutional networks. In Adv. Neural Inform. Process. Syst. 379--387."},{"key":"e_1_3_2_1_4_1","volume-title":"Deformable Convolutional Networks. In Int. Conf. Comput. Vis. 764--773","author":"Dai Jifeng","year":"2017","unstructured":"Jifeng Dai, Haozhi Qi, Yuwen Xiong, Yi Li, Guodong Zhang, Han Hu, and Yichen Wei. 2017. Deformable Convolutional Networks. In Int. Conf. Comput. Vis. 764--773."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2022.3146575"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_2_1_7_1","volume-title":"Int. Conf. Learn. Represent.","author":"Dosovitskiy Alexey","year":"2021","unstructured":"Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, Xiaohua Zhai, Thomas Unterthiner, Mostafa Dehghani, Matthias Minderer, Georg Heigold, Sylvain Gelly, Jakob Uszkoreit, and Neil Houlsby. 2021. An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. In Int. Conf. Learn. Represent."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00667"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.169"},{"key":"e_1_3_2_1_10_1","volume-title":"L\u00f3pez","author":"Gonz\u00e1lez Alejandro","year":"2016","unstructured":"Alejandro Gonz\u00e1lez, Zhijie Fang, Yainuvis Socarras, Joan Serrat, David Vazquez, Jiaolong Xu, and Antonio M. L\u00f3pez. 2016. Pedestrian Detection at Day\/Night Time with Visible and FIR Cameras: A Comparison. Sensors (2016)."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2018.11.017"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.322"},{"key":"e_1_3_2_1_13_1","volume-title":"Deep Residual Learning for Image Recognition. In IEEE Conf. Comput. Vis. Pattern Recog. 770--778","author":"He Kaiming","year":"2016","unstructured":"Kaiming He, Xiangyu Zhang, Shaoqing Ren, and Jian Sun. 2016. Deep Residual Learning for Image Recognition. In IEEE Conf. Comput. Vis. Pattern Recog. 770--778."},{"key":"e_1_3_2_1_14_1","volume-title":"Densely Connected Convolutional Networks. In IEEE Conf. Comput. Vis. Pattern Recog. 2261--2269","author":"Huang Gao","unstructured":"Gao Huang, Zhuang Liu, Laurens Van Der Maaten, and Kilian Q. Weinberger. 2017. Densely Connected Convolutional Networks. In IEEE Conf. Comput. Vis. Pattern Recog. 2261--2269."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298706"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2021.3076466"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01264-9_45"},{"key":"e_1_3_2_1_18_1","volume-title":"Brit. Mach. Vis. Conf.","author":"Li Chengyang","year":"2018","unstructured":"Chengyang Li, Dan Song, Ruofeng Tong, and Min Tang. 2018. Multispectral Pedestrian Detection via Simultaneous Detection and Segmentation. In Brit. Mach. Vis. Conf."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2018.08.005"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01240-3_45"},{"key":"e_1_3_2_1_21_1","volume-title":"Detr for pedestrian detection. arXiv:2012.06785","author":"Lin Matthieu","year":"2020","unstructured":"SMatthieu Lin, Chuming Li, Xingyuan Bu, Ming Sun, Chen Lin, Junjie Yan, Wanli Ouyang, and Zhidong Deng. 2020. Detr for pedestrian detection. arXiv:2012.06785 (2020)."},{"key":"e_1_3_2_1_22_1","volume-title":"Focal Loss for Dense Object Detection. In Int. Conf. Comput. Vis. IEEE Computer Society, 2999--3007","author":"Lin Tsung-Yi","year":"2017","unstructured":"Tsung-Yi Lin, Priya Goyal, Ross B. Girshick, Kaiming He, and Piotr Doll\u00e1r. 2017b. Focal Loss for Dense Object Detection. In Int. Conf. Comput. Vis. IEEE Computer Society, 2999--3007."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.106"},{"key":"e_1_3_2_1_24_1","volume-title":"Multispectral Deep Neural Networks for Pedestrian Detection. In Brit. Mach. Vis. Conf.","author":"Liu Jingjing","unstructured":"Jingjing Liu, Shaoting Zhang, Shu Wang, and Dimitris N. Metaxas. 2016. Multispectral Deep Neural Networks for Pedestrian Detection. In Brit. Mach. Vis. Conf."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2021.3060162"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01264-9_38"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00533"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.639"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00963"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2020.3038649"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2020.3034487"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00507"},{"key":"e_1_3_2_1_33_1","unstructured":"Shaoqing Ren Kaiming He Ross Girshick and Jian Sun. 2015. Faster r-cnn: Towards real-time object detection with region proposal networks. In Adv. Neural Inform. Process. Syst."},{"key":"e_1_3_2_1_34_1","volume-title":"Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556","author":"Simonyan Karen","year":"2014","unstructured":"Karen Simonyan and Andrew Zisserman. 2014. Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556 (2014)."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01234-2_33"},{"key":"e_1_3_2_1_36_1","volume-title":"IEEE Conf. Comput. Vis. Pattern Recog. 1--9.","author":"Szegedy C.","unstructured":"C. Szegedy, Wei Liu, Yangqing Jia, P. Sermanet, S. Reed, D. Anguelov, D. Erhan, V. Vanhoucke, and A. Rabinovich. 2015. Going deeper with convolutions. In IEEE Conf. Comput. Vis. Pattern Recog. 1--9."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00972"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547895"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58520-4_6"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2020.3040854"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.634"},{"key":"e_1_3_2_1_42_1","volume-title":"BAANet: Learning Bi-directional Adaptive Attention Gates for Multispectral Pedestrian Detection. In International Conference on Robotics and Automation (ICRA). 2920--2926","author":"Yang Xiaoxiao","year":"2022","unstructured":"Xiaoxiao Yang, Yeqiang Qian, Huijie Zhu, Chunxiang Wang, and Ming Yang. 2022. BAANet: Learning Bi-directional Adaptive Attention Gates for Multispectral Pedestrian Detection. In International Conference on Robotics and Automation (ICRA). 2920--2926."},{"key":"e_1_3_2_1_43_1","volume-title":"IEEE Winter Conference on Applications of Computer Vision.","author":"Zhang Heng","year":"2021","unstructured":"Heng Zhang, Elisa Fromont, Sebastie Lefevre, and Bruno Avignon. 2021. Guided attentive feature fusion for multispectral pedestrian detection. In IEEE Winter Conference on Applications of Computer Vision."},{"key":"e_1_3_2_1_44_1","volume-title":"Jianke Zhu, Yao Hu, and Steven C.H. Hoi.","author":"Zhang Jialiang","year":"2019","unstructured":"Jialiang Zhang, Lixiang Lin, Yang Li, Yun chen Chen, Jianke Zhu, Yao Hu, and Steven C.H. Hoi. 2019a. Attribute-aware pedestrian detection in a crowd. arXiv:1912.08661 (2019)."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2018.09.015"},{"key":"e_1_3_2_1_46_1","volume-title":"Weakly Aligned Cross-Modal Learning for Multispectral Pedestrian Detection. In international conference on computer vision.","author":"Zhang Lu","year":"2019","unstructured":"Lu Zhang, Xiangyu Zhu, Xiangyu Chen, Xu Yang, Zhen Lei, and Zhiyong Liu. 2019c. Weakly Aligned Cross-Modal Learning for Multispectral Pedestrian Detection. In international conference on computer vision."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.474"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00731"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01246-5_9"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58523-5_46"}],"event":{"name":"MM '23: The 31st ACM International Conference on Multimedia","location":"Ottawa ON Canada","acronym":"MM '23","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 31st ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3613444","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3581783.3613444","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T23:59:25Z","timestamp":1755820765000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3613444"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,26]]},"references-count":50,"alternative-id":["10.1145\/3581783.3613444","10.1145\/3581783"],"URL":"https:\/\/doi.org\/10.1145\/3581783.3613444","relation":{},"subject":[],"published":{"date-parts":[[2023,10,26]]},"assertion":[{"value":"2023-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}