{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T02:00:47Z","timestamp":1777428047187,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":50,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62476035, 62206037, U24B20140 and 62306274"],"award-info":[{"award-number":["62476035, 62206037, U24B20140 and 62306274"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100004543","name":"China Scholarship Council","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100004543","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Open Research Fund of the State Key Laboratory of Brain-Machine Intelligence, Zhejiang University","award":["BMI2400012"],"award-info":[{"award-number":["BMI2400012"]}]},{"name":"Horizon Europe HarmonicAI project","award":["101131117"],"award-info":[{"award-number":["101131117"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3754854","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T06:56:44Z","timestamp":1761375404000},"page":"10837-10846","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["Advanced SpikingYOLOX: Extending Spiking Neural Network on Object Detection with Spike-based Partial Self-Attention and 2D-Spiking Transformer"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-3562-5932","authenticated-orcid":false,"given":"Wei","family":"Miao","sequence":"first","affiliation":[{"name":"Dalian University of Technology, Dalian, Liaoning, China and University of Jyv\u00e4skyl\u00e4, Jyv\u00e4skyl\u00e4, Finland"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3683-3779","authenticated-orcid":false,"given":"Jiangrong","family":"Shen","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong University, Xi'an, Shaanxi, China and Zhejiang University, Hangzhou, Zhejiang, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1305-0010","authenticated-orcid":false,"given":"Hongming","family":"Xu","sequence":"additional","affiliation":[{"name":"Dalian University of Technology, Dalian, Liaoning, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0327-1167","authenticated-orcid":false,"given":"Tommi","family":"K\u00e4rkk\u00e4inen","sequence":"additional","affiliation":[{"name":"University of Jyv\u00e4skyl\u00e4, Jyv\u00e4skyl\u00e4, Finland"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9245-5544","authenticated-orcid":false,"given":"Qi","family":"Xu","sequence":"additional","affiliation":[{"name":"Dalian University of Technology, Dalian, Liaoning, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-9900-6143","authenticated-orcid":false,"given":"Yi","family":"Xu","sequence":"additional","affiliation":[{"name":"Dalian University of Technology, Dalian, Liaoning, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0058-2429","authenticated-orcid":false,"given":"Fengyu","family":"Cong","sequence":"additional","affiliation":[{"name":"Dalian University of Technology, Dalian, Liaoning, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Yolov4: Optimal speed and accuracy of object detection. arXiv preprint arXiv:2004.10934","author":"Bochkovskiy Alexey","year":"2020","unstructured":"Alexey Bochkovskiy, Chien-Yao Wang, and Hong-Yuan Mark Liao. 2020. Yolov4: Optimal speed and accuracy of object detection. arXiv preprint arXiv:2004.10934 (2020)."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"e_1_3_2_1_3_1","first-page":"4479","article-title":"Fast fourier convolution","volume":"33","author":"Chi Lu","year":"2020","unstructured":"Lu Chi, Borui Jiang, and Yadong Mu. 2020. Fast fourier convolution. Advances in Neural Information Processing Systems 33 (2020), 4479--4488.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_4_1","volume-title":"Adam: A method for stochastic optimization. (No Title)","author":"Diederik P Kingma","year":"2014","unstructured":"P Kingma Diederik. 2014. Adam: A method for stochastic optimization. (No Title) (2014)."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"crossref","unstructured":"Peter U Diehl Daniel Neil Jonathan Binas Matthew Cook Shih-Chii Liu and Michael Pfeiffer. 2015. Fast-classifying high-accuracy spiking deep networks through weight and threshold balancing. In 2015 International joint conference on neural networks (IJCNN). ieee 1--8.","DOI":"10.1109\/IJCNN.2015.7280696"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01352"},{"key":"e_1_3_2_1_7_1","volume-title":"An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929","author":"Dosovitskiy Alexey","year":"2020","unstructured":"Alexey Dosovitskiy. 2020. An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.3390\/s23073447"},{"key":"e_1_3_2_1_9_1","volume-title":"Spikingjelly: An open-source machine learning infrastructure platform for spike-based intelligence. Science Advances 9, 40","author":"Fang Wei","year":"2023","unstructured":"Wei Fang, Yanqi Chen, Jianhao Ding, Zhaofei Yu, Timoth\u00e9e Masquelier, Ding Chen, Liwei Huang, Huihui Zhou, Guoqi Li, and Yonghong Tian. 2023. Spikingjelly: An open-source machine learning infrastructure platform for spike-based intelligence. Science Advances 9, 40 (2023), eadi1480."},{"key":"e_1_3_2_1_10_1","first-page":"21056","article-title":"Deep residual learning in spiking neural networks","volume":"34","author":"Fang Wei","year":"2021","unstructured":"Wei Fang, Zhaofei Yu, Yanqi Chen, Tiejun Huang, Timoth\u00e9e Masquelier, and Yonghong Tian. 2021. Deep residual learning in spiking neural networks. Advances in Neural Information Processing Systems 34 (2021), 21056--21069.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"crossref","unstructured":"Kai Han Yunhe Wang Hanting Chen Xinghao Chen Jianyuan Guo Zhenhua Liu Yehui Tang An Xiao Chunjing Xu Yixing Xu et al. 2022. A survey on vision transformer. IEEE transactions on pattern analysis and machine intelligence 45 1 (2022) 87--110.","DOI":"10.1109\/TPAMI.2022.3152247"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_13_1","volume-title":"Proceedings, Part IV 14","author":"He Kaiming","year":"2016","unstructured":"Kaiming He, Xiangyu Zhang, Shaoqing Ren, and Jian Sun. 2016. Identity mappings in deep residual networks. In Computer Vision--ECCV 2016: 14th European Conference, Amsterdam, The Netherlands, October 11--14, 2016, Proceedings, Part IV 14. Springer, 630--645."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01664"},{"key":"e_1_3_2_1_15_1","volume-title":"Advancing spiking neural networks toward deep residual learning","author":"Hu Yifan","year":"2024","unstructured":"Yifan Hu, Lei Deng, Yujie Wu, Man Yao, and Guoqi Li. 2024. Advancing spiking neural networks toward deep residual learning. IEEE Transactions on Neural Networks and Learning Systems (2024)."},{"key":"e_1_3_2_1_16_1","volume-title":"Solar","volume":"4","author":"Hussain Muhammad","year":"2024","unstructured":"Muhammad Hussain and Rahima Khanam. 2024. In-depth review of yolov1 to yolov10 variants for enhanced photovoltaic defect detection. In Solar, Vol. 4. MDPI, 351--386."},{"key":"e_1_3_2_1_17_1","unstructured":"Glenn Jocher and Jing Qiu. 2024. Ultralytics YOLO11. https:\/\/github.com\/ultralytics\/ultralytics"},{"key":"e_1_3_2_1_18_1","volume-title":"Yolov11: An overview of the key architectural enhancements. arXiv preprint arXiv:2410.17725","author":"Khanam Rahima","year":"2024","unstructured":"Rahima Khanam and Muhammad Hussain. 2024. Yolov11: An overview of the key architectural enhancements. arXiv preprint arXiv:2410.17725 (2024)."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2020.3047071"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6787"},{"key":"e_1_3_2_1_21_1","volume-title":"Seenn: Towards temporal spiking early exit neural networks. Advances in Neural Information Processing Systems 36","author":"Li Yuhang","year":"2024","unstructured":"Yuhang Li, Tamar Geller, Youngeun Kim, and Priyadarshini Panda. 2024. Seenn: Towards temporal spiking early exit neural networks. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_2_1_22_1","volume-title":"Spike calibration: Fast and accurate conversion of spiking neural network for object detection and segmentation. arXiv preprint arXiv:2207.02702","author":"Li Yang","year":"2022","unstructured":"Yang Li, Xiang He, Yiting Dong, Qingqun Kong, and Yi Zeng. 2022. Spike calibration: Fast and accurate conversion of spiking neural network for object detection and segmentation. arXiv preprint arXiv:2207.02702 (2022)."},{"key":"e_1_3_2_1_23_1","volume-title":"Proceedings, Part V 13","author":"Lin Tsung-Yi","year":"2014","unstructured":"Tsung-Yi Lin, Michael Maire, Serge Belongie, James Hays, Pietro Perona, Deva Ramanan, Piotr Doll\u00e1r, and C Lawrence Zitnick. 2014. Microsoft coco: Common objects in context. In Computer Vision--ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6--12, 2014, Proceedings, Part V 13. Springer, 740--755."},{"key":"e_1_3_2_1_24_1","volume-title":"European Conference on Computer Vision. Springer, 253--272","author":"Luo Xinhao","year":"2024","unstructured":"Xinhao Luo, Man Yao, Yuhong Chou, Bo Xu, and Guoqi Li. 2024. Integer-valued training and spike-driven inference spiking neural network for high-performance and energy-efficient object detection. In European Conference on Computer Vision. Springer, 253--272."},{"key":"e_1_3_2_1_25_1","volume-title":"Networks of spiking neurons: the third generation of neural network models. Neural networks 10, 9","author":"Maass Wolfgang","year":"1997","unstructured":"Wolfgang Maass. 1997. Networks of spiking neurons: the third generation of neural network models. Neural networks 10, 9 (1997), 1659--1671."},{"key":"e_1_3_2_1_26_1","volume-title":"The 39th Annual AAAI Conference on Artificial Intelligence. https:\/\/openreview.net\/forum?id=gA6AUMhnM6","author":"Miao Wei","unstructured":"Wei Miao, Jiangrong Shen, Qi Xu, Timo Hamalainen, Yi Xu, and Fengyu Cong. 2024. SpikingYOLOX: Improved YOLOX Object Detection with Fast Fourier Convolution and Spiking Neural Networks. In The 39th Annual AAAI Conference on Artificial Intelligence. https:\/\/openreview.net\/forum?id=gA6AUMhnM6"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1007\/s00530-023-01211-w"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2019.2931595"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.91"},{"key":"e_1_3_2_1_30_1","volume-title":"Towards spikebased machine intelligence with neuromorphic computing. Nature 575, 7784","author":"Roy Kaushik","year":"2019","unstructured":"Kaushik Roy, Akhilesh Jaiswal, and Priyadarshini Panda. 2019. Towards spikebased machine intelligence with neuromorphic computing. Nature 575, 7784 (2019), 607--617."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1038\/s43588-021-00184-y"},{"key":"e_1_3_2_1_32_1","volume-title":"Going deeper in spiking neural networks: VGG and residual architectures. Frontiers in neuroscience 13","author":"Sengupta Abhronil","year":"2019","unstructured":"Abhronil Sengupta, Yuting Ye, Robert Wang, Chiao Liu, and Kaushik Roy. 2019. Going deeper in spiking neural networks: VGG and residual architectures. Frontiers in neuroscience 13 (2019), 95."},{"key":"e_1_3_2_1_33_1","volume-title":"Image processing, analysis and machine vision","author":"Sonka Milan","unstructured":"Milan Sonka, Vaclav Hlavac, and Roger Boyle. 2013. Image processing, analysis and machine vision. Springer."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00603"},{"key":"e_1_3_2_1_35_1","volume-title":"Resolution-robust Large Mask Inpainting with Fourier Convolutions. arXiv preprint arXiv:2109.07161","author":"Suvorov Roman","year":"2021","unstructured":"Roman Suvorov, Elizaveta Logacheva, Anton Mashikhin, Anastasia Remizova, Arsenii Ashukha, Aleksei Silvestrov, Naejin Kong, Harshith Goka, Kiwoong Park, and Victor Lempitsky. 2021. Resolution-robust Large Mask Inpainting with Fourier Convolutions. arXiv preprint arXiv:2109.07161 (2021)."},{"key":"e_1_3_2_1_36_1","volume-title":"Timoth\u00e9e Masquelier, and Anthony Maida.","author":"Tavanaei Amirhossein","year":"2019","unstructured":"Amirhossein Tavanaei, Masoud Ghodrati, Saeed Reza Kheradpisheh, Timoth\u00e9e Masquelier, and Anthony Maida. 2019. Deep learning in spiking neural networks. Neural networks 111 (2019), 47--63."},{"key":"e_1_3_2_1_37_1","volume-title":"Mlp-mixer: An all-mlp architecture for vision. Advances in neural information processing systems 34","author":"Tolstikhin Ilya O","year":"2021","unstructured":"Ilya O Tolstikhin, Neil Houlsby, Alexander Kolesnikov, Lucas Beyer, Xiaohua Zhai, Thomas Unterthiner, Jessica Yung, Andreas Steiner, Daniel Keysers, Jakob Uszkoreit, et al. 2021. Mlp-mixer: An all-mlp architecture for vision. Advances in neural information processing systems 34 (2021), 24261--24272."},{"key":"e_1_3_2_1_38_1","volume-title":"Attention is all you need. Advances in Neural Information Processing Systems","author":"Vaswani A","year":"2017","unstructured":"A Vaswani. 2017. Attention is all you need. Advances in Neural Information Processing Systems (2017)."},{"key":"e_1_3_2_1_39_1","volume-title":"Yolov10: Real-time end-to-end object detection. arXiv preprint arXiv:2405.14458","author":"Wang Ao","year":"2024","unstructured":"Ao Wang, Hui Chen, Lihao Liu, Kai Chen, Zijia Lin, Jungong Han, and Guiguang Ding. 2024. Yolov10: Real-time end-to-end object detection. arXiv preprint arXiv:2405.14458 (2024)."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01020"},{"key":"e_1_3_2_1_41_1","volume-title":"Spatio-temporal backpropagation for training high-performance spiking neural networks. Frontiers in neuroscience 12","author":"Wu Yujie","year":"2018","unstructured":"Yujie Wu, Lei Deng, Guoqi Li, Jun Zhu, and Luping Shi. 2018. Spatio-temporal backpropagation for training high-performance spiking neural networks. Frontiers in neuroscience 12 (2018), 331."},{"key":"e_1_3_2_1_42_1","volume-title":"Biologically inspired structure learning with reverse knowledge distillation for spiking neural networks. arXiv preprint arXiv:2304.09500","author":"Xu Qi","year":"2023","unstructured":"Qi Xu, Yaxin Li, Xuanye Fang, Jiangrong Shen, Jian K Liu, Huajin Tang, and Gang Pan. 2023. Biologically inspired structure learning with reverse knowledge distillation for spiking neural networks. arXiv preprint arXiv:2304.09500 (2023)."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1049\/el.2017.2621"},{"key":"e_1_3_2_1_44_1","volume-title":"Spike-driven transformer v2: Meta spiking neural network architecture inspiring the design of next-generation neuromorphic chips. arXiv preprint arXiv:2404.03663","author":"Yao Man","year":"2024","unstructured":"Man Yao, Jiakui Hu, Tianxiang Hu, Yifan Xu, Zhaokun Zhou, Yonghong Tian, Bo Xu, and Guoqi Li. 2024. Spike-driven transformer v2: Meta spiking neural network architecture inspiring the design of next-generation neuromorphic chips. arXiv preprint arXiv:2404.03663 (2024)."},{"key":"e_1_3_2_1_45_1","volume-title":"Spike-driven transformer. Advances in neural information processing systems 36","author":"Yao Man","year":"2024","unstructured":"Man Yao, Jiakui Hu, Zhaokun Zhou, Li Yuan, Yonghong Tian, Bo Xu, and Guoqi Li. 2024. Spike-driven transformer. Advances in neural information processing systems 36 (2024)."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01055"},{"key":"e_1_3_2_1_47_1","volume-title":"International Conference on Learning Representations.","author":"Zhang Hongyi","year":"2018","unstructured":"Hongyi Zhang, Moustapha Cisse, Yann N Dauphin, and David Lopez-Paz. 2018. mixup: Beyond Empirical Risk Minimization. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_48_1","volume-title":"Spikformer: When spiking neural network meets transformer. arXiv preprint arXiv:2209.15425","author":"Zhou Zhaokun","year":"2022","unstructured":"Zhaokun Zhou, Yuesheng Zhu, Chao He, YaoweiWang, Shuicheng Yan, Yonghong Tian, and Li Yuan. 2022. Spikformer: When spiking neural network meets transformer. arXiv preprint arXiv:2209.15425 (2022)."},{"key":"e_1_3_2_1_49_1","volume-title":"Deformable detr: Deformable transformers for end-to-end object detection. arXiv preprint arXiv:2010.04159","author":"Zhu Xizhou","year":"2020","unstructured":"Xizhou Zhu, Weijie Su, Lewei Lu, Bin Li, Xiaogang Wang, and Jifeng Dai. 2020. Deformable detr: Deformable transformers for end-to-end object detection. arXiv preprint arXiv:2010.04159 (2020)."},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2023.3238524"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3754854","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:36:26Z","timestamp":1765308986000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3754854"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":50,"alternative-id":["10.1145\/3746027.3754854","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3754854","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}