{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:30:34Z","timestamp":1765308634092,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":54,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62276121,62402252"],"award-info":[{"award-number":["62276121,62402252"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"TianYuan funds for Mathematics of the National Science Foundation of China","award":["12326604"],"award-info":[{"award-number":["12326604"]}]},{"name":"Shenzhen International Research Cooperation Project","award":["GJHZ20220913142611021"],"award-info":[{"award-number":["GJHZ20220913142611021"]}]},{"name":"Pengcheng Laboratory Research Project"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755045","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T05:50:47Z","timestamp":1761371447000},"page":"286-295","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["DS-Det: Single-Query Paradigm and Attention Disentangled Learning for Flexible Object Detection"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0682-2158","authenticated-orcid":false,"given":"Guiping","family":"Cao","sequence":"first","affiliation":[{"name":"RITAS, Southern University of Science and Technology, Shenzhen, China and Pengcheng Laboratory, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8564-0346","authenticated-orcid":false,"given":"Xiangyuan","family":"Lan","sequence":"additional","affiliation":[{"name":"Pengcheng Laboratory, Shenzhen, China and Pazhou Laboratory (Huangpu), Guangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2408-8302","authenticated-orcid":false,"given":"Wenjian","family":"Huang","sequence":"additional","affiliation":[{"name":"RITAS, Southern University of Science and Technology, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9317-0268","authenticated-orcid":false,"given":"Jianguo","family":"Zhang","sequence":"additional","affiliation":[{"name":"RITAS and Department of Computer Science and Engineering, Southern University of Science and Technology, Shenzhen, China and Pengcheng Laboratory, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6238-8499","authenticated-orcid":false,"given":"Dongmei","family":"Jiang","sequence":"additional","affiliation":[{"name":"Pengcheng Laboratory, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6110-4036","authenticated-orcid":false,"given":"Yaowei","family":"Wang","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology (Shenzhen), Shenzhen, China and Pengcheng Laboratory, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"Jamie Ryan Kiros, and Geoffrey E Hinton","author":"Ba Jimmy Lei","year":"2016","unstructured":"Jimmy Lei Ba, Jamie Ryan Kiros, and Geoffrey E Hinton. 2016. Layer normalization. arXiv preprint arXiv:1607.06450 (2016)."},{"key":"e_1_3_2_2_2_1","volume-title":"Align-DETR: Improving DETR with simple IoU-aware BCE loss. arXiv preprint arXiv:2304.07527","author":"Cai Zhi","year":"2023","unstructured":"Zhi Cai, Songtao Liu, Guodong Wang, Zheng Ge, Xiangyu Zhang, and Di Huang. 2023. Align-DETR: Improving DETR with simple IoU-aware BCE loss. arXiv preprint arXiv:2304.07527 (2023)."},{"key":"e_1_3_2_2_3_1","volume-title":"Proceedings of the Thirty-Third International Joint Conference on Artificial Intelligence. 605-613","author":"Cao Guiping","year":"2024","unstructured":"Guiping Cao, Wenjian Huang, Xiangyuan Lan, Jianguo Zhang, Dongmei Jiang, and Yaowei Wang. 2024. MLP-DINO: Category Modeling and Query Graphing with Deep MLP for Object Detection. In Proceedings of the Thirty-Third International Joint Conference on Artificial Intelligence. 605-613."},{"key":"e_1_3_2_2_4_1","volume-title":"Cross-DINO: Cross the Deep MLP and Transformer for Small Object Detection. arXiv preprint arXiv:2505.21868","author":"Cao Guiping","year":"2025","unstructured":"Guiping Cao, Wenjian Huang, Xiangyuan Lan, Jianguo Zhang, Dongmei Jiang, and Yaowei Wang. 2025. Cross-DINO: Cross the Deep MLP and Transformer for Small Object Detection. arXiv preprint arXiv:2505.21868 (2025)."},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00144"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i1.19893"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00610"},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01816"},{"key":"e_1_3_2_2_10_1","volume-title":"Towards large-scale small object detection: Survey and benchmarks","author":"Cheng Gong","year":"2023","unstructured":"Gong Cheng, Xiang Yuan, Xiwen Yao, Kebing Yan, Qinghua Zeng, Xingxing Xie, and Junwei Han. 2023. Towards large-scale small object detection: Survey and benchmarks. IEEE Transactions on Pattern Analysis and Machine Intelligence (2023)."},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICME46284.2020.9102793"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.169"},{"key":"e_1_3_2_2_13_1","volume-title":"Mamba: Linear-time sequence modeling with selective state spaces. arXiv preprint arXiv:2312.00752","author":"Gu Albert","year":"2023","unstructured":"Albert Gu and Tri Dao. 2023. Mamba: Linear-time sequence modeling with selective state spaces. arXiv preprint arXiv:2312.00752 (2023)."},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2022.3211006"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.322"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1007\/s00530-022-00891-0"},{"key":"e_1_3_2_2_18_1","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Hu Zhengdong","year":"2024","unstructured":"Zhengdong Hu, Yifan Sun, Jingdong Wang, and Yi Yang. 2024. DAC-DETR: Divide the attention layers and conquer. Advances in Neural Information Processing Systems, Vol. 36 (2024)."},{"key":"e_1_3_2_2_19_1","volume-title":"European Conference on Computer Vision. Springer, 290-305","author":"Huang Yi-Xin","year":"2024","unstructured":"Yi-Xin Huang, Hou-I Liu, Hong-Han Shuai, and Wen-Huang Cheng. 2024. Dq-detr: Detr with dynamic query for tiny object detection. In European Conference on Computer Vision. Springer, 290-305."},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01887"},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01325"},{"key":"e_1_3_2_2_22_1","first-page":"740","volume-title":"Zurich","author":"Lin Tsung-Yi","year":"2014","unstructured":"Tsung-Yi Lin, Michael Maire, Serge Belongie, James Hays, Pietro Perona, Deva Ramanan, Piotr Doll\u00e1r, and C Lawrence Zitnick. 2014. Microsoft coco: Common objects in context. In Computer Vision-ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6-12, 2014, Proceedings, Part V 13. Springer, 740-755."},{"key":"e_1_3_2_2_23_1","volume-title":"Dab-detr: Dynamic anchor boxes are better queries for detr. arXiv preprint arXiv:2201.12329","author":"Liu Shilong","year":"2022","unstructured":"Shilong Liu, Feng Li, Hao Zhang, Xiao Yang, Xianbiao Qi, Hang Su, Jun Zhu, and Lei Zhang. 2022. Dab-detr: Dynamic anchor boxes are better queries for detr. arXiv preprint arXiv:2201.12329 (2022)."},{"key":"e_1_3_2_2_24_1","unstructured":"Shilong Liu Tianhe Ren Jiayu Chen Zhaoyang Zeng Hao Zhang Feng Li Hongyang Li Jun Huang Hang Su Jun Zhu et al. 2023. Detection Transformer with Stable Matching. arXiv preprint arXiv:2304.04742 (2023)."},{"key":"e_1_3_2_2_25_1","volume-title":"European conference on computer vision. Springer, 38-55","author":"Liu Shilong","year":"2024","unstructured":"Shilong Liu, Zhaoyang Zeng, Tianhe Ren, Feng Li, Hao Zhang, Jie Yang, Qing Jiang, Chunyuan Li, Jianwei Yang, Hang Su, et al., 2024b. Grounding dino: Marrying dino with grounded pre-training for open-set object detection. In European conference on computer vision. Springer, 38-55."},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46448-0_2"},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2021.114602"},{"key":"e_1_3_2_2_28_1","volume-title":"Vmamba: Visual state space model. Advances in neural information processing systems","author":"Liu Yue","year":"2024","unstructured":"Yue Liu, Yunjie Tian, Yuzhong Zhao, Hongtian Yu, Lingxi Xie, Yaowei Wang, Qixiang Ye, Jianbin Jiao, and Yunfan Liu. 2024a. Vmamba: Visual state space model. Advances in neural information processing systems, Vol. 37 (2024), 103031-103063."},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2023.3251100"},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00363"},{"key":"e_1_3_2_2_32_1","volume-title":"Xingyi Zhou, and Philipp Kr\u00e4henb\u00fchl.","author":"Ouyang-Zhang Jeffrey","year":"2022","unstructured":"Jeffrey Ouyang-Zhang, Jang Hyun Cho, Xingyi Zhou, and Philipp Kr\u00e4henb\u00fchl. 2022. Nms strikes back. arXiv preprint arXiv:2212.06137 (2022)."},{"key":"e_1_3_2_2_33_1","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Pu Yifan","year":"2024","unstructured":"Yifan Pu, Weicong Liang, Yiduo Hao, Yuhui Yuan, Yukang Yang, Chao Zhang, Han Hu, and Gao Huang. 2024. Rank-DETR for high quality object detection. Advances in Neural Information Processing Systems, Vol. 36 (2024)."},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.91"},{"key":"e_1_3_2_2_35_1","volume-title":"Faster r-cnn: Towards real-time object detection with region proposal networks. Advances in neural information processing systems","author":"Ren Shaoqing","year":"2015","unstructured":"Shaoqing Ren, Kaiming He, Ross Girshick, and Jian Sun. 2015. Faster r-cnn: Towards real-time object detection with region proposal networks. Advances in neural information processing systems, Vol. 28 (2015)."},{"key":"e_1_3_2_2_36_1","unstructured":"Tianhe Ren Shilong Liu Feng Li Hao Zhang Ailing Zeng Jie Yang Xingyu Liao Ding Jia Hongyang Li He Cao et al. 2023. detrex: Benchmarking Detection Transformers. arXiv preprint arXiv:2306.07265 (2023)."},{"key":"e_1_3_2_2_37_1","volume-title":"Sparse detr: Efficient end-to-end object detection with learnable sparsity. arXiv preprint arXiv:2111.14330","author":"Roh Byungseok","year":"2021","unstructured":"Byungseok Roh, JaeWoong Shin, Wuhyun Shin, and Saehoon Kim. 2021. Sparse detr: Efficient end-to-end object detection with learnable sparsity. arXiv preprint arXiv:2111.14330 (2021)."},{"key":"e_1_3_2_2_38_1","first-page":"344","volume-title":"Padua","author":"Rukhovich Danila","year":"2021","unstructured":"Danila Rukhovich, Konstantin Sofiiuk, Danil Galeev, Olga Barinova, and Anton Konushin. 2021. Iterdet: iterative scheme for object detection in crowded environments. In Structural, Syntactic, and Statistical Pattern Recognition: Joint IAPR International Workshops, S SSPR 2020, Padua, Italy, January 21-22, 2021, Proceedings. Springer, 344-354."},{"key":"e_1_3_2_2_39_1","volume-title":"Crowdhuman: A benchmark for detecting human in a crowd. arXiv preprint arXiv:1805.00123","author":"Shao Shuai","year":"2018","unstructured":"Shuai Shao, Zijian Zhao, Boxun Li, Tete Xiao, Gang Yu, Xiangyu Zhang, and Jian Sun. 2018. Crowdhuman: A benchmark for detecting human in a crowd. arXiv preprint arXiv:1805.00123 (2018)."},{"key":"e_1_3_2_2_40_1","volume-title":"Proceedings of the IEEE\/CVF winter conference on applications of computer vision. 3531-3539","author":"Shen Zhuoran","year":"2021","unstructured":"Zhuoran Shen, Mingyuan Zhang, Haiyu Zhao, Shuai Yi, and Hongsheng Li. 2021. Efficient attention: Attention with linear complexities. In Proceedings of the IEEE\/CVF winter conference on applications of computer vision. 3531-3539."},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00605"},{"key":"e_1_3_2_2_42_1","volume-title":"Attention is all you need. Advances in neural information processing systems","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems, Vol. 30 (2017)."},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00579"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i3.20158"},{"key":"e_1_3_2_2_45_1","volume-title":"Efficient detr: improving end-to-end object detector with dense prior. arXiv preprint arXiv:2104.01318","author":"Yao Zhuyu","year":"2021","unstructured":"Zhuyu Yao, Jiangbo Ai, Boxun Li, and Chi Zhang. 2021. Efficient detr: improving end-to-end object detector with dense prior. arXiv preprint arXiv:2104.01318 (2021)."},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00423"},{"key":"e_1_3_2_2_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00102"},{"key":"e_1_3_2_2_48_1","volume-title":"Dino: Detr with improved denoising anchor boxes for end-to-end object detection. arXiv preprint arXiv:2203.03605","author":"Zhang Hao","year":"2022","unstructured":"Hao Zhang, Feng Li, Shilong Liu, Lei Zhang, Hang Su, Jun Zhu, Lionel M Ni, and Heung-Yeung Shum. 2022a. Dino: Detr with improved denoising anchor boxes for end-to-end object detection. arXiv preprint arXiv:2203.03605 (2022)."},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00708"},{"key":"e_1_3_2_2_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2019.2929005"},{"key":"e_1_3_2_2_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01611"},{"key":"e_1_3_2_2_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00614"},{"key":"e_1_3_2_2_53_1","volume-title":"Deformable detr: Deformable transformers for end-to-end object detection. arXiv preprint arXiv:2010.04159","author":"Zhu Xizhou","year":"2020","unstructured":"Xizhou Zhu, Weijie Su, Lewei Lu, Bin Li, Xiaogang Wang, and Jifeng Dai. 2020. Deformable detr: Deformable transformers for end-to-end object detection. arXiv preprint arXiv:2010.04159 (2020)."},{"key":"e_1_3_2_2_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00621"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Dublin Ireland","acronym":"MM '25"},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755045","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:26:03Z","timestamp":1765308363000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755045"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":54,"alternative-id":["10.1145\/3746027.3755045","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755045","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}