{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,4]],"date-time":"2026-08-04T15:31:31Z","timestamp":1785857491959,"version":"3.56.0"},"reference-count":84,"publisher":"Springer Science and Business Media LLC","issue":"8","license":[{"start":{"date-parts":[[2024,2,20]],"date-time":"2024-02-20T00:00:00Z","timestamp":1708387200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,2,20]],"date-time":"2024-02-20T00:00:00Z","timestamp":1708387200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2024,8]]},"DOI":"10.1007\/s11263-024-02005-x","type":"journal-article","created":{"date-parts":[[2024,2,20]],"date-time":"2024-02-20T12:02:32Z","timestamp":1708430552000},"page":"2825-2844","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":24,"title":["Semantic-Aligned Matching for Enhanced DETR Convergence and Multi-Scale Feature Fusion"],"prefix":"10.1007","volume":"132","author":[{"given":"Gongjie","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhipeng","family":"Luo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiaxing","family":"Huang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6766-2506","authenticated-orcid":false,"given":"Shijian","family":"Lu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Eric P.","family":"Xing","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,2,20]]},"reference":[{"key":"2005_CR1","doi-asserted-by":"crossref","unstructured":"Bertinetto, L., Valmadre, J., Henriques, J. F., Vedaldi, A., & Torr, P. H. (2016). Fully-convolutional siamese networks for object tracking. Eccv, 850\u2013865.","DOI":"10.1007\/978-3-319-48881-3_56"},{"key":"2005_CR2","doi-asserted-by":"publisher","first-page":"1483","DOI":"10.1109\/TPAMI.2019.2956516","volume":"43","author":"Z Cai","year":"2021","unstructured":"Cai, Z., & Vasconcelos, N. (2021). Cascade R-CNN: High quality object detection and instance segmentation. IEEE Transactions on Pattern Analysis and Machine Intelligence, 43, 1483\u20131498.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2005_CR3","doi-asserted-by":"publisher","first-page":"185","DOI":"10.1609\/aaai.v36i1.19893","volume":"36","author":"X Cao","year":"2022","unstructured":"Cao, X., Yuan, P., Feng, B., & Niu, K. (2022). CFDETR: Coarse-to-fine transformers for endto- end object detection. Aaai, 36, 185\u2013193.","journal-title":"Aaai"},{"key":"2005_CR4","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., & Zagoruyko, S. (2020). End-toend object detection with transformers. Eccv,213\u2013229.","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"2005_CR5","doi-asserted-by":"crossref","unstructured":"Chen, X., Yan, B., Zhu, J., Wang, D., Yang, X., & Lu, H. (2021). Transformer tracking. Cvpr, 8126\u20138135.","DOI":"10.1109\/CVPR46437.2021.00803"},{"key":"2005_CR6","first-page":"5621","volume":"33","author":"Y Chen","year":"2020","unstructured":"Chen, Y., Zhang, Z., Cao, Y., Wang, L., Lin, S., & Hu, H. (2020). RepPoints v2: Verification meets regression for object detection. Neurips, 33, 5621\u20135631.","journal-title":"Neurips"},{"key":"2005_CR7","doi-asserted-by":"crossref","unstructured":"Chung, D., Tahboub, K., & Delp, E. J. (2017). A two stream siamese convolutional neural network for person re-identification. Iccv, 1983\u20131991.","DOI":"10.1109\/ICCV.2017.218"},{"key":"2005_CR8","doi-asserted-by":"crossref","unstructured":"Dai, Z., Cai, B., Lin, Y., & Chen, J. (2021). UPDETR: Unsupervised pre-training for object detection with transformers. Cvpr, 1601\u20131610.","DOI":"10.1109\/CVPR46437.2021.00165"},{"key":"2005_CR9","doi-asserted-by":"crossref","unstructured":"Dai, X., Chen, Y., Yang, J., Zhang, P., Yuan, L., & Zhang, L. (2021). Dynamic DETR: End-to-end object detection with dynamic attention. Iccv, 2988\u20132997.","DOI":"10.1109\/ICCV48922.2021.00298"},{"key":"2005_CR10","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.-J., Li, K., & Fei-Fei, L. (2009). ImageNet: A large-scale hierarchical image database. Cvpr, 248\u2013255.","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"2005_CR11","doi-asserted-by":"crossref","unstructured":"Dong, X., & Shen, J. (2018). Triplet loss in siamese network for object tracking. Eccv, 459\u2013474.","DOI":"10.1007\/978-3-030-01261-8_28"},{"issue":"2","key":"2005_CR12","doi-asserted-by":"publisher","first-page":"303","DOI":"10.1007\/s11263-009-0275-4","volume":"88","author":"M Everingham","year":"2010","unstructured":"Everingham, M., Van Gool, L., Williams, C. K. I., Winn, J., & Zisserman, A. (2010). The Pascal Visual Object Classes (VOC) challenge. International Journal of Computer Vision, 88(2), 303\u2013338.","journal-title":"International Journal of Computer Vision"},{"key":"2005_CR13","doi-asserted-by":"crossref","unstructured":"Fan, Q., Zhuo, W., Tang, C.-K., & Tai, Y.-W. (2020). Few-shot object detection with attention-RPN and multi-relation detector. Cvpr, 4013\u20134022.","DOI":"10.1109\/CVPR42600.2020.00407"},{"key":"2005_CR14","doi-asserted-by":"crossref","unstructured":"Gao, P., Zheng, M., Wang, X., Dai, J., & Li, H. (2021). Fast convergence of DETR with spatially modulated co-attention. Iccv, 3621\u20133630.","DOI":"10.1109\/ICCV48922.2021.00360"},{"key":"2005_CR15","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., & Girshick, R. (2017). Mask R-CNN. Iccv, 2961\u20132969.","DOI":"10.1109\/ICCV.2017.322"},{"key":"2005_CR16","doi-asserted-by":"crossref","unstructured":"He, A., Luo, C., Tian, X., & Zeng, W. (2018). A twofold siamese network for real-time object tracking. Cvpr, 4834\u20134843.","DOI":"10.1109\/CVPR.2018.00508"},{"key":"2005_CR17","doi-asserted-by":"crossref","unstructured":"He, S., Luo, H., Wang, P., Wang, F., Li, H., & Jiang, W. (2021). TransReID: Transformerbased object re-identification. Iccv, 15013\u201315022.","DOI":"10.1109\/ICCV48922.2021.01474"},{"key":"2005_CR18","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2016). Deep residual learning for image recognition. Cvpr, 770\u2013778.","DOI":"10.1109\/CVPR.2016.90"},{"key":"2005_CR19","unstructured":"Hsieh, T.-I., Lo, Y.-C., Chen, H.-T., & Liu, T.-L. (2019). One-shot object detection with co-attention and co-excitation. Neurips, 32."},{"key":"2005_CR20","doi-asserted-by":"crossref","unstructured":"Hu, H., Gu, J., Zhang, Z., Dai, J., & Wei, Y. (2018). Relation networks for object detection. Cvpr, 3588\u20133597.","DOI":"10.1109\/CVPR.2018.00378"},{"key":"2005_CR21","doi-asserted-by":"crossref","unstructured":"Kang, B., Liu, Z., Wang, X., Yu, F., Feng, J., & Darrell, T. (2019). Few-shot object detection via feature reweighting. Iccv, 8420\u20138429.","DOI":"10.1109\/ICCV.2019.00851"},{"key":"2005_CR22","unstructured":"Kingma, D. P., & Ba, J. (2015). Adam: A method for stochastic optimization. Iclr."},{"key":"2005_CR23","unstructured":"Koch, G., Zemel, R., & Salakhutdinov, R. (2015). Siamese neural networks for one-shot image recognition. Icml Deep Learning Workshop,2."},{"key":"2005_CR24","doi-asserted-by":"crossref","unstructured":"Li, B., Wu, W., Wang, Q., Zhang, F., Xing, J., & Yan, J. (2019). SiamRPN++: Evolution of siamese visual tracking with very deep networks. Cvpr, 4282\u20134291.","DOI":"10.1109\/CVPR.2019.00441"},{"key":"2005_CR25","doi-asserted-by":"crossref","unstructured":"Li, B., Yan, J., Wu, W., Zhu, Z., & Hu, X. (2018). High performance visual tracking with siamese region proposal network. Cvpr, 8971\u20138980.","DOI":"10.1109\/CVPR.2018.00935"},{"key":"2005_CR26","doi-asserted-by":"crossref","unstructured":"Li, F., Zhang, H., Liu, S., Guo, J., Ni, L. M., & Zhang, L. (2022). DN-DETR: Accelerate DETR training by introducing query denoising. Cvpr, 13619\u201313627.","DOI":"10.1109\/CVPR52688.2022.01325"},{"key":"2005_CR27","doi-asserted-by":"crossref","unstructured":"Li, F., Zhang, H., Xu, H., Liu, S., Zhang, L., Ni, L. M., & Shum, H.-Y. (2023). Mask DINO: Towards a unified transformer-based framework for object detection and segmentation. Cvpr, 3041\u20133050.","DOI":"10.1109\/CVPR52729.2023.00297"},{"issue":"2","key":"2005_CR28","doi-asserted-by":"publisher","first-page":"532","DOI":"10.1109\/TPAMI.2019.2937086","volume":"43","author":"M Liao","year":"2021","unstructured":"Liao, M., Lyu, P., He, M., Yao, C., Wu, W., & Bai, X. (2021). Mask TextSpotter: An end-toend trainable neural network for spotting text with arbitrary shapes. IEEE Transactions on Pattern Analysis and Machine Intelligence, 43(2), 532\u2013548.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2005_CR29","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., Doll\u00e1r, P., Girshick, R., He, K., Hariharan, B., & Belongie, S. (2017). Feature pyramid networks for object detection. Cvpr, 2117\u20132125.","DOI":"10.1109\/CVPR.2017.106"},{"key":"2005_CR30","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., Goyal, P., Girshick, R., He, K., & Doll\u00e1r, P. (2017). Focal loss for dense object detection. Iccv, 2980\u20132988.","DOI":"10.1109\/ICCV.2017.324"},{"key":"2005_CR31","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., Maire, M., Belongie, S. J., Bourdev, L. D., Girshick, R. B., Hays, J., Perona, P., Ramanan, D., Doll\u00e1r, P., & Zitnick, C. L. (2014). Microsoft COCO: Common objects in context. Eccv, 740\u2013755.","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"2005_CR32","doi-asserted-by":"crossref","unstructured":"Liu, W., Anguelov, D., Erhan, D., Szegedy, C., Reed, S., Fu, C.-Y., & Berg, A. C. (2016). SSD: Single shot multibox detector. Eccv, 21\u201337.","DOI":"10.1007\/978-3-319-46448-0_2"},{"key":"2005_CR33","doi-asserted-by":"crossref","unstructured":"Liu, S., Huang, D., & Wang, Y. (2018). Receptive field block net for accurate and fast object detection. Eccv, 385\u2013400.","DOI":"10.1007\/978-3-030-01252-6_24"},{"key":"2005_CR34","unstructured":"Liu, S., Li, F., Zhang, H., Yang, X., Qi, X., Su, H., Zhu, J., & Zhang, L. (2022). DAB-DETR: Dynamic anchor boxes are better queries for DETR. Iclr."},{"key":"2005_CR35","doi-asserted-by":"publisher","first-page":"261","DOI":"10.1007\/s11263-019-01247-4","volume":"128","author":"L Liu","year":"2020","unstructured":"Liu, L., Ouyang, W., Wang, X., Fieguth, P., Chen, J., Liu, X., & Pietik\u00e4inen, M. (2020). Deep learning for generic object detection: A survey. International Journal of Computer Vision, 128, 261\u2013318.","journal-title":"International Journal of Computer Vision"},{"key":"2005_CR36","unstructured":"Loshchilov, I., & Hutter, F. (2019). Decoupled weight decay regularization. Iclr."},{"key":"2005_CR37","doi-asserted-by":"crossref","unstructured":"Meng, D., Chen, X., Fan, Z., Zeng, G., Li, H., Yuan, Y., Sun, L., & Wang, J. (2021). Conditional DETR for fast training convergence. Iccv, 3651\u20133660.","DOI":"10.1109\/ICCV48922.2021.00363"},{"key":"2005_CR38","doi-asserted-by":"crossref","unstructured":"Pang, J., Chen, K., Shi, J., Feng, H., Ouyang, W., & Lin, D. (2019). Libra R-CNN: Towards balanced learning for object detection. Cvpr, 821\u2013830.","DOI":"10.1109\/CVPR.2019.00091"},{"key":"2005_CR39","doi-asserted-by":"crossref","unstructured":"Perez-Rua, J.-M., Zhu, X., Hospedales, T. M., & Xiang, T. (2020). Incremental few-shot object detection. Cvpr, 13846\u201313855.","DOI":"10.1109\/CVPR42600.2020.01386"},{"key":"2005_CR40","doi-asserted-by":"crossref","unstructured":"Redmon, J., & Farhadi, A. (2017). YOLO 9000: Better, faster, stronger. Cvpr, 7263\u20137271.","DOI":"10.1109\/CVPR.2017.690"},{"key":"2005_CR41","doi-asserted-by":"publisher","first-page":"1137","DOI":"10.1109\/TPAMI.2016.2577031","volume":"39","author":"S Ren","year":"2015","unstructured":"Ren, S., He, K., Girshick, R. B., & Sun, J. (2015). Faster R-CNN: Towards real-time object detection with region proposal networks. IEEE Transactions on Pattern Analysis and Machine Intelligence, 39, 1137\u20131149.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2005_CR42","unstructured":"Roh, B., Shin, J., Shin, W., & Kim, S. (2022). Sparse DETR: Efficient end-to-end object detection with learnable sparsity. Iclr."},{"key":"2005_CR43","doi-asserted-by":"crossref","unstructured":"Schroff, F., Kalenichenko, D., & Philbin, J. (2015). FaceNet: A unified embedding for face recognition and clustering. Cvpr, 815\u2013823.","DOI":"10.1109\/CVPR.2015.7298682"},{"key":"2005_CR44","doi-asserted-by":"crossref","unstructured":"Shen, C., Jin, Z., Zhao, Y., Fu, Z., Jiang, R., Chen, Y., & Hua, X.-S. (2017). Deep siamese network with multi-level similarity perception for person re-identification. Acm mm, 1942\u20131950.","DOI":"10.1145\/3123266.3123452"},{"key":"2005_CR45","doi-asserted-by":"crossref","unstructured":"Shen, Y., Xiao, T., Li, H., Yi, S., & Wang, X. (2017). Learning deep neural networks for vehicle Re-ID with visual-spatio-temporal path proposals. Iccv, 1900\u20131909.","DOI":"10.1109\/ICCV.2017.210"},{"key":"2005_CR46","unstructured":"Snell, J., Swersky, K., & Zemel, R. (2017). Prototypical networks for few-shot learning. Neurips,30."},{"key":"2005_CR47","doi-asserted-by":"crossref","unstructured":"Song, L., Gong, D., Li, Z., Liu, C., & Liu, W. (2019). Occlusion robust face recognition based on mask learning with pairwise differential siamese network. Iccv, 773\u2013782.","DOI":"10.1109\/ICCV.2019.00086"},{"key":"2005_CR48","unstructured":"Song, H., Sun, D., Chun, S., Jampani, V., Han, D., Heo, B., Kim, W., & Yang, M.-H. (2022). ViDT: An efficient and effective fully transformerbased object detector. Iclr."},{"issue":"1","key":"2005_CR49","first-page":"1929","volume":"15","author":"N Srivastava","year":"2014","unstructured":"Srivastava, N., Hinton, G., Krizhevsky, A., Sutskever, I., & Salakhutdinov, R. (2014). Dropout: A simple way to prevent neural networks from overfitting. The Journal of Machine Learning Research (JMLR), 15(1), 1929\u20131958.","journal-title":"The Journal of Machine Learning Research (JMLR)"},{"key":"2005_CR50","doi-asserted-by":"crossref","unstructured":"Sun, Z., Cao, S., Yang, Y., & Kitani, K. M. (2021). Rethinking Transformer-based set prediction for object detection. Iccv, 3611\u20133620.","DOI":"10.1109\/ICCV48922.2021.00359"},{"key":"2005_CR51","doi-asserted-by":"crossref","unstructured":"Sun, P., Zhang, R., Jiang, Y., Kong, T., Xu, C., Zhan, W., Tomizuka, M., Li, L., Yuan, Z., Wang, C., & Luo, P. (2021). Sparse R-CNN: End-to-end object detection with learnable proposals. Cvpr, 14454\u201314463.","DOI":"10.1109\/CVPR46437.2021.01422"},{"key":"2005_CR52","doi-asserted-by":"crossref","unstructured":"Sung, F., Yang, Y., Zhang, L., Xiang, T., Torr, P. H., & Hospedales, T. M. (2018). Learning to compare: Relation network for few-shot learning. Cvpr, 1199\u20131208.","DOI":"10.1109\/CVPR.2018.00131"},{"key":"2005_CR53","doi-asserted-by":"crossref","unstructured":"Tan, M., Pang, R., & Le, Q. V. (2020). EfficientDet: Scalable and efficient object detection. Cvpr, 10781\u201310790.","DOI":"10.1109\/CVPR42600.2020.01079"},{"key":"2005_CR54","doi-asserted-by":"crossref","unstructured":"Tao, R., Gavves, E., & Smeulders, A. W. (2016). Siamese instance search for tracking. Cvpr, 1420\u20131429.","DOI":"10.1109\/CVPR.2016.158"},{"key":"2005_CR55","doi-asserted-by":"crossref","unstructured":"Tian, Z., Shen, C., Chen, H., & He, T. (2019). FCOS: Fully convolutional one-stage object detection. Iccv, 9626\u20139635.","DOI":"10.1109\/ICCV.2019.00972"},{"key":"2005_CR56","doi-asserted-by":"crossref","unstructured":"Tychsen-Smith, L., & Petersson, L. (2018). Improving object localization with fitness NMS and bounded IoU loss. Cvpr, 6877\u20136885.","DOI":"10.1109\/CVPR.2018.00719"},{"key":"2005_CR57","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A. N., Kaiser, L., & Polosukhin, I. (2017). Attention is all you need. Neurips,30."},{"key":"2005_CR58","doi-asserted-by":"crossref","unstructured":"Voigtlaender, P., Luiten, J., Torr, P. H., & Leibe, B. (2020). Siam R-CNN: Visual tracking by redetection. Cvpr, 6578\u20136588.","DOI":"10.1109\/CVPR42600.2020.00661"},{"key":"2005_CR59","unstructured":"Wang, X., Huang, T. E., Darrell, T., Gonzalez, J. E., & Yu, F. (2020). Frustratingly simple few-shot object detection. Icml."},{"key":"2005_CR60","doi-asserted-by":"crossref","unstructured":"Wang, T., Yuan, L., Chen, Y., Feng, J., & Yan, S. (2021). PnP-DETR: Towards efficient visual analysis with Transformers. Iccv, 4661\u20134670.","DOI":"10.1109\/ICCV48922.2021.00462"},{"key":"2005_CR61","doi-asserted-by":"crossref","unstructured":"Wang, N., Zhou, W., Wang, J., & Li, H. (2021). Transformer meets tracker: Exploiting temporal context for robust visual tracking. Cvpr, 1571\u20131580.","DOI":"10.1109\/CVPR46437.2021.00162"},{"key":"2005_CR62","doi-asserted-by":"publisher","first-page":"2567","DOI":"10.1609\/aaai.v36i3.20158","volume":"36","author":"Y Wang","year":"2022","unstructured":"Wang, Y., Zhang, X., Yang, T., & Sun, J. (2022). Anchor DETR: Query design for transformer-based detector. Aaai, 36, 2567\u20132575.","journal-title":"Aaai"},{"key":"2005_CR63","doi-asserted-by":"crossref","unstructured":"Wu, R., Zhang, G., Lu, S., & Chen, T. (2020). Cascade EF-GAN: Progressive facial expression editing with local focuses. Cvpr, 5020\u20135029.","DOI":"10.1109\/CVPR42600.2020.00507"},{"issue":"6","key":"2005_CR64","doi-asserted-by":"publisher","first-page":"1412","DOI":"10.1109\/TMM.2018.2877886","volume":"21","author":"L Wu","year":"2018","unstructured":"Wu, L., Wang, Y., Gao, J., & Li, X. (2018). Where-and-when to look: Deep siamese attention networks for video-based person re-identification. IEEE Transactions on Multimedia, 21(6), 1412\u20131424.","journal-title":"IEEE Transactions on Multimedia"},{"key":"2005_CR65","doi-asserted-by":"crossref","unstructured":"Xiao, Y., & Marlet, R. (2020). Few-shot object detection and viewpoint estimation for objects in the wild. Eccv, 192\u2013210.","DOI":"10.1007\/978-3-030-58520-4_12"},{"key":"2005_CR66","doi-asserted-by":"crossref","unstructured":"Yan, X., Chen, Z., Xu, A., Wang, X., Liang, X., & Lin, L. (2019). Meta R-CNN: Towards general solver for instance-level low-shot learning. Iccv, 9577\u20139586.","DOI":"10.1109\/ICCV.2019.00967"},{"key":"2005_CR67","doi-asserted-by":"crossref","unstructured":"Yang, Z., Liu, S., Hu, H., Wang, L., & Lin, S. (2019). RepPoints: Point set representation for object detection. Iccv, 9657\u20139666.","DOI":"10.1109\/ICCV.2019.00975"},{"key":"2005_CR68","unstructured":"Yao, Z., Ai, J., Li, B., & Zhang, C. (2021). Efficient DETR: improving end-to-end object detector with dense prior. arXiv:2104.01318 ."},{"key":"2005_CR69","doi-asserted-by":"crossref","unstructured":"Zhang, Z., & Peng, H. (2019). Deeper and wider siamese networks for real-time visual tracking. Cvpr, 4591\u20134600.","DOI":"10.1109\/CVPR.2019.00472"},{"key":"2005_CR70","doi-asserted-by":"crossref","unstructured":"Zhang, G., Cui, K., Wu, R., Lu, S., & Tian, Y. (2021). PNPDet: Efficient few-shot detection without forgetting via plug-and-play sub-networks. Wacv, 3822\u20133831.","DOI":"10.1109\/WACV48630.2021.00387"},{"key":"2005_CR71","unstructured":"Zhang, H., Li, F., Liu, S., Zhang, L., Su, H., Zhu, J., Ni, L. M., & Shum, H.-Y. (2023). DINO: DETR with improved denoising anchor boxes for end-to-end object detection. Iclr."},{"key":"2005_CR72","unstructured":"Zhang, G., Lin, J., Wu, S., Song, Y., Luo, Z., Xue, Y., Lu, S., & Wang, Z. (2023). Online map vectorization for autonomous driving: A rasterization perspective. Neurips. Retrieved from https:\/\/openreview.net\/forum?id=YvO5yTVv5Y"},{"key":"2005_CR73","doi-asserted-by":"crossref","unstructured":"Zhang, G., Luo, Z., Cui, K., & Lu, S. (2021). Meta- DETR: Image-level few-shot object detection with inter-class correlation exploitation. arXiv:2103.11731 .","DOI":"10.1109\/TPAMI.2022.3195735"},{"key":"2005_CR74","doi-asserted-by":"crossref","unstructured":"Zhang, G., Luo, Z., Tian, Z., Zhang, J., Zhang, X., & Lu, S. (2023). Towards efficient use of multi-scale features in transformer-based object detectors. Cvpr, 6206\u20136216.","DOI":"10.1109\/CVPR52729.2023.00601"},{"key":"2005_CR75","doi-asserted-by":"crossref","unstructured":"Zhang, G., Luo, Z., Yu, Y., Cui, K., & Lu, S. (2022). Accelerating DETR convergence via semantic-aligned matching. Cvpr, 949\u2013958.","DOI":"10.1109\/CVPR52688.2022.00102"},{"key":"2005_CR76","doi-asserted-by":"crossref","unstructured":"Zhang, S., Wen, L., Bian, X., Lei, Z., & Li, S. Z. (2018). Single-shot refinement neural network for object detection. Cvpr, 4203\u20134212.","DOI":"10.1109\/CVPR.2018.00442"},{"issue":"11","key":"2005_CR77","doi-asserted-by":"publisher","first-page":"12832","DOI":"10.1109\/TPAMI.2022.3195735","volume":"45","author":"G Zhang","year":"2023","unstructured":"Zhang, G., Luo, Z., Cui, K., Lu, S., & Xing, E. P. (2023). Meta-DETR: Image-level fewshot detection with inter-class correlation exploitation. IEEE Transactions on Pattern Analysis and Machine Intelligence, 45(11), 12832\u201312843. https:\/\/doi.org\/10.1109\/TPAMI.2022.3195735","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2005_CR78","doi-asserted-by":"crossref","unstructured":"Zhang, G., Lu, S., & Zhang, W. (2019). CAD-Net: A context-aware detection network for objects in remote sensing imagery. IEEE Transactions on Geoscience and Remote Sensing, 57(12), 10015\u201310024.","DOI":"10.1109\/TGRS.2019.2930982"},{"key":"2005_CR79","doi-asserted-by":"publisher","first-page":"9259","DOI":"10.1609\/aaai.v33i01.33019259","volume":"33","author":"Q Zhao","year":"2019","unstructured":"Zhao, Q., Sheng, T., Wang, Y., Tang, Z., Chen, Y., Cai, L., & Ling, H. (2019). M2Det: A singleshot object detector based on multi-level feature pyramid network. Aaai, 33, 9259\u20139266.","journal-title":"Aaai"},{"key":"2005_CR80","doi-asserted-by":"crossref","unstructured":"Zheng, M., Karanam, S., Wu, Z., & Radke, R. J. (2019). Re-identification with consistent attentive siamese networks. Cvpr, 5735\u20135744.","DOI":"10.1109\/CVPR.2019.00588"},{"key":"2005_CR81","unstructured":"Zhou, X., Wang, D., & Kr\u00e4henb\u00fchl, P. (2019). Objects as points. arxiv:1904.07850."},{"key":"2005_CR82","doi-asserted-by":"crossref","unstructured":"Zhou, X., Zhuo, J., & Krahenbuhl, P. (2019). Bottom-up object detection by grouping extreme and center points. Cvpr, 850\u2013859","DOI":"10.1109\/CVPR.2019.00094"},{"key":"2005_CR83","unstructured":"Zhu, X., Su, W., Lu, L., Li, B., Wang, X., & Dai, J. (2021). Deformable DETR: Deformable transformers for end-to-end object detection. Iclr."},{"key":"2005_CR84","doi-asserted-by":"crossref","unstructured":"Zhu, Z., Wang, Q., Li, B., Wu, W., Yan, J., & Hu, W. (2018). Distractor-aware siamese networks for visual object tracking. Eccv, 101\u2013117.","DOI":"10.1007\/978-3-030-01240-3_7"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-024-02005-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-024-02005-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-024-02005-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,7,11]],"date-time":"2024-07-11T14:10:52Z","timestamp":1720707052000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-024-02005-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,2,20]]},"references-count":84,"journal-issue":{"issue":"8","published-print":{"date-parts":[[2024,8]]}},"alternative-id":["2005"],"URL":"https:\/\/doi.org\/10.1007\/s11263-024-02005-x","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,2,20]]},"assertion":[{"value":"13 January 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 January 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"20 February 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}