{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,18]],"date-time":"2026-04-18T06:39:10Z","timestamp":1776494350483,"version":"3.51.2"},"reference-count":88,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T00:00:00Z","timestamp":1772755200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T00:00:00Z","timestamp":1772755200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62071333"],"award-info":[{"award-number":["62071333"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["22120220654"],"award-info":[{"award-number":["22120220654"]}],"id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Tongji University national artificial intelligence production-education integration innovation project"},{"DOI":"10.13039\/100017329","name":"Foundation of Science and Technology on Near-Surface Detection Laboratory","doi-asserted-by":"publisher","award":["6142414221606"],"award-info":[{"award-number":["6142414221606"]}],"id":[{"id":"10.13039\/100017329","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2026,4]]},"DOI":"10.1007\/s11263-026-02733-2","type":"journal-article","created":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T09:59:30Z","timestamp":1772791170000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Exploiting Unlabeled Data with Multiple Expert Teachers for Open Vocabulary Aerial Object Detection and Its Orientation Adaptation"],"prefix":"10.1007","volume":"134","author":[{"given":"Yan","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5037-0972","authenticated-orcid":false,"given":"Weiwei","family":"Guo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xue","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ning","family":"Liao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shaofeng","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yi","family":"Yu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wenxian","family":"Yu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Junchi","family":"Yan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,3,6]]},"reference":[{"key":"2733_CR1","doi-asserted-by":"crossref","unstructured":"Azimi, S. M., Vig, E., Bahmanyar, R., K\u00f6rner, M., & Reinartz, P. (2018). Towards multi-class object detection in unconstrained remote sensing imagery. In: Asian Conference on Computer Vision, pp. 150\u2013165. Springer.","DOI":"10.1007\/978-3-030-20893-6_10"},{"key":"2733_CR2","doi-asserted-by":"crossref","unstructured":"Bilen, H., & Vedaldi, A. (2016). Weakly supervised deep detection networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2846\u20132854.","DOI":"10.1109\/CVPR.2016.311"},{"key":"2733_CR3","doi-asserted-by":"crossref","unstructured":"Cai, Z., & Vasconcelos, N. (2018). Cascade r-cnn: Delving into high quality object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6154\u20136162.","DOI":"10.1109\/CVPR.2018.00644"},{"key":"2733_CR4","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., & Zagoruyko, S. (2020). End-to-end object detection with transformers. In: European Conference on Computer Vision, pp. 213\u2013229. Springer.","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"2733_CR5","doi-asserted-by":"crossref","unstructured":"Chen, K., Jiang, X., Hu, Y., Tang, X., Gao, Y., Chen, J., & Xie, W. (2023). Ovarnet: Towards open-vocabulary object attribute recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 23518\u201323527.","DOI":"10.1109\/CVPR52729.2023.02252"},{"key":"2733_CR6","unstructured":"Chen, K., Wang, J., Pang, J., Cao, Y., Xiong, Y., Li, X., Sun, S., Feng, W., Liu, Z., Xu, J., Zhang, Z., Cheng, D., Zhu, C., Cheng, T., Zhao, Q., Li, B., Lu, X., Zhu, R., Wu, Y., Dai, J., Wang, J., Shi, J., Ouyang, W., Loy, C. C., & Lin, D. (2019). MMDetection: Open mmlab detection toolbox and benchmark. arXiv preprint arXiv:1906.07155."},{"key":"2733_CR7","doi-asserted-by":"crossref","unstructured":"Cheng, T., Song, L., Ge, Y., Liu, W., Wang, X., & Shan, Y. (2024). Yolo-world: Real-time open-vocabulary object detection. In: Proc. IEEE Conf. Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR52733.2024.01599"},{"issue":"10","key":"2733_CR8","doi-asserted-by":"publisher","first-page":"1865","DOI":"10.1109\/JPROC.2017.2675998","volume":"105","author":"G Cheng","year":"2017","unstructured":"Cheng, G., Han, J., & Lu, X. (2017). Remote sensing image scene classification: Benchmark and state of the art. Proceedings of the IEEE, 105(10), 1865\u20131883. https:\/\/doi.org\/10.1109\/JPROC.2017.2675998","journal-title":"Proceedings of the IEEE"},{"issue":"12","key":"2733_CR9","doi-asserted-by":"publisher","first-page":"7405","DOI":"10.1109\/TGRS.2016.2601622","volume":"54","author":"G Cheng","year":"2016","unstructured":"Cheng, G., Zhou, P., & Han, J. (2016). Learning rotation-invariant convolutional neural networks for object detection in vhr optical remote sensing images. IEEE Transactions on Geoscience and Remote Sensing, 54(12), 7405\u20137415.","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"2733_CR10","doi-asserted-by":"publisher","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.-J., Li, K., & Fei-Fei, L. (2009). Imagenet: A large-scale hierarchical image database. In: 2009 IEEE Conference on Computer Vision and Pattern Recognition, pp. 248\u2013255. https:\/\/doi.org\/10.1109\/CVPR.2009.5206848.","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"2733_CR11","doi-asserted-by":"crossref","unstructured":"Ding, J., Xue, N., Long, Y., Xia, G.-S., & Lu, Q. (2019). Learning roi transformer for oriented object detection in aerial images. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2849\u20132858.","DOI":"10.1109\/CVPR.2019.00296"},{"key":"2733_CR12","doi-asserted-by":"crossref","unstructured":"Du, Y., Wei, F., Zhang, Z., Shi, M., Gao, Y., & Li, G. (2022). Learning to prompt for open-vocabulary object detection with vision-language model. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14084\u201314093.","DOI":"10.1109\/CVPR52688.2022.01369"},{"key":"2733_CR13","doi-asserted-by":"crossref","unstructured":"Feng, C., Zhong, Y., Jie, Z., Chu, X., Ren, H., Wei, X., Xie, W., & Ma, L. (2022). Promptdet: Towards open-vocabulary detection using uncurated images. In: European Conference on Computer Vision, pp. 701\u2013717. Springer.","DOI":"10.1007\/978-3-031-20077-9_41"},{"key":"2733_CR14","doi-asserted-by":"crossref","unstructured":"Girshick, R. (2015). Fast r-cnn. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 1440\u20131448.","DOI":"10.1109\/ICCV.2015.169"},{"key":"2733_CR15","doi-asserted-by":"crossref","unstructured":"Girshick, R., Donahue, J., Darrell, T., & Malik, J. (2014). Rich feature hierarchies for accurate object detection and semantic segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 580\u2013587.","DOI":"10.1109\/CVPR.2014.81"},{"key":"2733_CR16","unstructured":"Gu, X., Lin, T.-Y., Kuo, W., & Cui, Y. (2021). Open-vocabulary object detection via vision and language knowledge distillation. arXiv preprint arXiv:2104.13921."},{"key":"2733_CR17","doi-asserted-by":"crossref","unstructured":"Gupta, A., Dollar, P., & Girshick, R. (2019). Lvis: A dataset for large vocabulary instance segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5356\u20135364.","DOI":"10.1109\/CVPR.2019.00550"},{"key":"2733_CR18","doi-asserted-by":"crossref","unstructured":"Han, J., Ding, J., Xue, N., & Xia, G.-S. (2021). Redet: A rotation-equivariant detector for aerial object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 2786\u20132795.","DOI":"10.1109\/CVPR46437.2021.00281"},{"key":"2733_CR19","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2021.3062048","author":"J Han","year":"2021","unstructured":"Han, J., Ding, J., Li, J., & Xia, G.-S. (2021). Align deep features for oriented object detection. IEEE Transactions on Geoscience and Remote Sensing. https:\/\/doi.org\/10.1109\/TGRS.2021.3062048","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"2733_CR20","doi-asserted-by":"crossref","unstructured":"He, K., Fan, H., Wu, Y., Xie, S., & Girshick, R. (2020). Momentum contrast for unsupervised visual representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9729\u20139738.","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"2733_CR21","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., & Girshick, R. (2017). Mask r-cnn. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2961\u20132969.","DOI":"10.1109\/ICCV.2017.322"},{"key":"2733_CR22","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2016). Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778.","DOI":"10.1109\/CVPR.2016.90"},{"key":"2733_CR23","unstructured":"Jeong, J., Lee, S., Kim, J., & Kwak, N. (2019). Consistency-based semi-supervised learning for object detection. Advances in neural information processing systems,32."},{"key":"2733_CR24","doi-asserted-by":"crossref","unstructured":"Kang, M., Leng, X., Lin, Z., & Ji, K. (2017). A modified faster r-cnn based on cfar algorithm for sar ship detection. In: 2017 International Workshop on Remote Sensing with Intelligent Processing (RSIP), pp. 1\u20134. IEEE.","DOI":"10.1109\/RSIP.2017.7958815"},{"key":"2733_CR25","doi-asserted-by":"crossref","unstructured":"Law, H., & Deng, J. (2018). Cornernet: Detecting objects as paired keypoints. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 734\u2013750.","DOI":"10.1007\/978-3-030-01264-9_45"},{"key":"2733_CR26","unstructured":"Lee, H., Song, M., Koo, J., & Seo, J. (2023). Hausdorff distance matching with adaptive query denoising for rotated detection transformer."},{"key":"2733_CR27","doi-asserted-by":"crossref","unstructured":"Li, Y., Guo, W., Yang, X., Liao, N., He, D., Zhou, J., & Yu, W. (2025). Toward open vocabulary aerial object detection with clip-activated student-teacher learning. In: European Conference on Computer Vision, pp. 431\u2013448. Springer.","DOI":"10.1007\/978-3-031-73016-0_25"},{"key":"2733_CR28","doi-asserted-by":"publisher","unstructured":"Li, L. H., Zhang, P., Zhang, H., Yang, J., Li, C., Zhong, Y., Wang, L., Yuan, L., Zhang, L., Hwang, J.-N., Chang, K.-W., & Gao, J. (2022). Grounded language-image pre-training. In: 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 10955\u201310965. https:\/\/doi.org\/10.1109\/CVPR52688.2022.01069.","DOI":"10.1109\/CVPR52688.2022.01069"},{"issue":"12","key":"2733_CR29","doi-asserted-by":"publisher","first-page":"11507","DOI":"10.1109\/TPAMI.2024.3393024","volume":"46","author":"Y Li","year":"2024","unstructured":"Li, Y., Luo, J., Zhang, Y., Tan, Y., Yu, J.-G., & Bai, S. (2024). Learning to holistically detect bridges from large-size vhr remote sensing imagery. IEEE Transactions on Pattern Analysis and Machine Intelligence, 46(12), 11507\u201311523. https:\/\/doi.org\/10.1109\/TPAMI.2024.3393024","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2733_CR30","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., Goyal, P., Girshick, R., He, K., & Doll\u00e1r, P. (2017). Focal loss for dense object detection. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2980\u20132988.","DOI":"10.1109\/ICCV.2017.324"},{"key":"2733_CR31","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., Maire, M., Belongie, S., Hays, J., Perona, P., Ramanan, D., Doll\u00e1r, P., & Zitnick, C. L. (2014). Microsoft coco: Common objects in context. In: Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6-12, 2014, Proceedings, Part V 13, pp. 740\u2013755. Springer.","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"2733_CR32","doi-asserted-by":"crossref","unstructured":"Liu, W., Anguelov, D., Erhan, D., Szegedy, C., Reed, S., Fu, C.-Y., & Berg, A. C. (2016). Ssd: Single shot multibox detector. In: Computer Vision\u2013ECCV 2016: 14th European Conference, Amsterdam, The Netherlands, October 11\u201314, 2016, Proceedings, Part I 14, pp. 21\u201337. Springer.","DOI":"10.1007\/978-3-319-46448-0_2"},{"key":"2733_CR33","doi-asserted-by":"crossref","unstructured":"Liu, F., Chen, D., Guan, Z., Zhou, X., Zhu, J., & Zhou, J. (2023). Remoteclip: A vision language foundation model for remote sensing. arXiv preprint arXiv:2306.11029.","DOI":"10.1109\/TGRS.2024.3390838"},{"key":"2733_CR34","unstructured":"Liu, Y.-C., Ma, C.-Y., He, Z., Kuo, C.-W., Chen, K., Zhang, P., Wu, B., Kira, Z., & Vajda, P. (2021). Unbiased teacher for semi-supervised object detection. arXiv preprint arXiv:2102.09480."},{"key":"2733_CR35","doi-asserted-by":"crossref","unstructured":"Liu, S., Zeng, Z., Ren, T., Li, F., Zhang, H., Yang, J., Li, C., Yang, J., Su, H., Zhu, J., and others. (2023). Grounding dino: Marrying dino with grounded pre-training for open-set object detection. arXiv preprint arXiv:2303.05499.","DOI":"10.1007\/978-3-031-72970-6_3"},{"key":"2733_CR36","doi-asserted-by":"crossref","unstructured":"Liu, Y., Zhang, M.-h., Xu, P., & Guo, Z.-w. (2017). Sar ship detection using sea-land segmentation-based convolutional neural network. In: 2017 International Workshop on Remote Sensing with Intelligent Processing (RSIP), pp. 1\u20134. IEEE.","DOI":"10.1109\/RSIP.2017.7958806"},{"key":"2733_CR37","doi-asserted-by":"publisher","first-page":"296","DOI":"10.1016\/j.isprsjprs.2019.11.023","volume":"159","author":"K Li","year":"2020","unstructured":"Li, K., Wan, G., Cheng, G., Meng, L., & Han, J. (2020). Object detection in optical remote sensing images: A survey and a new benchmark. ISPRS journal of photogrammetry and remote sensing, 159, 296\u2013307.","journal-title":"ISPRS journal of photogrammetry and remote sensing"},{"issue":"3","key":"2733_CR38","doi-asserted-by":"publisher","first-page":"1832","DOI":"10.1109\/TPAMI.2024.3508072","volume":"47","author":"Y Li","year":"2025","unstructured":"Li, Y., Wang, L., Wang, T., Yang, X., Luo, J., Wang, Q., Deng, Y., Wang, W., Sun, X., Li, H., Dang, B., Zhang, Y., Yu, Y., & Yan, J. (2025). Star: A first-ever dataset and a large-scale benchmark for scene graph generation in large-size satellite imagery. IEEE Transactions on Pattern Analysis and Machine Intelligence, 47(3), 1832\u20131849. https:\/\/doi.org\/10.1109\/TPAMI.2024.3508072","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2733_CR39","unstructured":"Machine\u00a0Learning, A., & Mining, D. (2023). Zero-shot Object Detection Challenge. http:\/\/aiskyeye.com\/challenge-2023\/zero-shot-object-detection\/, Last accessed on 2023-11-09."},{"key":"2733_CR40","doi-asserted-by":"crossref","unstructured":"Pan, J., Liu, Y., Fu, Y., Ma, M., Li, J., Paudel, D.P., Van\u00a0Gool, L., & Huang, X. (2024). Locate anything on earth: Advancing open-vocabulary object detection for remote sensing community. arXiv preprint arXiv:2408.09110.","DOI":"10.1609\/aaai.v39i6.32672"},{"issue":"11","key":"2733_CR41","doi-asserted-by":"publisher","first-page":"7869","DOI":"10.1109\/TCSVT.2022.3186070","volume":"32","author":"W Qian","year":"2022","unstructured":"Qian, W., Yang, X., Peng, S., Zhang, X., & Yan, J. (2022). Rsdet++: Point-based modulated loss for more accurate rotated object detection. IEEE Transactions on Circuits and Systems for Video Technology, 32(11), 7869\u20137879.","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"2733_CR42","unstructured":"Radford, A., Kim, J. W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., and others. (2021). Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PMLR"},{"key":"2733_CR43","doi-asserted-by":"crossref","unstructured":"Rahman, S., Khan, S., & Porikli, F. (2018). Zero-shot object detection: Learning to simultaneously recognize and localize novel concepts. In: Asian Conference on Computer Vision, pp. 547\u2013563. Springer.","DOI":"10.1007\/978-3-030-20887-5_34"},{"key":"2733_CR44","doi-asserted-by":"crossref","unstructured":"Redmon, J., Divvala, S., Girshick, R., & Farhadi, A. (2016). You only look once: Unified, real-time object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 779\u2013788.","DOI":"10.1109\/CVPR.2016.91"},{"key":"2733_CR45","doi-asserted-by":"crossref","unstructured":"Reilly, V., Idrees, H., & Shah, M. (2010). Detection and tracking of large number of targets in wide area surveillance. In: Computer Vision\u2013ECCV 2010: 11th European Conference on Computer Vision, Heraklion, Crete, Greece, September 5-11, 2010, Proceedings, Part III 11, pp. 186\u2013199. Springer.","DOI":"10.1007\/978-3-642-15558-1_14"},{"key":"2733_CR46","unstructured":"Ren, S., He, K., Girshick, R., & Sun, J. (2015). Faster r-cnn: Towards real-time object detection with region proposal networks. Advances in neural information processing systems,28."},{"key":"2733_CR47","doi-asserted-by":"publisher","first-page":"183","DOI":"10.1016\/j.compind.2018.03.014","volume":"98","author":"EJ Sadgrove","year":"2018","unstructured":"Sadgrove, E. J., Falzon, G., Miron, D., & Lamb, D. W. (2018). Real-time object detection in agricultural\/remote environments using the multiple-expert colour feature extreme learning machine (mec-elm). Computers in Industry, 98, 183\u2013191.","journal-title":"Computers in Industry"},{"key":"2733_CR48","unstructured":"Sohn, K., Zhang, Z., Li, C.-L., Zhang, H., Lee, C.-Y., & Pfister, T. (2020). A simple semi-supervised learning framework for object detection. arXiv preprint arXiv:2005.04757."},{"key":"2733_CR49","first-page":"596","volume":"33","author":"K Sohn","year":"2020","unstructured":"Sohn, K., Berthelot, D., Carlini, N., Zhang, Z., Zhang, H., Raffel, C. A., Cubuk, E. D., Kurakin, A., & Li, C.-L. (2020). Fixmatch: Simplifying semi-supervised learning with consistency and confidence. Advances in neural information processing systems, 33, 596\u2013608.","journal-title":"Advances in neural information processing systems"},{"key":"2733_CR50","doi-asserted-by":"publisher","unstructured":"Sommer, L. W., Schuchert, T., & Beyerer, J. (2017). Fast deep vehicle detection in aerial images. In: 2017 IEEE Winter Conference on Applications of Computer Vision (WACV), pp. 311\u2013319. https:\/\/doi.org\/10.1109\/WACV.2017.41","DOI":"10.1109\/WACV.2017.41"},{"key":"2733_CR51","unstructured":"Tarvainen, A., & Valpola, H. (2017). Mean teachers are better role models: Weight-averaged consistency targets improve semi-supervised deep learning results. Advances in neural information processing systems,30."},{"key":"2733_CR52","doi-asserted-by":"crossref","unstructured":"Tian, Z., Shen, C., Chen, H., & He, T. (2019). Fcos: Fully convolutional one-stage object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9627\u20139636.","DOI":"10.1109\/ICCV.2019.00972"},{"key":"2733_CR53","doi-asserted-by":"crossref","unstructured":"Wang, Z., Prabha, R., Huang, T., Wu, J., & Rajagopal, R. (2024). Skyscript: A large and semantically diverse vision-language dataset for remote sensing. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 38, pp. 5805\u20135813.","DOI":"10.1609\/aaai.v38i6.28393"},{"key":"2733_CR54","unstructured":"Wei, G., Yuan, X., Liu, Y., Shang, Z., Yao, K., Li, C., Yan, Q., Zhao, C., Zhang, H., & Xiao, R. (2024). OVA-DETR: Open vocabulary aerial object detection using image-text alignment and fusion. arXiv:2408.12246."},{"key":"2733_CR55","doi-asserted-by":"crossref","unstructured":"Wu, S., Zhang, W., Jin, S., Liu, W.,& Loy, C. C. (2023). Aligning bag of regions for open-vocabulary object detection. In: CVPR.","DOI":"10.1109\/CVPR52729.2023.01464"},{"key":"2733_CR56","doi-asserted-by":"crossref","unstructured":"Wu, X., Zhu, F., Zhao, R., & Li, H. (2023). Cora: Adapting clip for open-vocabulary detection with region prompting and anchor pre-matching. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7031\u20137040.","DOI":"10.1109\/CVPR52729.2023.00679"},{"key":"2733_CR57","doi-asserted-by":"publisher","unstructured":"Xia, G.-S., Bai, X., Ding, J., Zhu, Z., Belongie, S., Luo, J., Datcu, M., Pelillo, M., & Zhang, L. (2018). Dota: A large-scale dataset for object detection in aerial images. In: 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3974\u20133983. https:\/\/doi.org\/10.1109\/CVPR.2018.00418.","DOI":"10.1109\/CVPR.2018.00418"},{"key":"2733_CR58","doi-asserted-by":"crossref","unstructured":"Xie, X., Cheng, G., Wang, J., Yao, X., & Han, J. (2021). Oriented r-cnn for object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 3520\u20133529.","DOI":"10.1109\/ICCV48922.2021.00350"},{"key":"2733_CR59","doi-asserted-by":"crossref","unstructured":"Xu, C., Wang, J., Yang, W., Yu, H., Yu, L., & Xia, G.-S. (2022). Rfla: Gaussian receptive field based label assignment for tiny object detection. In: European Conference on Computer Vision, pp. 526\u2013543. Springer.","DOI":"10.1007\/978-3-031-20077-9_31"},{"key":"2733_CR60","doi-asserted-by":"crossref","unstructured":"Xu, M., Zhang, Z., Hu, H., Wang, J., Wang, L., Wei, F., Bai, X., & Liu, Z. (2021). End-to-end semi-supervised object detection with soft teacher. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3060\u20133069.","DOI":"10.1109\/ICCV48922.2021.00305"},{"key":"2733_CR61","doi-asserted-by":"crossref","unstructured":"Yang, F., Fan, H., Chu, P., Blasch, E., & Ling, H. (2019). Clustered object detection in aerial images. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8311\u20138320.","DOI":"10.1109\/ICCV.2019.00840"},{"key":"2733_CR62","doi-asserted-by":"crossref","unstructured":"Yang, X., Hou, L., Zhou, Y., Wang, W., & Yan, J. (2021). Dense label encoding for boundary discontinuity free rotation detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15819\u201315829.","DOI":"10.1109\/CVPR46437.2021.01556"},{"key":"2733_CR63","unstructured":"Yang, X., Yan, J., Ming, Q., Wang, W., Zhang, X., & Tian, Q. (2021). Rethinking rotated object detection with gaussian wasserstein distance loss. In: International Conference on Machine Learning, pp. 11830\u201311841. PMLR."},{"key":"2733_CR64","doi-asserted-by":"crossref","unstructured":"Yang, X., Yang, J., Yan, J., Zhang, Y., Zhang, T., Guo, Z., Sun, X., & Fu, K. (2019). Scrdet: Towards more robust detection for small, cluttered and rotated objects. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8232\u20138241.","DOI":"10.1109\/ICCV.2019.00832"},{"key":"2733_CR65","doi-asserted-by":"publisher","first-page":"132","DOI":"10.3390\/rs10010132","volume":"10","author":"X Yang","year":"2018","unstructured":"Yang, X., Sun, H., Fu, K., Yang, J., Sun, X., Yan, M., & Guo, Z. (2018). Automatic ship detection in remote sensing images from google earth of complex scenes based on multiscale rotation dense feature pyramid networks. Remote. Sens., 10, 132.","journal-title":"Remote. Sens."},{"issue":"5","key":"2733_CR66","doi-asserted-by":"publisher","first-page":"1340","DOI":"10.1007\/s11263-022-01593-w","volume":"130","author":"X Yang","year":"2022","unstructured":"Yang, X., & Yan, J. (2022). On the arbitrary-oriented object detection: Classification based approaches revisited. International Journal of Computer Vision, 130(5), 1340\u20131365.","journal-title":"International Journal of Computer Vision"},{"key":"2733_CR67","doi-asserted-by":"publisher","first-page":"3163","DOI":"10.1609\/aaai.v35i4.16426","volume":"35","author":"X Yang","year":"2021","unstructured":"Yang, X., Yan, J., Feng, Z., & He, T. (2021). R3det: Refined single-stage detector with feature refinement for rotating object. Proceedings of the AAAI Conference on Artificial Intelligence, 35, 3163\u20133171.","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"2733_CR68","first-page":"18381","volume":"34","author":"X Yang","year":"2021","unstructured":"Yang, X., Yang, X., Yang, J., Ming, Q., Wang, W., Tian, Q., & Yan, J. (2021). Learning high-precision bounding box for rotated object detection via kullback-leibler divergence. Advances in Neural Information Processing Systems, 34, 18381\u201318394.","journal-title":"Advances in Neural Information Processing Systems"},{"issue":"4","key":"2733_CR69","first-page":"4335","volume":"45","author":"X Yang","year":"2022","unstructured":"Yang, X., Zhang, G., Yang, X., Zhou, Y., Wang, W., Tang, J., He, T., & Yan, J. (2022). Detecting rotated objects as gaussian distributions and its 3-d generalization. IEEE Transactions on Pattern Analysis and Machine Intelligence, 45(4), 4335\u20134354.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2733_CR70","first-page":"9125","volume":"35","author":"L Yao","year":"2022","unstructured":"Yao, L., Han, J., Wen, Y., Liang, X., Xu, D., Zhang, W., Li, Z., Xu, C., & Xu, H. (2022). Detclip: Dictionary-enriched visual-concept paralleled pre-training for open-world detection. Advances in Neural Information Processing Systems, 35, 9125\u20139138.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2733_CR71","doi-asserted-by":"crossref","unstructured":"Ye, K., Zhang, M., Kovashka, A., Li, W., Qin, D., & Berent, J. (2019). Cap2det: Learning to amplify weak caption supervision for object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9686\u20139695.","DOI":"10.1109\/ICCV.2019.00978"},{"key":"2733_CR72","doi-asserted-by":"crossref","unstructured":"Zang, Y., Li, W., Zhou, K., Huang, C., & Loy, C. C. (2022). Open-vocabulary detr with conditional matching. In: European Conference on Computer Vision, pp. 106\u2013122. Springer.","DOI":"10.1007\/978-3-031-20077-9_7"},{"key":"2733_CR73","doi-asserted-by":"crossref","unstructured":"Zareian, A., Rosa, K. D., Hu, D. H., & Chang, S.-F. (2021). Open-vocabulary object detection using captions. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14393\u201314402.","DOI":"10.1109\/CVPR46437.2021.01416"},{"key":"2733_CR74","doi-asserted-by":"publisher","first-page":"423","DOI":"10.5194\/isprs-archives-XLI-B7-423-2016","volume":"41","author":"R Zhang","year":"2016","unstructured":"Zhang, R., Yao, J., Zhang, K., Feng, C., & Zhang, J. (2016). S-cnn-based ship detection from high-resolution remote sensing images. The International Archives of the Photogrammetry, Remote Sensing and Spatial Information Sciences, 41, 423\u2013430.","journal-title":"The International Archives of the Photogrammetry, Remote Sensing and Spatial Information Sciences"},{"key":"2733_CR75","first-page":"36067","volume":"35","author":"H Zhang","year":"2022","unstructured":"Zhang, H., Zhang, P., Hu, X., Chen, Y.-C., Li, L., Dai, X., Wang, L., Yuan, L., Hwang, J.-N., & Gao, J. (2022). Glipv2: Unifying localization and vision-language understanding. Advances in Neural Information Processing Systems, 35, 36067\u201336080.","journal-title":"Advances in Neural Information Processing Systems"},{"issue":"4","key":"2733_CR76","doi-asserted-by":"publisher","first-page":"8","DOI":"10.1109\/MGRS.2023.3312347","volume":"11","author":"X Zhang","year":"2023","unstructured":"Zhang, X., Zhang, T., Wang, G., Zhu, P., Tang, X., Jia, X., & Jiao, L. (2023). Remote sensing object detection meets deep learning: A metareview of challenges and advances. IEEE Geoscience and Remote Sensing Magazine, 11(4), 8\u201344. https:\/\/doi.org\/10.1109\/MGRS.2023.3312347","journal-title":"IEEE Geoscience and Remote Sensing Magazine"},{"key":"2733_CR77","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2024.3449154","author":"Z Zhang","year":"2024","unstructured":"Zhang, Z., Zhao, T., Guo, Y., & Yin, J. (2024). Rs5m and georsclip: A large scale vision-language dataset and a large vision-language model for remote sensing. IEEE Transactions on Geoscience and Remote Sensing. https:\/\/doi.org\/10.1109\/TGRS.2024.3449154","journal-title":"IEEE Transactions on Geoscience and Remote Sensing"},{"key":"2733_CR78","doi-asserted-by":"crossref","unstructured":"Zhao, S., Zhang, Z., Schulter, S., Zhao, L., Vijay\u00a0Kumar, B., Stathopoulos, A., Chandraker, M., & Metaxas, D. N. (2022). Exploiting unlabeled data with vision and language models for object detection. In: European Conference on Computer Vision, pp. 159\u2013175. Springer.","DOI":"10.1007\/978-3-031-20077-9_10"},{"issue":"8","key":"2733_CR79","doi-asserted-by":"publisher","first-page":"693","DOI":"10.1016\/S0262-8856(03)00064-7","volume":"21","author":"T Zhao","year":"2003","unstructured":"Zhao, T., & Nevatia, R. (2003). Car detection in low resolution aerial images. Image and Vision Computing, 21(8), 693\u2013703.","journal-title":"Image and Vision Computing"},{"key":"2733_CR80","doi-asserted-by":"crossref","unstructured":"Zheng, Y., Huang, R., Han, C., Huang, X., & Cui, L. (2020). Background learnable cascade for zero-shot object detection. In: Proceedings of the Asian Conference on Computer Vision.","DOI":"10.1007\/978-3-030-69535-4_7"},{"key":"2733_CR81","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.isprsjprs.2020.04.019","volume":"166","author":"Z Zheng","year":"2020","unstructured":"Zheng, Z., Zhong, Y., Ma, A., Han, X., Zhao, J., Liu, Y., & Zhang, L. (2020). Hynet: Hyper-scale object detection network framework for multiple spatial resolution remote sensing imagery. ISPRS Journal of Photogrammetry and Remote Sensing, 166, 1\u201314. https:\/\/doi.org\/10.1016\/j.isprsjprs.2020.04.019","journal-title":"ISPRS Journal of Photogrammetry and Remote Sensing"},{"key":"2733_CR82","doi-asserted-by":"crossref","unstructured":"Zhong, Y., Yang, J., Zhang, P., Li, C., Codella, N., Li, L. H., Zhou, L., Dai, X., Yuan, L., Li, Y., and others. (2022). Regionclip: Region-based language-image pretraining. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16793\u201316803.","DOI":"10.1109\/CVPR52688.2022.01629"},{"key":"2733_CR83","doi-asserted-by":"crossref","unstructured":"Zhou, X., Girdhar, R., Joulin, A., Kr\u00e4henb\u00fchl, P., & Misra, I. (2022). Detecting twenty-thousand classes using image-level supervision. In: European Conference on Computer Vision, pp. 350\u2013368. Springer.","DOI":"10.1007\/978-3-031-20077-9_21"},{"key":"2733_CR84","unstructured":"Zhou, X., Wang, D., & Kr\u00e4henb\u00fchl, P. (2019). Objects as points. arXiv preprint arXiv:1904.07850."},{"key":"2733_CR85","doi-asserted-by":"crossref","unstructured":"Zhou, Y., Yang, X., Zhang, G., Wang, J., Liu, Y., Hou, L., Jiang, X., Liu, X., Yan, J., Lyu, C., Zhang, W., & Chen, K. (2022). Mmrotate: A rotated object detection benchmark using pytorch. In: Proceedings of the 30th ACM International Conference on Multimedia.","DOI":"10.1145\/3503161.3548541"},{"key":"2733_CR86","doi-asserted-by":"crossref","unstructured":"Zhou, Q., Yu, C., Wang, Z., Qian, Q., & Li, H. (2021). Instant-teaching: An end-to-end semi-supervised object detection framework. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4081\u20134090.","DOI":"10.1109\/CVPR46437.2021.00407"},{"issue":"11","key":"2733_CR87","doi-asserted-by":"publisher","first-page":"7380","DOI":"10.1109\/TPAMI.2021.3119563","volume":"44","author":"P Zhu","year":"2021","unstructured":"Zhu, P., Wen, L., Du, D., Bian, X., Fan, H., Hu, Q., & Ling, H. (2021). Detection and tracking meet drones challenge. IEEE Transactions on Pattern Analysis and Machine Intelligence, 44(11), 7380\u20137399.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2733_CR88","doi-asserted-by":"crossref","unstructured":"Zou, Z., & Shi, Z. (2017). Random access memories: A new paradigm for target detection in high resolution aerial remote sensing images. IEEE Transactions on Image Processing, 27(3), 1100\u20131111.","DOI":"10.1109\/TIP.2017.2773199"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-026-02733-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-026-02733-2","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-026-02733-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,18]],"date-time":"2026-04-18T05:43:00Z","timestamp":1776490980000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-026-02733-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,6]]},"references-count":88,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2026,4]]}},"alternative-id":["2733"],"URL":"https:\/\/doi.org\/10.1007\/s11263-026-02733-2","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,3,6]]},"assertion":[{"value":"29 March 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"1 January 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 March 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"156"}}