{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,16]],"date-time":"2025-09-16T20:51:22Z","timestamp":1758055882583,"version":"3.44.0"},"reference-count":73,"publisher":"Springer Science and Business Media LLC","issue":"12","license":[{"start":{"date-parts":[[2025,5,22]],"date-time":"2025-05-22T00:00:00Z","timestamp":1747872000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,5,22]],"date-time":"2025-05-22T00:00:00Z","timestamp":1747872000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"Intelligent Identification System for Benthic Diatom in the Yellow River Basin: Development and Application","award":["2023HHCG013"],"award-info":[{"award-number":["2023HHCG013"]}]},{"name":"Engineering Project for Improving the Innovation Capability of Technology-oriented Small and Medium-sized Enterprises","award":["2023TSGC0293"],"award-info":[{"award-number":["2023TSGC0293"]}]},{"DOI":"10.13039\/100017445","name":"Science Fund for Distinguished Young Scholars of Shandong Province","doi-asserted-by":"publisher","award":["ZR2024MF025"],"award-info":[{"award-number":["ZR2024MF025"]}],"id":[{"id":"10.13039\/100017445","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2025,9]]},"DOI":"10.1007\/s00371-025-03932-7","type":"journal-article","created":{"date-parts":[[2025,5,22]],"date-time":"2025-05-22T03:26:21Z","timestamp":1747884381000},"page":"9373-9394","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Enhancing Fine-Grained Visual Classification via Curriculum Learning and Global\u2013Local Feature Interaction"],"prefix":"10.1007","volume":"41","author":[{"given":"Xueqing","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shuo","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fengjuan","family":"Feng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jianlei","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,5,22]]},"reference":[{"issue":"1","key":"3932_CR1","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1038\/s41467-022-27980-y","volume":"13","author":"D Tuia","year":"2022","unstructured":"Tuia, D., Kellenberger, B., Beery, S., et al.: Perspectives in machine learning for wildlife conservation. Nat. Commun. 13(1), 1\u201315 (2022). https:\/\/doi.org\/10.1038\/s41467-022-27980-y","journal-title":"Nat. Commun."},{"key":"3932_CR2","doi-asserted-by":"publisher","unstructured":"Karlinsky, L., Shtok, J., Tzadok, A.: Fine-grained recognition of thousands of object categories with single-example training. In: Paper presented at the 2017 IEEE Conference on Computer Vision and Pattern Recognition, Honolulu, HI, USA, 21\u201326 July (2017). https:\/\/doi.org\/10.1109\/CVPR.2017.109","DOI":"10.1109\/CVPR.2017.109"},{"key":"3932_CR3","doi-asserted-by":"publisher","unstructured":"Sochor, J., Herout, A., Havel, J.: BoxCars: 3D boxes as CNN input for improved fine-grained vehicle recognition. In: Paper presented at the 2016 IEEE Conference on Computer Vision and Pattern Recognition, Las Vegas, NV, USA, 27\u201330 June (2016). https:\/\/doi.org\/10.1109\/CVPR.2016.328","DOI":"10.1109\/CVPR.2016.328"},{"key":"3932_CR4","doi-asserted-by":"crossref","unstructured":"Yadav, S.S., Jadhav, S.M.: Deep convolutional neural network based medical image classifcation for disease diagnosis. J. Big Data 6(1), 1\u201318 (2019)","DOI":"10.1186\/s40537-019-0276-2"},{"key":"3932_CR5","doi-asserted-by":"publisher","unstructured":"Zhang, H., Xu, T., Elhoseiny, M., et al.: SPDA-CNN: unifying semantic part detection and abstraction for fine-grained recognition. In: Paper presented at the 2016 IEEE Conference on Computer Vision and Pattern Recognition, Las Vegas, NV, USA, 27\u201330 June (2016). https:\/\/doi.org\/10.1109\/CVPR.2016.328","DOI":"10.1109\/CVPR.2016.328"},{"key":"3932_CR6","doi-asserted-by":"publisher","unstructured":"Huang, S., Xu, Z., Tao, D. et al.: Part-stacked CNN for fine-grained visual categorization. In: Paper presented at the 2016 IEEE Conference on Computer Vision and Pattern Recognition, Las Vegas, NV, USA, 27\u201330 June (2016). https:\/\/doi.org\/10.1109\/CVPR.2016.328","DOI":"10.1109\/CVPR.2016.328"},{"key":"3932_CR7","doi-asserted-by":"publisher","first-page":"704","DOI":"10.1016\/j.patcog.2017.10.002","volume":"76","author":"XS Wei","year":"2018","unstructured":"Wei, X.S., Xie, C.W., Wu, J., et al.: Mask-CNN: localizing parts and selecting descriptors for fine-grained bird species categorization. Pattern Recognit. 76, 704\u2013714 (2018). https:\/\/doi.org\/10.1016\/j.patcog.2017.10.002","journal-title":"Pattern Recognit."},{"issue":"8","key":"3932_CR8","doi-asserted-by":"publisher","first-page":"5112","DOI":"10.1109\/TNNLS.2021.3126046","volume":"34","author":"X Guan","year":"2023","unstructured":"Guan, X., Yang, Y., Li, J.J., et al.: On the imaginary wings: text-assisted complex-valued fusion network for fine-grained visual classification. IEEE Trans. Neural Netw. Learn. Syst. 34(8), 5112\u20135121 (2023). https:\/\/doi.org\/10.1109\/TNNLS.2021.3126046","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"3932_CR9","doi-asserted-by":"publisher","unstructured":"Lin, T.Y., RoyChowdhury, A., Maji, S.: Bilinear CNN models for fine-grained visual recognition. In: Paper presented at the 2015 IEEE International Conference on Computer Vision, Santiago, Chile, 07\u201313 December (2015). https:\/\/doi.org\/10.1109\/ICCV.2015.170","DOI":"10.1109\/ICCV.2015.170"},{"key":"3932_CR10","doi-asserted-by":"publisher","unstructured":"Sun, G., Cholakkal, H., Khan, S., et al.: Fine-grained recognition: accounting for subtle differences between similar classes. In: Paper presented at the AAAI Conference on Artificial Intelligence , California, USA (2020). https:\/\/doi.org\/10.1609\/aaai.v34i07.6882","DOI":"10.1609\/aaai.v34i07.6882"},{"key":"3932_CR11","doi-asserted-by":"publisher","unstructured":"Zhu, L.Y., Chen, T.R., Yin, J.X. et al.: Learning Gabor texture features for fine-grained recognition. In: Paper presented at the 2023 IEEE\/CVF International Conference on Computer Vision , Paris, France, 01\u201306 October (2023). https:\/\/doi.org\/10.1109\/ICCV51070.2023.00156","DOI":"10.1109\/ICCV51070.2023.00156"},{"key":"3932_CR12","doi-asserted-by":"publisher","first-page":"9015","DOI":"10.1109\/TMM.2023.3244340","volume":"25","author":"Q Xu","year":"2023","unstructured":"Xu, Q., Wang, J.H., Jiang, B., et al.: Fine-grained visual classification via internal ensemble learning transformer. IEEE Trans. Multimed. 25, 9015\u20139028 (2023). https:\/\/doi.org\/10.1109\/TMM.2023.3244340","journal-title":"IEEE Trans. Multimed."},{"key":"3932_CR13","doi-asserted-by":"publisher","unstructured":"Sun, M., Yuan, Y., Zhou, F., Ding, E.: Multi-attention multi-class constraint for fine-grained image recognition. In: Paper presented at the 2018 European Conference on Computer Vision, Munich, Germany, 08\u201314 September (2018). https:\/\/doi.org\/10.1007\/978-3-030-01270-0_49","DOI":"10.1007\/978-3-030-01270-0_49"},{"key":"3932_CR14","doi-asserted-by":"publisher","unstructured":"Zhang, T., Chang, D.L., Ma, Z.Y. et al.: Progressive co-attention network for fine-grained visual classification. In: Paper presented at the 2021 International Conference on Visual Communications and Image Processing, Munich, Germany, 05\u201308 December (2021). https:\/\/doi.org\/10.1109\/VCIP53242.2021.9675376","DOI":"10.1109\/VCIP53242.2021.9675376"},{"key":"3932_CR15","doi-asserted-by":"publisher","unstructured":"Zhao, Y.F., Yan, K., Huang, F.Y. et al.: Graph-based high-order relation discovery for fine-grained recognition. In: Paper presented at the 2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Nashville, TN, USA, 20\u201325 June (2021). https:\/\/doi.org\/10.1109\/CVPR46437.2021.01483","DOI":"10.1109\/CVPR46437.2021.01483"},{"key":"3932_CR16","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2024.112827","volume":"309","author":"S Chowdhury","year":"2025","unstructured":"Chowdhury, S., Soni, B.: R-vqa: a robust visual question answering model. Knowledge-Based Syst. 309, 112827 (2025). https:\/\/doi.org\/10.1016\/j.knosys.2024.112827","journal-title":"Knowledge-Based Syst."},{"key":"3932_CR17","doi-asserted-by":"publisher","DOI":"10.1016\/j.engappai.2024.109948","volume":"142","author":"S Chowdhury","year":"2025","unstructured":"Chowdhury, S., Soni, B.: Envqa: improving visual question answering model by enriching the visual feature. Eng. Appl. Artif. Intell. 142, 109948 (2025). https:\/\/doi.org\/10.1016\/j.engappai.2024.109948","journal-title":"Eng. Appl. Artif. Intell."},{"issue":"6","key":"3932_CR18","doi-asserted-by":"publisher","first-page":"70010","DOI":"10.1111\/coin.70010","volume":"40","author":"S Chowdhury","year":"2024","unstructured":"Chowdhury, S., Soni, B.: Beyond words: Esc-net revolutionizes vqa by rlevating visual features and defying language priors. Comput. Intell. 40(6), 70010 (2024). https:\/\/doi.org\/10.1111\/coin.70010","journal-title":"Comput. Intell."},{"key":"3932_CR19","doi-asserted-by":"publisher","first-page":"10479","DOI":"10.1007\/s13369-023-07661-8","volume":"48","author":"S Chowdhury","year":"2023","unstructured":"Chowdhury, S., Soni, B.: Qsfvqa: a time efficient, scalable and optimized vqa framework. Arab. J. Sci. Eng. 48, 10479\u201310491 (2023). https:\/\/doi.org\/10.1007\/s13369-023-07661-8","journal-title":"Arab. J. Sci. Eng."},{"key":"3932_CR20","doi-asserted-by":"publisher","unstructured":"Lin, D., Shen, X.Y., Lu, C.W. et al.: Deep LAC: Deep localization, alignment and classification for fine-grained recognition. In: Paper presented at the 2015 IEEE Conference on Computer Vision and Pattern Recognition, Boston, MA, USA, 07\u201312 June (2015). https:\/\/doi.org\/10.1109\/CVPR.2015.7298775","DOI":"10.1109\/CVPR.2015.7298775"},{"key":"3932_CR21","doi-asserted-by":"publisher","unstructured":"Liu, X., Wang, J., Wen, S.L. et al.: Localizing by describing: attribute-guided attention localization for fine-grained recognition. In: Paper presented at the AAAI Conference on Artificial Intelligence , California, USA (2017). https:\/\/doi.org\/10.1609\/aaai.v31i1.11202","DOI":"10.1609\/aaai.v31i1.11202"},{"key":"3932_CR22","doi-asserted-by":"publisher","unstructured":"Gao, Y., Han, X.T., Wang, X. et al.: Channel interaction networks for fine-grained image categorization. In: Paper presented at the AAAI Conference on Artificial Intelligence, California, USA (2020). https:\/\/doi.org\/10.1609\/aaai.v34i07.6712","DOI":"10.1609\/aaai.v34i07.6712"},{"key":"3932_CR23","doi-asserted-by":"publisher","unstructured":"Huang, S.L., Wang, X.C., Tao, D.C.: Snapmix: Semantically proportional mixing for augmenting fine-grained data. In: Paper presented at the AAAI Conference on Artificial Intelligence , California, USA (2021). https:\/\/doi.org\/10.1609\/aaai.v35i2.16255","DOI":"10.1609\/aaai.v35i2.16255"},{"key":"3932_CR24","doi-asserted-by":"publisher","unstructured":"Liang, Y.Z., Zhu, L.C., Wang, X.H., et al.: A simple episodic linear probe improves visual recognition in the wild. In: Paper presented at the 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, New Orleans, LA, USA, 18\u201324 June (2022). https:\/\/doi.org\/10.1109\/CVPR52688.2022.00934","DOI":"10.1109\/CVPR52688.2022.00934"},{"key":"3932_CR25","doi-asserted-by":"publisher","unstructured":"Sun, H.B., He, X.T., Peng, Y.X.: Sim-trans: Structure information modeling transformer for finegrained visual categorization. In: Paper presented at the 30th ACM International Conference on Multimedia, New York, NY, USA, 10\u201314 October (2022). https:\/\/doi.org\/10.1145\/3503161.3548308","DOI":"10.1145\/3503161.3548308"},{"key":"3932_CR26","doi-asserted-by":"publisher","unstructured":"Kim, Y.J., Ha, J.W.: Contrastive fine-grained class clustering via generative adversarial networks. In: Paper presented at the International Conference on Learning Representations, 25\u201329 April (2022). https:\/\/doi.org\/10.48550\/arXiv.2112.14971","DOI":"10.48550\/arXiv.2112.14971"},{"issue":"8","key":"3932_CR27","doi-asserted-by":"publisher","first-page":"9932","DOI":"10.1109\/TPAMI.2023.3237871","volume":"45","author":"WQ Min","year":"2023","unstructured":"Min, W.Q., Wang, Z.L., Liu, Y.X., et al.: Large scale visual food recognition. IEEE Trans. Pattern Anal. Mach. Intell. 45(8), 9932\u20139949 (2023). https:\/\/doi.org\/10.1109\/TPAMI.2023.3237871","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"3932_CR28","doi-asserted-by":"publisher","unstructured":"Han, Y.Z., Han, D.C., Liu, Z.Y., et al.: Dynamic perceiver for efficient visual recognition. In: Paper presented at the 2023 IEEE\/CVF International Conference on Computer Vision, Paris, France, 1\u20136 October (2023). https:\/\/doi.org\/10.1109\/ICCV51070.2023.00551","DOI":"10.1109\/ICCV51070.2023.00551"},{"key":"3932_CR29","doi-asserted-by":"publisher","unstructured":"Zheng, H.L., Fu, J.L., Zha, Z.J. et al.: Looking for the devil in the details: learning trilinear attention sampling network for fine-grained image recognition. In: Paper presented at the 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Long Beach, CA, USA, 15\u201320 June (2019). https:\/\/doi.org\/10.1109\/CVPR.2019.00515","DOI":"10.1109\/CVPR.2019.00515"},{"key":"3932_CR30","doi-asserted-by":"publisher","unstructured":"He, J., Chen, J.N., Liu, S. et al.: TransFG: a transformer architecture for fine-grained recognition. In: Paper presented at the 36th AAAI Conference on Artificial Intelligence, Vancouver, BC, Canada, 28 February\u20131 March (2022). https:\/\/doi.org\/10.1609\/aaai.v36i1.19967","DOI":"10.1609\/aaai.v36i1.19967"},{"key":"3932_CR31","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2023.3238548","author":"H Liu","year":"2023","unstructured":"Liu, H., Zhang, C., Deng, Y.J., et al.: Transifc: invariant cues-aware feature concentration learning for efficient fine-grained bird image classification. IEEE Trans. Multimed. (2023). https:\/\/doi.org\/10.1109\/TMM.2023.3238548","journal-title":"IEEE Trans. Multimed."},{"issue":"9","key":"3932_CR32","doi-asserted-by":"publisher","first-page":"5009","DOI":"10.1109\/TCSVT.2023.3248791","volume":"33","author":"RY Ji","year":"2023","unstructured":"Ji, R.Y., Li, J.Y., Zhang, L.B., et al.: Dual transformer with multi-grained assembly for fne-grained visual classifcation. IEEE Trans. Circuits Syst. Video Technol. 33(9), 5009\u20135021 (2023). https:\/\/doi.org\/10.1109\/TCSVT.2023.3248791","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"3932_CR33","doi-asserted-by":"publisher","first-page":"3113","DOI":"10.1109\/TMM.2023.3307235","volume":"26","author":"CM Wang","year":"2024","unstructured":"Wang, C.M., Fu, H.Y., Ma, H.D.: Learning mutually exclusive part representations for fine-grained image classification. IEEE Trans. Multimed. 26, 3113\u20133124 (2024). https:\/\/doi.org\/10.1109\/TMM.2023.3307235","journal-title":"IEEE Trans. Multimed."},{"key":"3932_CR34","doi-asserted-by":"publisher","first-page":"5312","DOI":"10.1109\/TIP.2024.3459788","volume":"33","author":"HB Sun","year":"2024","unstructured":"Sun, H.B., He, X.T., Xu, J.L., et al.: Sim-ofe: structure information mining and object-aware feature enhancement for fine-grained visual categorization. IEEE Trans. Image Process. 33, 5312\u20135326 (2024). https:\/\/doi.org\/10.1109\/TIP.2024.3459788","journal-title":"IEEE Trans. Image Process."},{"key":"3932_CR35","doi-asserted-by":"publisher","unstructured":"Yang, S.K., Liu, S., Wang, C.H.: Re-rank coarse classification with local region enhanced features for fine-grained image recognition. 1\u201310 (2021) https:\/\/doi.org\/10.48550\/arXiv.2102.09875","DOI":"10.48550\/arXiv.2102.09875"},{"key":"3932_CR36","doi-asserted-by":"publisher","first-page":"9470","DOI":"10.1109\/TIP.2021.3126490","volume":"30","author":"YF Zhao","year":"2021","unstructured":"Zhao, Y.F., Li, J., Chen, X.W., et al.: Part-guided relational transformers for fine-grained visual recognition. IEEE Trans. Image Process. 30, 9470\u20139481 (2021). https:\/\/doi.org\/10.1109\/TIP.2021.3126490","journal-title":"IEEE Trans. Image Process."},{"key":"3932_CR37","doi-asserted-by":"publisher","unstructured":"Yang, X.H., Wang, Y.W., Chen, K., et al.: Fine-grained object classification via self-supervised pose alignment. In: Paper presented at the 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, New Orleans, LA, USA, 18\u201324 June (2022). https:\/\/doi.org\/10.1109\/CVPR52688.2022.00725","DOI":"10.1109\/CVPR52688.2022.00725"},{"issue":"2","key":"3932_CR38","doi-asserted-by":"publisher","first-page":"579","DOI":"10.1109\/TPAMI.2019.2933510","volume":"44","author":"JW Han","year":"2022","unstructured":"Han, J.W., Yao, X.W., Cheng, G., et al.: P-cnn: part-based convolutional neural networks for fine-grained visual categorization. IEEE Trans. Pattern Anal. Mach. Intell. 44(2), 579\u2013590 (2022). https:\/\/doi.org\/10.1109\/TPAMI.2019.2933510","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"3932_CR39","doi-asserted-by":"publisher","first-page":"748","DOI":"10.1109\/TIP.2021.3135477","volume":"31","author":"M Liu","year":"2022","unstructured":"Liu, M., Zhang, C.J., Bai, H.H., et al.: Cross-part learning for fine-grained image classification. IEEE Trans. Image Process. 31, 748\u2013758 (2022). https:\/\/doi.org\/10.1109\/TIP.2021.3135477","journal-title":"IEEE Trans. Image Process."},{"key":"3932_CR40","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2023.109305","volume":"137","author":"X Ke","year":"2023","unstructured":"Ke, X., Cai, Y.H., Chen, B.T., et al.: Granularity-aware distillation and structure modeling region proposal network for fine-grained image classification. Pattern Recognit. 137, 109305 (2023). https:\/\/doi.org\/10.1016\/j.patcog.2023.109305","journal-title":"Pattern Recognit."},{"key":"3932_CR41","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2023.109979","volume":"145","author":"ZC Zhang","year":"2024","unstructured":"Zhang, Z.C., Chen, Z.D., Wang, Y.X., et al.: A vision transformer for fine-grained classification by reducing noise and enhancing discriminative information. Pattern Recognit. 145, 109979 (2024). https:\/\/doi.org\/10.1016\/j.patcog.2023.109979","journal-title":"Pattern Recognit."},{"issue":"1","key":"3932_CR42","doi-asserted-by":"publisher","first-page":"40","DOI":"10.1016\/j.neucom.2023.03.035","volume":"536","author":"L Lu","year":"2023","unstructured":"Lu, L., Cai, Y.C., Huang, H., et al.: An efficient fine-grained vehicle recognition method based on part-level feature optimization. Neurocomputing 536(1), 40\u201349 (2023). https:\/\/doi.org\/10.1016\/j.neucom.2023.03.035","journal-title":"Neurocomputing"},{"key":"3932_CR43","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2023.109550","volume":"140","author":"DC Liu","year":"2023","unstructured":"Liu, D.C., Zhao, L.J., Wang, Y., et al.: Learn from each other to classify better: cross-layer mutual attention learning for fine-grained visual classification. Pattern Recognit. 140, 109550 (2023). https:\/\/doi.org\/10.1016\/j.patcog.2023.109550","journal-title":"Pattern Recognit."},{"key":"3932_CR44","doi-asserted-by":"publisher","first-page":"4529","DOI":"10.1109\/TIP.2024.3441813","volume":"33","author":"JH Wang","year":"2024","unstructured":"Wang, J.H., Xu, Q., Jiang, B., et al.: Multi-granularity part sampling attention for fine-grained visual classification. IEEE Trans. Image Process. 33, 4529\u20134542 (2024). https:\/\/doi.org\/10.1109\/TIP.2024.3441813","journal-title":"IEEE Trans. Image Process."},{"key":"3932_CR45","doi-asserted-by":"publisher","unstructured":"Hu, J., Shen, L., Sun, G.: Squeeze-and-excitation networks. In: Paper presented at the 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Salt Lake, UT, USA, 18\u201323 June (2018). https:\/\/doi.org\/10.1109\/CVPR.2018.00745","DOI":"10.1109\/CVPR.2018.00745"},{"issue":"12","key":"3932_CR46","doi-asserted-by":"publisher","first-page":"3446","DOI":"10.1109\/TMI.2021.3087857","volume":"40","author":"RH Liu","year":"2021","unstructured":"Liu, R.H., Liu, M.Y., Sheng, B., et al.: Nhbs-net: a feature fusion attention network for ultrasound neonatal hip bone segmentation. IEEE Trans. Med. Imaging 40(12), 3446\u20133458 (2021). https:\/\/doi.org\/10.1109\/TMI.2021.3087857","journal-title":"IEEE Trans. Med. Imaging"},{"key":"3932_CR47","doi-asserted-by":"publisher","DOI":"10.1007\/s00371-024-03511-2","author":"SL Wang","year":"2024","unstructured":"Wang, S.L., Hou, Q.W., Li, J.A., et al.: Tsid-net: a two-stage single image dehazing framework with style transfer and contrastive knowledge transfer. Vis. Comput. (2024). https:\/\/doi.org\/10.1007\/s00371-024-03511-2","journal-title":"Vis. Comput."},{"key":"3932_CR48","doi-asserted-by":"publisher","unstructured":"He, K.M., Zhang, X.Y., Ren, S.Q. et al.: Deep residual learning for image recognition. In: Paper presented at the 2016 IEEE Conference on Computer Vision and Pattern Recognition, Las Vegas, NV, USA, 27\u201330 June (2016). https:\/\/doi.org\/10.1109\/CVPR.2016.90","DOI":"10.1109\/CVPR.2016.90"},{"key":"3932_CR49","doi-asserted-by":"publisher","unstructured":"Bengio, Y., Louradour, J., Collobert, R. et al.: Curriculum learning. In: Paper presented at the 26th Annual International Conference on Machine Learning, New York, NY, USA, 14\u201318 June (2009). https:\/\/doi.org\/10.1145\/1553374.1553380","DOI":"10.1145\/1553374.1553380"},{"key":"3932_CR50","unstructured":"Muller, R., Kornblith, S., Hinton, G. et al.: When does label smoothing help? In: Paper presented at the 33rd International Conference on Neural Information Processing Systems, Vancouner, BC, Canada, 08\u201314 December (2019)"},{"key":"3932_CR51","doi-asserted-by":"publisher","unstructured":"Szegedy, C., Vanhoucke, V., Loffe, S. et al.: Rethinking the inception architecture for computer vision. In: Paper presented at the 2016 IEEE Conference on Computer Vision and Pattern Recognition, Las Vegas, NV, USA, 27\u201330 June (2016). https:\/\/doi.org\/10.1109\/CVPR.2016.308","DOI":"10.1109\/CVPR.2016.308"},{"key":"3932_CR52","doi-asserted-by":"publisher","unstructured":"Zoph, B., Vasudevan, V., Shlens, J. et al.: Learning transferable architectures for scalable image recognition. In: Paper presented at the 2018 IEEE Conference on Computer Vision and Pattern Recognition, Salt Lake City, UT, USA, 18\u201323 June (2018). https:\/\/doi.org\/10.1109\/CVPR.2018.00907","DOI":"10.1109\/CVPR.2018.00907"},{"key":"3932_CR53","doi-asserted-by":"publisher","unstructured":"Yang, Z., Luo, T.G., Wang, D. et al.: Learning to navigate for fine-grained classification. In: Paper presented at the 15th European Conference on Computer Vision, Munich, Germany, 8\u201314 September (2018). https:\/\/doi.org\/10.1007\/978-3-030-01264-9_26","DOI":"10.1007\/978-3-030-01264-9_26"},{"key":"3932_CR54","doi-asserted-by":"publisher","unstructured":"Tang, Z.C., Yang, H.L., Yu, C. et al.: Weakly supervised posture mining for fine-grained classification. In: Paper presented at the 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Vancouver, BC, Canada, 17\u201324 June (2023). https:\/\/doi.org\/10.1109\/CVPR52729.2023.02273","DOI":"10.1109\/CVPR52729.2023.02273"},{"issue":"3","key":"3932_CR55","doi-asserted-by":"publisher","first-page":"2237","DOI":"10.1002\/cav.2237","volume":"35","author":"XQ Lin","year":"2024","unstructured":"Lin, X.Q., Zhang, Y., Wang, S., et al.: Multiagent trajectory prediction with global-local scene-enhanced social interaction graph network. Comput. Anim. Virtual Worlds 35(3), 2237 (2024). https:\/\/doi.org\/10.1002\/cav.2237","journal-title":"Comput. Anim. Virtual Worlds"},{"key":"3932_CR56","unstructured":"Wah, C., Branson, S., Welinder, P. et al.: The caltech-ucsd birds-200-2011 dataset. California Institute of Technology CNS-TR-2011-001 (2011)"},{"key":"3932_CR57","doi-asserted-by":"publisher","unstructured":"Krause, J., Stark, M., Deng, J. et al.: 3D object representations for fine-grained categorization. In: Paper presented at the 2013 IEEE International Conference on Computer Vision Workshops, Sydney, NYW, Australia, 02\u201308 December (2013). https:\/\/doi.org\/10.1109\/ICCVW.2013.77","DOI":"10.1109\/ICCVW.2013.77"},{"key":"3932_CR58","doi-asserted-by":"publisher","unstructured":"Maji, S., Rahtu, E., Vedaldi, A.: Fine-grained visual classification of aircraft. ArXiv 1\u20136 (2013) https:\/\/doi.org\/10.48550\/arXiv.1306.5151","DOI":"10.48550\/arXiv.1306.5151"},{"key":"3932_CR59","doi-asserted-by":"publisher","unstructured":"Deng, J., Dong, W., Socher, R. et al.: Imagenet: a large-scale hierarchical image database. In: Paper presented at the 2009 IEEE Conference on Computer Vision and Pattern Recognition, Miami, FL, USA, 20\u201325 June (2009). https:\/\/doi.org\/10.1109\/CVPR.2009.5206848","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"3932_CR60","doi-asserted-by":"publisher","unstructured":"Chen, Y., Bai, Y.L., Zhang, W. et al.: Destruction and construction learning for fine-grained image recognition. In: Paper presented at the 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, Long Beach, CA, USA, 15\u201320 June (2019). https:\/\/doi.org\/10.1109\/CVPR.2019.00530","DOI":"10.1109\/CVPR.2019.00530"},{"key":"3932_CR61","doi-asserted-by":"publisher","unstructured":"Ding, Y., Zhou, Y.Z., Zhu, Y. et al.: Selective sparse sampling for fine-grained image recognition. In: Paper presented at the 2019 IEEE\/CVF International Conference on Computer Vision, Seoul,Korea, 27 October\u201302 November (2019). https:\/\/doi.org\/10.1109\/ICCV.2019.00670","DOI":"10.1109\/ICCV.2019.00670"},{"key":"3932_CR62","doi-asserted-by":"publisher","unstructured":"Zhuang, P.Q., Wang, Y.L., Qiao, Y.: Learning attentive pairwise interaction for fine-grained classification. In: Paper presented at the 34th AAAI Conference on Artificial Intelligence, New York, USA, 07\u201312 February (2020). https:\/\/doi.org\/10.1609\/aaai.v34i07.7016","DOI":"10.1609\/aaai.v34i07.7016"},{"key":"3932_CR63","doi-asserted-by":"publisher","unstructured":"Du, R.Y., Chang, D.L., Bhunia, A.K. et al.: Fine-Grained visual classification via progressive multi-granularity training of jigsaw patches. In: Paper presented at the 16th European Conference on Computer Vision, 23\u201328 August (2020). https:\/\/doi.org\/10.1007\/978-3-030-58565-5_10","DOI":"10.1007\/978-3-030-58565-5_10"},{"key":"3932_CR64","doi-asserted-by":"publisher","unstructured":"Song, J.W., Yang, R.Y.: Feature boosting, suppression, and diversification for fine-grained visual classification. In: Paper presented at the 2021 International Joint Conference on Neural Networks, Shenzhen, China, 18\u201322 June (2021). https:\/\/doi.org\/10.1109\/IJCNN52387.2021.9534004","DOI":"10.1109\/IJCNN52387.2021.9534004"},{"key":"3932_CR65","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2021.108219","volume":"121","author":"LB Zhang","year":"2022","unstructured":"Zhang, L.B., Huang, S.L., Liu, W.: Learning sequentially diversified representations for fine-grained categorization. Pattern Recognit. 121, 108219 (2022). https:\/\/doi.org\/10.1016\/j.patcog.2021.108219","journal-title":"Pattern Recognit."},{"key":"3932_CR66","doi-asserted-by":"crossref","unstructured":"Liu, D.H., Wang, Y., J., K.: Recursive multi-scale channel-spatial attention for fine-grained image lassification. IEICE Transactions on Information and Systems 105-D, 713\u2013726 (2022) https:\/\/api.semanticscholar.org\/CorpusID:247241017","DOI":"10.1587\/transinf.2021EDP7166"},{"key":"3932_CR67","doi-asserted-by":"publisher","first-page":"4409","DOI":"10.1109\/TMM.2021.3117064","volume":"24","author":"LB Zhang","year":"2022","unstructured":"Zhang, L.B., Huang, S.L., Liu, W.: Enhancing mixture-of-experts by lever-aging attention for fine-grained recognition. IEEE Trans. Multimed. 24, 4409\u20134421 (2022). https:\/\/doi.org\/10.1109\/TMM.2021.3117064","journal-title":"IEEE Trans. Multimed."},{"key":"3932_CR68","doi-asserted-by":"publisher","DOI":"10.1016\/j.displa.2023.102468","volume":"79","author":"QX Zhu","year":"2023","unstructured":"Zhu, Q.X., Kuang, W.L., Li, Z.X.: A collaborative gated attention network for fine-grained visual classification. Displays 79, 102468 (2023). https:\/\/doi.org\/10.1016\/j.displa.2023.102468","journal-title":"Displays"},{"issue":"6","key":"3932_CR69","doi-asserted-by":"publisher","first-page":"2798","DOI":"10.1109\/TCSVT.2022.3227737","volume":"33","author":"M Wang","year":"2023","unstructured":"Wang, M., Zhao, P., Lu, X., et al.: Fine-grained visual categorization: a spatial\u2013frequency feature fusion perspective. IEEE Trans. Circuits Syst. Video Technol. 33(6), 2798\u20132812 (2023). https:\/\/doi.org\/10.1109\/TCSVT.2022.3227737","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"3932_CR70","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A. et al.: An image is worth 16x16 words: transformers for image recognition at scale. In: Paper presented at the 2021 International Conference on Learning Representations, Vienna, Austria, 04 May (2021)"},{"key":"3932_CR71","doi-asserted-by":"publisher","unstructured":"He, J.R., Chen, J.N., Liu, S. et al.: TransFG: a transformer architecture for fine-grained recognition. In: Paper presented at the 22th AAI Conference on Artificial Intelligence, 22\u201329 February (2022). https:\/\/doi.org\/10.1609\/aaai.v36i1.19967","DOI":"10.1609\/aaai.v36i1.19967"},{"key":"3932_CR72","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2023.3238548","author":"H Liu","year":"2023","unstructured":"Liu, H., Zhang, C., Deng, Y., et al.: Transifc: invariant cues-aware feature concentration learning for efficient fine-grained bird image classification. IEEE Trans. Multimed. (2023). https:\/\/doi.org\/10.1109\/TMM.2023.3238548","journal-title":"IEEE Trans. Multimed."},{"key":"3932_CR73","doi-asserted-by":"publisher","unstructured":"Selvaraju, R.R., Cogswell, M., Das, A. et al.: Grad-cam: visual explanations from deep networks via gradient-based localization. In: Paper presented at the 2017 IEEE International Conference on Computer Vision, Venice, Italy, 22\u201329 October (2017). https:\/\/doi.org\/10.1109\/ICCV.2017.74","DOI":"10.1109\/ICCV.2017.74"}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-025-03932-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-025-03932-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-025-03932-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,15]],"date-time":"2025-09-15T09:38:20Z","timestamp":1757929100000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-025-03932-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,22]]},"references-count":73,"journal-issue":{"issue":"12","published-print":{"date-parts":[[2025,9]]}},"alternative-id":["3932"],"URL":"https:\/\/doi.org\/10.1007\/s00371-025-03932-7","relation":{},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"type":"print","value":"0178-2789"},{"type":"electronic","value":"1432-2315"}],"subject":[],"published":{"date-parts":[[2025,5,22]]},"assertion":[{"value":"21 April 2025","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 May 2025","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no relevant financial or non-financial interests to disclose.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"YES.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval and consent to participate"}},{"value":"YES.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}},{"value":"YES.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Code availability"}},{"value":"CUB, CAR, AIR are publicly available, and the ALGAE dataset is non-public.","order":6,"name":"Ethics","group":{"name":"EthicsHeading","label":"Materials availability"}}]}}