{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,16]],"date-time":"2026-04-16T20:15:48Z","timestamp":1776370548399,"version":"3.51.2"},"reference-count":107,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2025,3,5]],"date-time":"2025-03-05T00:00:00Z","timestamp":1741132800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,3,5]],"date-time":"2025-03-05T00:00:00Z","timestamp":1741132800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Iran J Comput Sci"],"published-print":{"date-parts":[[2025,6]]},"DOI":"10.1007\/s42044-025-00242-y","type":"journal-article","created":{"date-parts":[[2025,3,5]],"date-time":"2025-03-05T10:07:11Z","timestamp":1741169231000},"page":"271-301","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["From image classification to segmentation: a comprehensive empirical review"],"prefix":"10.1007","volume":"8","author":[{"family":"Ruksana","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jamin Rahman","family":"Jim","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Md Jonair","family":"Hossain","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Aritra","family":"Das","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Md Mohsin","family":"Kabir","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"M. F.","family":"Mridha","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,3,5]]},"reference":[{"key":"242_CR1","doi-asserted-by":"crossref","unstructured":"Yang, J., Shi, R., Ni, B.: Medmnist classification decathlon: a lightweight automl benchmark for medical image analysis. In: IEEE 18th International Symposium on Biomedical Imaging (ISBI), vol. 2021. IEEE, pp. 191\u2013195 (2021)","DOI":"10.1109\/ISBI48211.2021.9434062"},{"issue":"5","key":"242_CR2","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3234150","volume":"51","author":"S Pouyanfar","year":"2018","unstructured":"Pouyanfar, S., Sadiq, S., Yan, Y., Tian, H., Tao, Y., Reyes, M.P., Shyu, M.-L., Chen, S.-C., Iyengar, S.S.: A survey on deep learning: algorithms, techniques, and applications. ACM Comput. Surv. (CSUR) 51(5), 1\u201336 (2018)","journal-title":"ACM Comput. Surv. (CSUR)"},{"key":"242_CR3","unstructured":"Gidaris, S., Singh, P., Komodakis N.: Unsupervised representation learning by predicting image rotations, arXiv preprint arXiv:1803.07728 (2018)"},{"key":"242_CR4","doi-asserted-by":"publisher","first-page":"60","DOI":"10.1016\/j.media.2017.07.005","volume":"42","author":"G Litjens","year":"2017","unstructured":"Litjens, G., Kooi, T., Bejnordi, B.E., Setio, A.A.A., Ciompi, F., Ghafoorian, M., Van Der Laak, J.A., Van Ginneken, B., S\u00e1nchez, C.I.: A survey on deep learning in medical image analysis. Med. Image Anal. 42, 60\u201388 (2017)","journal-title":"Med. Image Anal."},{"key":"242_CR5","unstructured":"Hendrycks, D., Mazeika, M., Dietterich, T.: Deep anomaly detection with outlier exposure, arXiv preprint arXiv:1812.04606 (2018)"},{"key":"242_CR6","doi-asserted-by":"publisher","first-page":"92","DOI":"10.1016\/j.neucom.2022.06.111","volume":"503","author":"SR Dubey","year":"2022","unstructured":"Dubey, S.R., Singh, S.K., Chaudhuri, B.B.: Activation functions in deep learning: a comprehensive survey and benchmark. Neurocomputing 503, 92\u2013108 (2022)","journal-title":"Neurocomputing"},{"key":"242_CR7","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2020.107281","volume":"105","author":"H Qin","year":"2020","unstructured":"Qin, H., Gong, R., Liu, X., Bai, X., Song, J., Sebe, N.: Binary neural networks: a survey. Pattern Recognit. 105, 107281 (2020)","journal-title":"Pattern Recognit."},{"key":"242_CR8","doi-asserted-by":"crossref","unstructured":"He, Y., Liu, P., Wang, Z., Hu, Z., Yang, Y.: Filter pruning via geometric median for deep convolutional neural networks acceleration. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4340\u20134349 (2019)","DOI":"10.1109\/CVPR.2019.00447"},{"key":"242_CR9","doi-asserted-by":"crossref","unstructured":"Shen, M., Liu, X., Gong, R., Han, K.: Balanced binary neural networks with gated residual. In: ICASSP 2020-2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, pp. 4197\u20134201 (2020)","DOI":"10.1109\/ICASSP40776.2020.9054599"},{"key":"242_CR10","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"242_CR11","doi-asserted-by":"crossref","unstructured":"Liu, Z., Mao, H., Wu, C.-Y., Feichtenhofer, C., Darrell, T., Xie, S.: A convnet for the 2020s. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11976\u201311986 (2022)","DOI":"10.1109\/CVPR52688.2022.01167"},{"key":"242_CR12","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S., Guo, B.: Swin transformer: hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 10012\u201310022 (2021)","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"242_CR13","doi-asserted-by":"crossref","unstructured":"Rastegari, M., Ordonez, V., Redmon, J., Farhadi, A.: Xnor-net: Imagenet classification using binary convolutional neural networks. In: European Conference on Computer Vision. Springer, pp. 525\u2013542 (2016)","DOI":"10.1007\/978-3-319-46493-0_32"},{"key":"242_CR14","unstructured":"Ge, Y., Zhang, X., Choi, C.L., Cheung, K.C., Zhao, P., Zhu, F., Wang, X., Zhao, R., Li, H.: Self-distillation with batch knowledge ensembling improves imagenet classification, arXiv preprint arXiv:2104.13298 (2021)"},{"issue":"12","key":"242_CR15","doi-asserted-by":"publisher","first-page":"8766","DOI":"10.1109\/TPAMI.2020.3013679","volume":"44","author":"JR Clough","year":"2020","unstructured":"Clough, J.R., Byrne, N., Oksuz, I., Zimmer, V.A., Schnabel, J.A., King, A.P.: A topological loss function for deep-learning based image segmentation using persistent homology. IEEE Trans. Pattern Anal. Mach. Intell. 44(12), 8766\u20138778 (2020)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"242_CR16","unstructured":"Wang, Q., Zhang, J., Song, S., Zhang, Z.: Attentional neural network: feature selection using cognitive feedback. Adv. Neural Inf. Process. Syst. 27 (2014)"},{"key":"242_CR17","unstructured":"Smith, L., Gal, Y.: Understanding measures of uncertainty for adversarial example detection, arXiv preprint arXiv:1803.08533 (2018)"},{"key":"242_CR18","unstructured":"Giuste, F.O., Vizcarra, J.C.: Cifar-10 image classification using feature ensembles, arXiv preprint arXiv:2002.03846 (2020)"},{"key":"242_CR19","doi-asserted-by":"publisher","first-page":"610","DOI":"10.1109\/TSMC.1973.4309314","volume":"6","author":"RM Haralick","year":"1973","unstructured":"Haralick, R.M., Shanmugam, K., Dinstein, I.H.: Textural features for image classification. IEEE Trans. Syst. Man Cybern. 6, 610\u2013621 (1973)","journal-title":"IEEE Trans. Syst. Man Cybern."},{"key":"242_CR20","doi-asserted-by":"crossref","unstructured":"Cubuk, E.D., Zoph, B., Shlens, J., Le, Q.V.: Randaugment: practical automated data augmentation with a reduced search space. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshops, pp. 702\u2013703 (2020)","DOI":"10.1109\/CVPRW50498.2020.00359"},{"key":"242_CR21","doi-asserted-by":"crossref","unstructured":"Li, Y., Wu, C.-Y., Fan, H., Mangalam, K., Xiong, B., Malik, J., Feichtenhofer, C.: Mvitv2: Improved multiscale vision transformers for classification and detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4804\u20134814 (2022)","DOI":"10.1109\/CVPR52688.2022.00476"},{"issue":"1","key":"242_CR22","doi-asserted-by":"publisher","first-page":"41","DOI":"10.1038\/s41597-022-01721-8","volume":"10","author":"J Yang","year":"2023","unstructured":"Yang, J., Shi, R., Wei, D., Liu, Z., Zhao, L., Ke, B., Pfister, H., Ni, B.: Medmnist v2-a large-scale lightweight benchmark for 2d and 3d biomedical image classification. Sci. Data 10(1), 41 (2023)","journal-title":"Sci. Data"},{"key":"242_CR23","unstructured":"Ying, Z., You, J., Morris, C., Ren, X., Hamilton, W., Leskovec, J.: Hierarchical graph representation learning with differentiable pooling. Adv. Neural Inf. Process. Syst. 31 (2018)"},{"key":"242_CR24","doi-asserted-by":"crossref","unstructured":"Guerrero, P., Kleiman, Y., Ovsjanikov, M., Mitra, N.J.: Pcpnet learning local shape properties from raw point clouds. In: Computer Graphics Forum, vol.\u00a037. Wiley Online Library, pp. 75\u201385 (2018)","DOI":"10.1111\/cgf.13343"},{"key":"242_CR25","first-page":"18583","volume":"33","author":"R Taori","year":"2020","unstructured":"Taori, R., Dave, A., Shankar, V., Carlini, N., Recht, B., Schmidt, L.: Measuring robustness to natural distribution shifts in image classification. Adv. Neural Inf. Process. Syst. 33, 18583\u201318599 (2020)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"242_CR26","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Delving deep into rectifiers: surpassing human-level performance on imagenet classification. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 1026\u20131034 (2015)","DOI":"10.1109\/ICCV.2015.123"},{"key":"242_CR27","first-page":"1513","volume":"33","author":"K Tang","year":"2020","unstructured":"Tang, K., Huang, J., Zhang, H.: Long-tailed classification by keeping the good and removing the bad momentum causal effect. Adv. Neural Inf. Process. Syst. 33, 1513\u20131524 (2020)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"242_CR28","doi-asserted-by":"crossref","unstructured":"Howard, A., Sandler, M., Chu, G., Chen, L.-C., Chen, B., Tan, M., Wang, W., Zhu, Y., Pang, R., Vasudevan, V.: et\u00a0al., Searching for mobilenetv3. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1314\u20131324 (2019)","DOI":"10.1109\/ICCV.2019.00140"},{"key":"242_CR29","doi-asserted-by":"crossref","unstructured":"Kirillov, A., He, K., Girshick, R., Rother, C., Doll\u00e1r, P.: Panoptic segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9404\u20139413 (2019)","DOI":"10.1109\/CVPR.2019.00963"},{"key":"242_CR30","unstructured":"Gerken, J., Carlsson, O., Linander, H., Ohlsson, F., Petersson, C., Persson, D.: Equivariance versus augmentation for spherical images. In: International Conference on Machine Learning, PMLR, pp. 7404\u20137421 (2022)"},{"key":"242_CR31","doi-asserted-by":"publisher","DOI":"10.1016\/j.dsp.2019.102633","volume":"98","author":"SM Peyghambarzadeh","year":"2020","unstructured":"Peyghambarzadeh, S.M., Azizmalayeri, F., Khotanlou, H., Salarpour, A.: Point-planenet: plane kernel based convolutional neural network for point clouds analysis. Digit. Signal Process. 98, 102633 (2020)","journal-title":"Digit. Signal Process."},{"key":"242_CR32","unstructured":"Vacanti, G., Van\u00a0Looveren, A.: Adversarial detection and correction by matching prediction distributions, arXiv preprint arXiv:2002.09364 (2020)"},{"key":"242_CR33","unstructured":"Ren, J., Fort, S., Liu, J., Roy, A.G., Padhy, S., Lakshminarayanan, B.: A simple fix to mahalanobis distance for improving near-ood detection, arXiv preprint arXiv:2106.09022 (2021)"},{"key":"242_CR34","first-page":"4203","volume":"35","author":"J Yang","year":"2022","unstructured":"Yang, J., Li, C., Dai, X., Gao, J.: Focal modulation networks. Adv. Neural Inf. Process. Syst. 35, 4203\u20134217 (2022)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"242_CR35","unstructured":"Zaheer, M.Z., Lee, J.-H., Astrid, M., Lee, S.-I.: Old is gold: Redefining the adversarially learned one-class classifier training paradigm. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14183\u201314193 (2020)"},{"key":"242_CR36","doi-asserted-by":"crossref","unstructured":"Chen, Y., Tian, Y., Pang, G., Carneiro, G.: Deep one-class classification via interpolated gaussian descriptor. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a036, pp. 383\u2013392 (2022)","DOI":"10.1609\/aaai.v36i1.19915"},{"key":"242_CR37","unstructured":"Wong, E., Kolter, Z.: Provable defenses against adversarial examples via the convex outer adversarial polytope. In: International Conference on Machine Learning, PMLR, pp. 5286\u20135295 (2018)"},{"key":"242_CR38","unstructured":"Van\u00a0Amersfoort, J., Smith, L., Teh, Y.W., Gal, Y.: Uncertainty estimation using a single deep deterministic neural network. In: International Conference on Machine Learning, PMLR, pp. 9690\u20139700 (2020)"},{"key":"242_CR39","doi-asserted-by":"crossref","unstructured":"Ridnik, T., Lawen, H., Noy, A., Ben\u00a0Baruch, E., Sharir, G., Friedman, I.: Tresnet: high performance gpu-dedicated architecture. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 1400\u20131409 (2021)","DOI":"10.1109\/WACV48630.2021.00144"},{"key":"242_CR40","doi-asserted-by":"crossref","unstructured":"Zoph, B., Vasudevan, V., Shlens, J., Le, Q.V.: Learning transferable architectures for scalable image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 8697\u20138710 (2018)","DOI":"10.1109\/CVPR.2018.00907"},{"key":"242_CR41","doi-asserted-by":"crossref","unstructured":"Chao, P., Kao, C.-Y., Ruan, Y.-S., Huang, C.-H., Lin, Y.-L.: Hardnet: a low memory traffic network. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3552\u20133561 (2019)","DOI":"10.1109\/ICCV.2019.00365"},{"key":"242_CR42","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S., et\u00a0al.: An image is worth 16x16 words: transformers for image recognition at scale. arxiv 2020, arXiv preprint arXiv:2010.11929 (2010)"},{"key":"242_CR43","doi-asserted-by":"crossref","unstructured":"Yang, C., Zhou, H., An, Z., Jiang, X., Xu, Y., Zhang, Q.: Cross-image relational knowledge distillation for semantic segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12319\u201312328 (2022)","DOI":"10.1109\/CVPR52688.2022.01200"},{"key":"242_CR44","doi-asserted-by":"crossref","unstructured":"Taghanaki, S.A., Abhishek, K., Azizi, S., Hamarneh, G.: A kernelized manifold mapping to diminish the effect of adversarial perturbations. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11340\u201311349 (2019)","DOI":"10.1109\/CVPR.2019.01160"},{"key":"242_CR45","first-page":"15750","volume":"33","author":"L Zhang","year":"2020","unstructured":"Zhang, L., Tanno, R., Xu, M.-C., Jin, C., Jacob, J., Cicarrelli, O., Barkhof, F., Alexander, D.: Disentangling human error from ground truth in segmentation of medical images. Adv. Neural Inf. Process. Syst. 33, 15750\u201315762 (2020)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"242_CR46","unstructured":"Salimans, T., Goodfellow, I., Zaremba, W., Cheung, V., Radford, A., Chen, X.: Improved techniques for training gans. Adv. Neural Inf. Process. Syst. 29 (2016)"},{"key":"242_CR47","doi-asserted-by":"crossref","unstructured":"Liu, C., Zoph, B., Neumann, M., Shlens, J., Hua, W., Li, L.-J., Fei-Fei, L., Yuille, A., Huang, J., Murphy, K. (algorithm)Progressive neural architecture search. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 19\u201334 (2018)","DOI":"10.1007\/978-3-030-01246-5_2"},{"key":"242_CR48","first-page":"5186","volume":"34","author":"T Nguyen","year":"2021","unstructured":"Nguyen, T., Novak, R., Xiao, L., Lee, J.: Dataset distillation with infinitely wide convolutional networks. Adv. Neural Inf. Process. Syst. 34, 5186\u20135198 (2021)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"242_CR49","unstructured":"French, G., Oliver, A., Salimans, T.: Milking cowmask for semi-supervised image classification, arXiv preprint arXiv:2003.12022 (2020)"},{"key":"242_CR50","unstructured":"Alayrac, J.-B., Uesato, J., Huang, P.-S., Fawzi, A., Stanforth, R., Kohli, P.: Are labels required for improving adversarial robustness? Adv. Neural Inf. Process. Syst. 32 (2019)"},{"key":"242_CR51","unstructured":"Elsayed, G., Krishnan, D., Mobahi, H., Regan, K., Bengio, S.: Large margin deep networks for classification. Adv. Neural Inf. Process. Syst. 31 (2018)"},{"key":"242_CR52","doi-asserted-by":"publisher","DOI":"10.1016\/j.cose.2023.103101","volume":"127","author":"E Soremekun","year":"2023","unstructured":"Soremekun, E., Udeshi, S., Chattopadhyay, S.: Towards backdoor attacks and defense in robust machine learning models. Comput. Secur. 127, 103101 (2023)","journal-title":"Comput. Secur."},{"key":"242_CR53","doi-asserted-by":"crossref","unstructured":"Yu, X., Yu, Z., Ramalingam, S.: Learning strict identity mappings in deep residual networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4432\u20134440 (2018)","DOI":"10.1109\/CVPR.2018.00466"},{"key":"242_CR54","doi-asserted-by":"crossref","unstructured":"Huang, G., Liu, Z., Van Der\u00a0Maaten, L., Weinberger, K.Q.: Densely connected convolutional networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4700\u20134708 (2017)","DOI":"10.1109\/CVPR.2017.243"},{"key":"242_CR55","unstructured":"Gupta, S., Konar, D., Aggarwal, V.: A scalable quantum non-local neural network for image classification, arXiv preprint arXiv:2407.18906 (2024)"},{"key":"242_CR56","doi-asserted-by":"crossref","unstructured":"Gong, X., Chang, S., Jiang, Y., Wang, Z.: Autogan: neural architecture search for generative adversarial networks. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3224\u20133234 (2019)","DOI":"10.1109\/ICCV.2019.00332"},{"key":"242_CR57","doi-asserted-by":"crossref","unstructured":"Tang, X., Panda, A., Sehwag, V., Mittal, P.: Differentially private image classification by learning priors from random processes. Adv. Neural Inf. Process. Syst. 36 (2024)","DOI":"10.29012\/jpc.910"},{"key":"242_CR58","doi-asserted-by":"crossref","unstructured":"Bird, J.J., Lotfi, A.: Cifake: Image classification and explainable identification of ai-generated synthetic images. IEEE Access (2024)","DOI":"10.1109\/ACCESS.2024.3356122"},{"key":"242_CR59","doi-asserted-by":"crossref","unstructured":"Zhang, H., Dana, K., Shi, J., Zhang, Z., Wang, X., Tyagi, A., Agrawal, A.: Context encoding for semantic segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7151\u20137160 (2018)","DOI":"10.1109\/CVPR.2018.00747"},{"issue":"2","key":"242_CR60","doi-asserted-by":"publisher","first-page":"86","DOI":"10.58602\/jics.v3i2.41","volume":"3","author":"H Zangana","year":"2025","unstructured":"Zangana, H., Mustafa, F.M., Omar, M.: Advances in adaptive resonance theory for object identification and recognition in image processing. Jurnal Ilmiah Comput. Sci. 3(2), 86\u2013100 (2025)","journal-title":"Jurnal Ilmiah Comput. Sci."},{"key":"242_CR61","doi-asserted-by":"crossref","unstructured":"Jacob, B., Kligys, S., Chen, B., Zhu, M., Tang, M., Howard, A., Adam, H., Kalenichenko, D.: Quantization and training of neural networks for efficient integer-arithmetic-only inference. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2704\u20132713 (2018)","DOI":"10.1109\/CVPR.2018.00286"},{"key":"242_CR62","first-page":"17864","volume":"34","author":"B Cheng","year":"2021","unstructured":"Cheng, B., Schwing, A., Kirillov, A.: Per-pixel classification is not all you need for semantic segmentation. Adv. Neural Inf. Process. Syst. 34, 17864\u201317875 (2021)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"242_CR63","doi-asserted-by":"crossref","unstructured":"Hassani, A., Walton, S., Li, J., Li, S., Shi, H.: Neighborhood attention transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6185\u20136194 (2023)","DOI":"10.1109\/CVPR52729.2023.00599"},{"key":"242_CR64","doi-asserted-by":"crossref","unstructured":"Du, X., Lin, T.-Y., Jin, P., Ghiasi, G., Tan, M., Cui, Y., Le, Q.V., Song, X.: Spinenet: learning scale-permuted backbone for recognition and localization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11592\u201311601 (2020)","DOI":"10.1109\/CVPR42600.2020.01161"},{"key":"242_CR65","doi-asserted-by":"crossref","unstructured":"Howard, A., Sandler, M., Chu, G., Chen, L.-C., Chen, B., Tan, M., Wang, W., Zhu, Y., Pang, R., Vasudevan, V., et\u00a0al.: Searching for mobilenetv3. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1314\u20131324 (2019)","DOI":"10.1109\/ICCV.2019.00140"},{"key":"242_CR66","doi-asserted-by":"crossref","unstructured":"Zhang, H., Wu, C., Zhang, Z., Zhu, Y., Lin, H., Zhang, Z., Sun, Y., He, T., Mueller, J., Manmatha, R., et\u00a0al.: Resnest: split-attention networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2736\u20132746 (2022)","DOI":"10.1109\/CVPRW56347.2022.00309"},{"key":"242_CR67","doi-asserted-by":"crossref","unstructured":"Srinivas, A., Lin, T.-Y., Parmar, N., Shlens, J., Abbeel, P., Vaswani, A.: Bottleneck transformers for visual recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16519\u201316529 (2021)","DOI":"10.1109\/CVPR46437.2021.01625"},{"key":"242_CR68","unstructured":"Reinders, C., Schubert, F., Rosenhahn, B.: Hydramix: multi-image feature mixing for small data image classification (2025)"},{"key":"242_CR69","unstructured":"Yin, N., Shen, L., Wang, M., Lan, L., Ma, Z., Chen, C., Hua, X.-S., Luo, X.: Coco: a coupled contrastive framework for unsupervised domain adaptive graph classification. In: International Conference on Machine Learning, PMLR, pp. 40040\u201340053 (2023)"},{"key":"242_CR70","doi-asserted-by":"crossref","unstructured":"Mao, X., Chen, Y., Zhu, Y., Chen, D., Su, H., Zhang, R., Xue, H.: Coco-o: a benchmark for object detectors under natural distribution shifts. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6339\u20136350 (2023)","DOI":"10.1109\/ICCV51070.2023.00583"},{"key":"242_CR71","unstructured":"Gu, X., Cui, Y., Huang, J., Rashwan, A., Yang, X., Zhou, X., Ghiasi, G., Kuo, W., Chen, H., Chen, L.-C., et\u00a0al.: Dataseg: taming a universal multi-dataset multi-task segmentation model. Adv. Neural Inf. Process. Syst. 36 (2024)"},{"issue":"3","key":"242_CR72","doi-asserted-by":"publisher","first-page":"279","DOI":"10.3390\/electronics10030279","volume":"10","author":"R Padilla","year":"2021","unstructured":"Padilla, R., Passos, W.L., Dias, T.L., Netto, S.L., Da Silva, E.A.: A comparative analysis of object detection metrics with a companion open-source toolkit. Electronics 10(3), 279 (2021)","journal-title":"Electronics"},{"key":"242_CR73","unstructured":"Veit, A., Matera, T., Neumann, L., Matas, J., Belongie, S.: Coco-text: dataset and benchmark for text detection and recognition in natural images, arXiv preprint arXiv:1601.07140 (2016)"},{"key":"242_CR74","doi-asserted-by":"crossref","unstructured":"Yu, W., Luo, M., Zhou, P., Si, C., Zhou, Y., Wang, X., Feng, J., Yan, S.: Metaformer is actually what you need for vision. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10819\u201310829 (2022)","DOI":"10.1109\/CVPR52688.2022.01055"},{"key":"242_CR75","doi-asserted-by":"crossref","unstructured":"Li, J., Hassani, A., Walton, S., Shi, H.: Convmlp: hierarchical convolutional mlps for vision. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6306\u20136315 (2023)","DOI":"10.1109\/CVPRW59228.2023.00671"},{"key":"242_CR76","first-page":"13564","volume":"33","author":"C Chi","year":"2020","unstructured":"Chi, C., Wei, F., Hu, H.: Relationnet++: bridging visual representations for object detection via transformer decoder. Adv. Neural Inf. Process. Syst. 33, 13564\u201313574 (2020)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"242_CR77","doi-asserted-by":"publisher","first-page":"6893","DOI":"10.1109\/TIP.2022.3216771","volume":"31","author":"T Liang","year":"2022","unstructured":"Liang, T., Chu, X., Liu, Y., Wang, Y., Tang, Z., Chu, W., Chen, J., Ling, H.: Cbnet: a composite backbone network architecture for object detection. IEEE Trans. Image Process. 31, 6893\u20136906 (2022)","journal-title":"IEEE Trans. Image Process."},{"key":"242_CR78","unstructured":"Zhou, X., Wang, D., Kr\u00e4henb\u00fchl, P.: Objects as points, arXiv preprint arXiv:1904.07850 (2019)"},{"issue":"3","key":"242_CR79","doi-asserted-by":"publisher","first-page":"415","DOI":"10.1007\/s41095-022-0274-8","volume":"8","author":"W Wang","year":"2022","unstructured":"Wang, W., Xie, E., Li, X., Fan, D.-P., Song, K., Liang, D., Lu, T., Luo, P., Shao, L.: Pvt v2: improved baselines with pyramid vision transformer. Comput. Vis. Media 8(3), 415\u2013424 (2022)","journal-title":"Comput. Vis. Media"},{"key":"242_CR80","doi-asserted-by":"crossref","unstructured":"Singh, S., Yadav, A., Jain, J., Shi, H., Johnson, J., Desai, K.: Benchmarking object detectors with coco: a new path forward. In: European Conference on Computer Vision, Springer, pp. 279\u2013295 (2024)","DOI":"10.1007\/978-3-031-72784-9_16"},{"key":"242_CR81","unstructured":"Pedamonti, D.: Comparison of non-linear activation functions for deep neural networks on mnist classification task, arXiv preprint arXiv:1804.02763 (2018)"},{"key":"242_CR82","unstructured":"Blundell, C., Cornebise, J., Kavukcuoglu, K., Wierstra, D.: Weight uncertainty in neural network. In: International Conference on Machine Learning, PMLR, pp. 1613\u20131622 (2015)"},{"key":"242_CR83","unstructured":"Wang, D., Shelhamer, E., Liu, S., Olshausen, B., Darrell, T.: Tent: fully test-time adaptation by entropy minimization, arXiv preprint arXiv:2006.10726 (2020)"},{"issue":"1","key":"242_CR84","doi-asserted-by":"publisher","first-page":"1003","DOI":"10.1038\/s41598-024-76178-3","volume":"15","author":"W Hussain","year":"2025","unstructured":"Hussain, W., Mushtaq, M.F., Shahroz, M., Akram, U., Ghith, E.S., Tlija, M., Kim, T.-H., Ashraf, I.: Ensemble genetic and cnn model-based image classification by enhancing hyperparameter tuning. Sci. Rep. 15(1), 1003 (2025)","journal-title":"Sci. Rep."},{"issue":"1","key":"242_CR85","doi-asserted-by":"publisher","first-page":"771","DOI":"10.1038\/s41597-024-03587-4","volume":"11","author":"DK Gupta","year":"2024","unstructured":"Gupta, D.K., Bamba, U., Thakur, A., Gupta, A., Agarwal, R., Sharan, S., Demir, E., Agarwal, K., Prasad, D.K.: An ultramnist classification benchmark to train cnns for very large images. Sci. Data 11(1), 771 (2024)","journal-title":"Sci. Data"},{"key":"242_CR86","unstructured":"Shen, K., Jobst, B., Shishenina, E., Pollmann, F.: Classification of the fashion-mnist dataset on a quantum computer, arXiv preprint arXiv:2403.02405 (2024)"},{"issue":"5","key":"242_CR87","first-page":"173","volume":"6","author":"MS Rana","year":"2023","unstructured":"Rana, M.S., Kabir, M.H., Sobur, A.: Comparison of the error rates of mnist datasets using different type of machine learning model. North Am. Acad. Res. 6(5), 173\u2013181 (2023)","journal-title":"North Am. Acad. Res."},{"key":"242_CR88","unstructured":"Cheon, M.: Demonstrating the efficacy of Kolmogorov\u2013Arnold networks in vision tasks, arXiv preprint arXiv:2406.14916 (2024)"},{"key":"242_CR89","doi-asserted-by":"crossref","unstructured":"Sabokrou, M., Khalooei, M., Fathy, M., Adeli, E.: Adversarially learned one-class classifier for novelty detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3379\u20133388 (2018)","DOI":"10.1109\/CVPR.2018.00356"},{"key":"242_CR90","unstructured":"Bbouzidi, S., Hcini, G., Jdey, I., Drira, F.: Convolutional neural networks and vision transformers for fashion mnist classification: a literature review, arXiv preprint arXiv:2406.03478 (2024)"},{"issue":"4","key":"242_CR91","doi-asserted-by":"publisher","first-page":"911","DOI":"10.3390\/electronics12040911","volume":"12","author":"DB Mulindwa","year":"2023","unstructured":"Mulindwa, D.B., Du, S.: An n-sigmoid activation function to improve the squeeze-and-excitation for 2d and 3d deep networks. Electronics 12(4), 911 (2023)","journal-title":"Electronics"},{"key":"242_CR92","doi-asserted-by":"crossref","unstructured":"Sandler, M., Howard, A., Zhu, M., Zhmoginov, A., Chen, L.-C.: Mobilenetv2: inverted residuals and linear bottlenecks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4510\u20134520 (2018)","DOI":"10.1109\/CVPR.2018.00474"},{"key":"242_CR93","unstructured":"Ruff, L., Vandermeulen, R., Goernitz, N., Deecke, L., Siddiqui, S.A., Binder, A., M\u00fcller, E., Kloft, M.: Deep one-class classification. In: International Conference on Machine Learning, PMLR, pp. 4393\u20134402 (2018)"},{"key":"242_CR94","unstructured":"Nalisnick, E., Matsukawa, A., Teh, Y.W., Gorur, D., Lakshminarayanan, B.: Do deep generative models know what they don\u2019t know?, arXiv preprint arXiv:1810.09136 (2018)"},{"key":"242_CR95","doi-asserted-by":"crossref","unstructured":"Liu, M., Zhu, M.: Mobile video object detection with temporally-aware feature maps. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5686\u20135695 (2018)","DOI":"10.1109\/CVPR.2018.00596"},{"issue":"10","key":"242_CR96","doi-asserted-by":"publisher","first-page":"2896","DOI":"10.1109\/TCSVT.2017.2736553","volume":"28","author":"K Kang","year":"2017","unstructured":"Kang, K., Li, H., Yan, J., Zeng, X., Yang, B., Xiao, T., Zhang, C., Wang, Z., Wang, R., Wang, X., et al.: T-cnn: tubelets with convolutional neural networks for object detection from videos. IEEE Trans. Circ. Syst. Video Technol. 28(10), 2896\u20132907 (2017)","journal-title":"IEEE Trans. Circ. Syst. Video Technol."},{"key":"242_CR97","unstructured":"Howard, A.G., Zhu, M., Chen, B., Kalenichenko, D., Wang, W., Weyand, T., Andreetto, M., Adam, H.: Mobilenets: efficient convolutional neural networks for mobile vision applications, arXiv preprint arXiv:1704.04861 (2017)"},{"key":"242_CR98","first-page":"26183","volume":"34","author":"Y Fang","year":"2021","unstructured":"Fang, Y., Liao, B., Wang, X., Fang, J., Qi, J., Wu, R., Niu, J., Liu, W.: You only look at one sequence: rethinking transformer in vision through object detection. Adv. Neural Inf. Process. Syst. 34, 26183\u201326197 (2021)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"242_CR99","doi-asserted-by":"crossref","unstructured":"Vasconcelos, C., Birodkar, V., Dumoulin, V.: Proper reuse of image classification features improves object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13628\u201313637 (2022)","DOI":"10.1109\/CVPR52688.2022.01326"},{"issue":"4","key":"242_CR100","doi-asserted-by":"publisher","first-page":"733","DOI":"10.1007\/s41095-023-0364-2","volume":"9","author":"M-H Guo","year":"2023","unstructured":"Guo, M.-H., Lu, C.-Z., Liu, Z.-N., Cheng, M.-M., Hu, S.-M.: Visual attention network. Comput. Vis. Media 9(4), 733\u2013752 (2023)","journal-title":"Comput. Vis. Media"},{"key":"242_CR101","unstructured":"Bao, H., Dong, L., Piao, S., Wei, F.: Beit: Bert pre-training of image transformers, arXiv preprint arXiv:2106.08254 (2021)"},{"key":"242_CR102","doi-asserted-by":"crossref","unstructured":"Liu, C., Chen, L.-C., Schroff, F., Adam, H., Hua, W., Yuille, A.L., Fei-Fei, L.: Auto-deeplab: hierarchical neural architecture search for semantic image segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 82\u201392 (2019)","DOI":"10.1109\/CVPR.2019.00017"},{"key":"242_CR103","doi-asserted-by":"crossref","unstructured":"Woo, S., Debnath, S., Hu, R., Chen, X., Liu, Z., Kweon, I.S., Xie, S.: Convnext v2: co-designing and scaling convnets with masked autoencoders. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16133\u201316142 (2023)","DOI":"10.1109\/CVPR52729.2023.01548"},{"issue":"5","key":"242_CR104","first-page":"6575","volume":"45","author":"L Yuan","year":"2022","unstructured":"Yuan, L., Hou, Q., Jiang, Z., Feng, J., Yan, S.: Volo: vision outlooker for visual recognition. IEEE Trans. Pattern Anal. Mach. Intell. 45(5), 6575\u20136586 (2022)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"242_CR105","unstructured":"Hatamizadeh, A., Yin, H., Heinrich, G., Kautz, J., Molchanov, P.: Global context vision transformers. In: International Conference on Machine Learning, PMLR, pp. 12633\u201312646 (2023)"},{"key":"242_CR106","doi-asserted-by":"crossref","unstructured":"He, T., Zhang, Z., Zhang, H., Zhang, Z., Xie, J., Li, M.: Bag of tricks for image classification with convolutional neural networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 558\u2013567 (2019)","DOI":"10.1109\/CVPR.2019.00065"},{"key":"242_CR107","unstructured":"Abai, Z., Rajmalwar, N.: Densenet models for tiny imagenet classification, arXiv preprint arXiv:1904.10429 (2019)"}],"container-title":["Iran Journal of Computer Science"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s42044-025-00242-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s42044-025-00242-y\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s42044-025-00242-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,4]],"date-time":"2025-06-04T21:25:39Z","timestamp":1749072339000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s42044-025-00242-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,3,5]]},"references-count":107,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2025,6]]}},"alternative-id":["242"],"URL":"https:\/\/doi.org\/10.1007\/s42044-025-00242-y","relation":{},"ISSN":["2520-8438","2520-8446"],"issn-type":[{"value":"2520-8438","type":"print"},{"value":"2520-8446","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,3,5]]},"assertion":[{"value":"8 November 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"12 February 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 March 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no known competing financial interests or personal relationships that could have influenced this work.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}