{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,16]],"date-time":"2026-01-16T01:06:23Z","timestamp":1768525583757,"version":"3.49.0"},"reference-count":203,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2022,10,19]],"date-time":"2022-10-19T00:00:00Z","timestamp":1666137600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,10,19]],"date-time":"2022-10-19T00:00:00Z","timestamp":1666137600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["61962008"],"award-info":[{"award-number":["61962008"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2023,4]]},"DOI":"10.1007\/s00530-022-01003-8","type":"journal-article","created":{"date-parts":[[2022,10,19]],"date-time":"2022-10-19T09:06:57Z","timestamp":1666170417000},"page":"693-724","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":19,"title":["A survey of visual neural networks: current trends, challenges and opportunities"],"prefix":"10.1007","volume":"29","author":[{"given":"Ping","family":"Feng","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhenjun","family":"Tang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,10,19]]},"reference":[{"issue":"3","key":"1003_CR1","doi-asserted-by":"publisher","first-page":"389","DOI":"10.1007\/s00530-020-00696-z","volume":"27","author":"X Liang","year":"2021","unstructured":"Liang, X., Tang, Z., Xie, X., Wu, J., Zhang, X.: Robust and fast image hashing with two-dimensional PCA. Multimedia Syst. 27(3), 389\u2013401 (2021)","journal-title":"Multimedia Syst."},{"key":"1003_CR2","doi-asserted-by":"crossref","unstructured":"Zoph, B., Vasudevan, V., Shlens, J., Le, Q.V.: Learning transferable architectures for scalable image recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8697\u20138710. (2018)","DOI":"10.1109\/CVPR.2018.00907"},{"key":"1003_CR3","doi-asserted-by":"crossref","unstructured":"Tan, M., Pang, R., Le, Q.V.: Efficientdet: Scalable and efficient object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10781\u201310790. (2020)","DOI":"10.1109\/CVPR42600.2020.01079"},{"key":"1003_CR4","doi-asserted-by":"publisher","DOI":"10.1007\/s00530-022-00915-9","author":"W Pang","year":"2022","unstructured":"Pang, W., He, Q., Li, Y.: Predicting skeleton trajectories using a Skeleton-Transformer for video anomaly detection. Multimed. Syst. (2022). https:\/\/doi.org\/10.1007\/s00530-022-00915-9","journal-title":"Multimed. Syst."},{"key":"1003_CR5","doi-asserted-by":"publisher","first-page":"834","DOI":"10.1109\/TPAMI.2017.2699184","volume":"40","author":"LC Chen","year":"2018","unstructured":"Chen, L.C., Papandreou, G., Kokkinos, I., Murphy, K., Yuille, A.L.: DeepLab: semantic image segmentation with deep convolutional nets, atrous convolution, and fully connected CRFs. IEEE Trans. Pattern Anal. Mach. Intell. 40, 834\u2013848 (2018)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"1003_CR6","doi-asserted-by":"publisher","first-page":"1655","DOI":"10.1109\/TPAMI.2018.2846566","volume":"41","author":"F Radenovi\u0107","year":"2018","unstructured":"Radenovi\u0107, F., Tolias, G., Chum, O.: Fine-tuning CNN image retrieval with no human annotation. IEEE Trans. Pattern Anal. Mach. Intell. 41, 1655\u20131668 (2018)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"1003_CR7","doi-asserted-by":"crossref","unstructured":"Radenovi\u0107, F., Iscen, A., Tolias, G., Avrithis, Y., Chum, O.: Revisiting oxford and paris: Large-scale image retrieval benchmarking. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5706\u20135715. (2018)","DOI":"10.1109\/CVPR.2018.00598"},{"key":"1003_CR8","doi-asserted-by":"crossref","unstructured":"Jarrett, K., Kavukcuoglu, K., Ranzato, M.A., LeCun, Y.: What is the best multi-stage architecture for object recognition? In: Proceedings of the 2009 IEEE 12th International Conference on Computer Vision, pp. 2146\u20132153. (2009)","DOI":"10.1109\/ICCV.2009.5459469"},{"key":"1003_CR9","doi-asserted-by":"crossref","unstructured":"Bengio, Y.: Learning deep architectures for AI. Now Publishers Inc, 2(1), 1\u2013127 (2009)","DOI":"10.1561\/2200000006"},{"key":"1003_CR10","doi-asserted-by":"publisher","DOI":"10.1007\/s00530-022-00916-8","author":"J Lv","year":"2022","unstructured":"Lv, J., Wang, X., Shao, C.: TMIF: transformer-based multi-modal interactive fusion for automatic rumor detection. Multimed. Syst. (2022). https:\/\/doi.org\/10.1007\/s00530-022-00916-8","journal-title":"Multimed. Syst."},{"key":"1003_CR11","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S.: An image is worth 16x16 words: Transformers for image recognition at scale. In: Proceedings of the International Conference on Learning Representations. (2021)"},{"key":"1003_CR12","unstructured":"Chen, M., Radford, A., Child, R., Wu, J., Jun, H., Luan, D., Sutskever, I.: Generative pretraining from pixels. In: Proceedings of the International Conference on Machine Learning, pp. 1691\u20131703 (2020)"},{"key":"1003_CR13","unstructured":"Goodfellow, I.J., Shlens, J., Szegedy, C.: Explaining and harnessing adversarial examples. In: Proceedings of the International Conference on Learning Representations. (2014)"},{"issue":"5","key":"1003_CR14","doi-asserted-by":"publisher","first-page":"726","DOI":"10.1109\/TETCI.2021.3100641","volume":"5","author":"Y Zhang","year":"2021","unstructured":"Zhang, Y., Ti\u0148o, P., Leonardis, A., Tang, K.: A survey on neural network interpretability. IEEE Trans. Emerg. Top. Comput. Intell. 5(5), 726\u2013742 (2021).","journal-title":"IEEE Trans. Emerg. Top. Comput. Intell."},{"key":"1003_CR15","doi-asserted-by":"publisher","first-page":"302","DOI":"10.1016\/j.neucom.2020.07.053","volume":"417","author":"M Alam","year":"2020","unstructured":"Alam, M., Samad, M.D., Vidyaratne, L., Glandon, A., Iftekharuddin, K.M.: Survey on deep neural networks in speech and vision systems. Neurocomputing 417, 302\u2013321 (2020)","journal-title":"Neurocomputing"},{"key":"1003_CR16","unstructured":"Bouvrie, J.: Notes on convolutional neural networks. (2006)"},{"key":"1003_CR17","volume-title":"Deep Learning","author":"I Goodfellow","year":"2016","unstructured":"Goodfellow, I., Bengio, Y., Courville, A.: Deep Learning. MIT press (2016)"},{"key":"1003_CR18","doi-asserted-by":"publisher","first-page":"436","DOI":"10.1038\/nature14539","volume":"521","author":"Y LeCun","year":"2015","unstructured":"LeCun, Y., Bengio, Y., Hinton, G.: Deep learning. Nature 521, 436\u2013444 (2015)","journal-title":"Nature"},{"key":"1003_CR19","doi-asserted-by":"crossref","unstructured":"Scherer, D., M\u00fcller, A., Behnke, S.: Evaluation of pooling operations in convolutional architectures for object recognition. In: Proceedings of the International Conference on Artificial Neural Networks, pp. 92\u2013101. Springer, (2010)","DOI":"10.1007\/978-3-642-15825-4_10"},{"key":"1003_CR20","unstructured":"Wang, T., Wu, D.J., Coates, A., Ng, A.Y.: End-to-end text recognition with convolutional neural networks. In: Proceedings of the 21st International Conference on Pattern Recognition, pp. 3304\u20133308, (2012)"},{"key":"1003_CR21","doi-asserted-by":"publisher","first-page":"1904","DOI":"10.1109\/TPAMI.2015.2389824","volume":"37","author":"K He","year":"2015","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Spatial pyramid pooling in deep convolutional networks for visual recognition. IEEE Trans. Pattern Anal. Mach. Intell. 37, 1904\u20131916 (2015)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"1003_CR22","unstructured":"Lin, M., Chen, Q., Yan, S.: Network in network. In: Proceedings of the International Conference on Learning Representations. (2014)"},{"key":"1003_CR23","doi-asserted-by":"publisher","first-page":"2352","DOI":"10.1162\/neco_a_00990","volume":"29","author":"W Rawat","year":"2017","unstructured":"Rawat, W., Wang, Z.: Deep convolutional neural networks for image classification: a comprehensive review. Neural Comput. 29, 2352\u20132449 (2017)","journal-title":"Neural Comput."},{"key":"1003_CR24","unstructured":"Nwankpa, C., Ijomah, W., Gachagan, A., Marshall, S.: Activation functions: comparison of trends in practice and research for deep learning. arXiv preprint arXiv:1811.03378 (2018)"},{"key":"1003_CR25","doi-asserted-by":"publisher","first-page":"107","DOI":"10.1142\/S0218488598000094","volume":"6","author":"S Hochreiter","year":"1998","unstructured":"Hochreiter, S.: The vanishing gradient problem during learning recurrent neural nets and problem solutions. Int. J. Uncertain. Fuzziness Knowl.-Based Syst. 6, 107\u2013116 (1998)","journal-title":"Int. J. Uncertain. Fuzziness Knowl.-Based Syst."},{"key":"1003_CR26","unstructured":"Ioffe, S., Szegedy, C.: Batch normalization: Accelerating deep network training by reducing internal covariate shift. In: Proceedings of the International Conference on Machine Learning, pp. 448\u2013456. (2015)"},{"key":"1003_CR27","first-page":"2488","volume":"31","author":"S Santurkar","year":"2018","unstructured":"Santurkar, S., Tsipras, D., Ilyas, A., Madry, A.: How does batch normalization help optimization? Adv. Neural Inf. Process. Syst. 31, 2488\u20132498 (2018)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"1003_CR28","doi-asserted-by":"publisher","first-page":"2278","DOI":"10.1109\/5.726791","volume":"86","author":"Y LeCun","year":"1998","unstructured":"LeCun, Y., Bottou, L., Bengio, Y., Haffner, P.: Gradient-based learning applied to document recognition. Proc. IEEE 86, 2278\u20132324 (1998)","journal-title":"Proc. IEEE"},{"key":"1003_CR29","first-page":"84","volume":"25","author":"A Krizhevsky","year":"2012","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: Imagenet classification with deep convolutional neural networks. Adv. Neural Inf. Process. Syst. 25, 84\u201390 (2012)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"1003_CR30","doi-asserted-by":"publisher","first-page":"211","DOI":"10.1007\/s11263-015-0816-y","volume":"115","author":"O Russakovsky","year":"2015","unstructured":"Russakovsky, O., Deng, J., Su, H., Krause, J., Satheesh, S., Ma, S., Huang, Z., Karpathy, A., Khosla, A., Bernstein, M.: Imagenet large scale visual recognition challenge. Int. J. Comput. Vision 115, 211\u2013252 (2015)","journal-title":"Int. J. Comput. Vision"},{"key":"1003_CR31","doi-asserted-by":"crossref","unstructured":"Zeiler, M.D., Fergus, R.: Visualizing and understanding convolutional networks. In: Proceedings of the European Conference on Computer Vision, pp. 818\u2013833. Springer, (2014)","DOI":"10.1007\/978-3-319-10590-1_53"},{"key":"1003_CR32","unstructured":"Simonyan, K., Zisserman, A.: Very deep convolutional networks for large-scale image recognition. In: Proceedings of the International Conference on Learning Representations. (2015)"},{"key":"1003_CR33","doi-asserted-by":"crossref","unstructured":"Szegedy, C., Liu, W., Jia, Y., Sermanet, P., Reed, S., Anguelov, D., Erhan, D., Vanhoucke, V., Rabinovich, A.: Going deeper with convolutions. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1\u20139. (2015)","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"1003_CR34","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778. (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"1003_CR35","doi-asserted-by":"crossref","unstructured":"Szegedy, C., Vanhoucke, V., Ioffe, S., Shlens, J., Wojna, Z.: Rethinking the inception architecture for computer vision. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2818\u20132826. (2016)","DOI":"10.1109\/CVPR.2016.308"},{"key":"1003_CR36","doi-asserted-by":"crossref","unstructured":"Szegedy, C., Ioffe, S., Vanhoucke, V., Alemi, A.A.: Inception-v4, inception-resnet and the impact of residual connections on learning. In: Proceedings of the Thirty-first AAAI Conference on Artificial Intelligence. pp. 4278\u20134284. (2017)","DOI":"10.1609\/aaai.v31i1.11231"},{"key":"1003_CR37","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Identity mappings in deep residual networks. In: Proceedings of the European Conference on Computer Vision, pp. 630\u2013645. Springer, (2016)","DOI":"10.1007\/978-3-319-46493-0_38"},{"key":"1003_CR38","doi-asserted-by":"crossref","unstructured":"Xie, S., Girshick, R., Doll\u00e1r, P., Tu, Z., He, K.: Aggregated residual transformations for deep neural networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1492\u20131500. (2017)","DOI":"10.1109\/CVPR.2017.634"},{"key":"1003_CR39","doi-asserted-by":"crossref","unstructured":"Zagoruyko, S., Komodakis, N.: Wide residual networks. In: British Machine Vision Conference. (2016)","DOI":"10.5244\/C.30.87"},{"key":"1003_CR40","unstructured":"Zhang, H., Wu, C., Zhang, Z., Zhu, Y., Lin, H., Zhang, Z., Sun, Y., He, T., Mueller, J., Manmatha, R.: Resnest: Split-attention networks. arXiv preprint arXiv:2004.08955 (2020)"},{"key":"1003_CR41","doi-asserted-by":"crossref","unstructured":"Huang, G., Liu, Z., Van Der Maaten, L., Weinberger, K.Q.: Densely connected convolutional networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4700\u20134708. (2017)","DOI":"10.1109\/CVPR.2017.243"},{"key":"1003_CR42","doi-asserted-by":"crossref","unstructured":"Hu, J., Shen, L., Sun, G.: Squeeze-and-excitation networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7132\u20137141. (2018)","DOI":"10.1109\/CVPR.2018.00745"},{"key":"1003_CR43","doi-asserted-by":"crossref","unstructured":"Chollet, F.: Xception: Deep learning with depthwise separable convolutions. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1251\u20131258. (2017)","DOI":"10.1109\/CVPR.2017.195"},{"key":"1003_CR44","unstructured":"Howard, A.G., Zhu, M., Chen, B., Kalenichenko, D., Wang, W., Weyand, T., Andreetto, M., Adam, H.: Mobilenets: Efficient convolutional neural networks for mobile vision applications. arXiv preprint arXiv:1704.04861 (2017)"},{"key":"1003_CR45","doi-asserted-by":"crossref","unstructured":"Su, J., Faraone, J., Liu, J., Zhao, Y., Thomas, D.B., Leong, P.H., Cheung, P.Y.: Redundancy-reduced mobilenet acceleration on reconfigurable logic for imagenet classification. In: International Symposium on Applied Reconfigurable Computing, pp. 16\u201328. Springer, (2018)","DOI":"10.1007\/978-3-319-78890-6_2"},{"key":"1003_CR46","unstructured":"Han, S., Mao, H., Dally, W.J.: Deep compression: Compressing deep neural networks with pruning, trained quantization and huffman coding. In: Proceedings of the International Conference on Learning Representations. (2016)"},{"key":"1003_CR47","unstructured":"Zhou, S., Wu, Y., Ni, Z., Zhou, X., Wen, H., Zou, Y.: Dorefa-net: Training low bitwidth convolutional neural networks with low bitwidth gradients. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. (2018)"},{"key":"1003_CR48","unstructured":"Tan, M., Le, Q.: Efficientnet: Rethinking model scaling for convolutional neural networks. In: Proceedings of the International Conference on Machine Learning, pp. 6105\u20136114. (2019)"},{"key":"1003_CR49","doi-asserted-by":"crossref","unstructured":"Pan, X., Luo, P., Shi, J., Tang, X.: Two at once: Enhancing learning and generalization capacities via ibn-net. In: Proceedings of the European Conference on Computer Vision, pp. 464\u2013479. (2018)","DOI":"10.1007\/978-3-030-01225-0_29"},{"key":"1003_CR50","first-page":"3965","volume":"34","author":"Z Dai","year":"2021","unstructured":"Dai, Z., Liu, H., Le, Q., Tan, M.: Coatnet: marrying convolution and attention for all data sizes. Adv. Neural. Inf. Process. Syst. 34, 3965\u20133977 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1003_CR51","unstructured":"Han, K., Wang, Y., Guo, J., Tang, Y., Wu, E.: Vision GNN: An Image is Worth Graph of Nodes. arXiv preprint arXiv:2206.00272 (2022)"},{"key":"1003_CR52","first-page":"5998","volume":"30","author":"A Vaswani","year":"2017","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, \u0141, Polosukhin, I.: Attention is all you need. Adv. Neural. Inf. Process. Syst. 30, 5998\u20136008 (2017)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1003_CR53","unstructured":"Ba, J.L., Kiros, J.R., Hinton, G.E.: Layer normalization. arXiv preprint arXiv:1607.06450 (2016)"},{"key":"1003_CR54","unstructured":"Hendrycks, D., Gimpel, K.: Gaussian error linear units (gelus). arXiv preprint arXiv:1606.08415 (2016)"},{"key":"1003_CR55","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., Zagoruyko, S.: End-to-end object detection with transformers. In: Proceedings of the European Conference on Computer Vision, pp. 213\u2013229. Springer, (2020)","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"1003_CR56","first-page":"91","volume":"28","author":"S Ren","year":"2015","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster r-cnn: towards real-time object detection with region proposal networks. Adv. Neural. Inf. Process. Syst. 28, 91\u201399 (2015)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1003_CR57","doi-asserted-by":"crossref","unstructured":"Girshick, R.: Fast r-cnn. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 1440\u20131448. (2015)","DOI":"10.1109\/ICCV.2015.169"},{"key":"1003_CR58","unstructured":"Chen, T., Kornblith, S., Norouzi, M., Hinton, G.: A simple framework for contrastive learning of visual representations. In: Proceedings of the International Conference on Machine Learning, pp. 1597\u20131607. (2020)"},{"key":"1003_CR59","first-page":"9","volume":"1","author":"A Radford","year":"2019","unstructured":"Radford, A., Wu, J., Child, R., Luan, D., Amodei, D., Sutskever, I.: Language models are unsupervised multitask learners. OpenAI blog 1, 9 (2019)","journal-title":"OpenAI blog"},{"key":"1003_CR60","doi-asserted-by":"crossref","unstructured":"Radosavovic, I., Kosaraju, R.P., Girshick, R., He, K., Doll\u00e1r, P.: Designing network design spaces. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10428\u201310436. (2020)","DOI":"10.1109\/CVPR42600.2020.01044"},{"key":"1003_CR61","unstructured":"Touvron, H., Cord, M., Douze, M., Massa, F., Sablayrolles, A., J\u00e9gou, H.: Training data-efficient image transformers & distillation through attention. In: Proceedings of the International Conference on Machine Learning, pp. 10347\u201310357. (2021)"},{"key":"1003_CR62","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S., Guo, B.: Swin transformer: Hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 10012\u201310022. (2021)","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"1003_CR63","unstructured":"Bello, I.: Lambdanetworks: Modeling long-range interactions without attention. arXiv preprint arXiv:2102.08602 (2021)"},{"key":"1003_CR64","unstructured":"Huang, Z., Ben, Y., Luo, G., Cheng, P., Yu, G., Fu, B.: Shuffle transformer: Rethinking spatial shuffle for vision transformer. arXiv preprint arXiv:2106.03650 (2021)"},{"key":"1003_CR65","doi-asserted-by":"crossref","unstructured":"Wang, W., Xie, E., Li, X., Fan, D.-P., Song, K., Liang, D., Lu, T., Luo, P., Shao, L.: Pyramid vision transformer: A versatile backbone for dense prediction without convolutions. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 568\u2013578. (2021)","DOI":"10.1109\/ICCV48922.2021.00061"},{"key":"1003_CR66","doi-asserted-by":"crossref","unstructured":"Wang, W., Xie, E., Li, X., Fan, D.-P., Song, K., Liang, D., Lu, T., Luo, P., Shao, L.: PVT v2: Improved baselines with Pyramid Vision Transformer. Computational Visual Media 1\u201310 (2022)","DOI":"10.1007\/s41095-022-0274-8"},{"key":"1003_CR67","unstructured":"Zhou, Daquan, et al.: Deepvit: Towards deeper vision transformer. arXiv preprint arXiv:2103.11886 (2021)."},{"key":"1003_CR68","unstructured":"Li, Y., Zhang, K., Cao, J., Timofte, R., Van Gool, L.: Localvit: Bringing locality to vision transformers. arXiv preprint arXiv:2104.05707 (2021)"},{"key":"1003_CR69","doi-asserted-by":"crossref","unstructured":"Zhang, J., Peng, H., Wu, K., Liu, M., Xiao, B., Fu, J., Yuan, L.: MiniViT: Compressing Vision Transformers with Weight Multiplexing. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12145\u201312154. (2022)","DOI":"10.1109\/CVPR52688.2022.01183"},{"key":"1003_CR70","first-page":"26831","volume":"34","author":"Y Bai","year":"2021","unstructured":"Bai, Y., Mei, J., Yuille, A.L., Xie, C.: Are Transformers more robust than CNNs? Adv. Neural. Inf. Process. Syst. 34, 26831\u201326843 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1003_CR71","unstructured":"Kipf, T.N., Welling, M.: Semi-supervised classification with graph convolutional networks. arXiv preprint arXiv:1609.02907 (2016)"},{"key":"1003_CR72","unstructured":"Hendrycks, D., Dietterich, T.: Benchmarking neural network robustness to common corruptions and perturbations. In: Proceedings of the International Conference on Learning Representations, (2019)"},{"key":"1003_CR73","doi-asserted-by":"crossref","unstructured":"Hendrycks, D., Zhao, K., Basart, S., Steinhardt, J., Song, D.: Natural adversarial examples. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15262\u201315271. (2021)","DOI":"10.1109\/CVPR46437.2021.01501"},{"key":"1003_CR74","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Xie, Y., Xing, F., McGough, M., Yang, L.: Mdnet: A semantically and visually interpretable medical image diagnosis network. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6428\u20136436. (2017)","DOI":"10.1109\/CVPR.2017.378"},{"key":"1003_CR75","doi-asserted-by":"publisher","first-page":"42","DOI":"10.1109\/MSP.2021.3134634","volume":"39","author":"L Ericsson","year":"2022","unstructured":"Ericsson, L., Gouk, H., Loy, C.C., Hospedales, T.M.: Self-supervised representation learning: introduction, advances, and challenges. IEEE Signal Process. Mag. 39, 42\u201362 (2022)","journal-title":"IEEE Signal Process. Mag."},{"key":"1003_CR76","unstructured":"Zhang, Y., Kang, B., Hooi, B., Yan, S., Feng, J.: Deep long-tailed learning: A survey. arXiv preprint arXiv:2110.04596 (2021)"},{"key":"1003_CR77","doi-asserted-by":"crossref","unstructured":"Liu, W., Wang, H., Shen, X., Tsang, I.: The emerging trends of multi-label learning. IEEE Transactions on Pattern Analysis and Machine Intelligence (2021)","DOI":"10.1109\/TPAMI.2021.3119334"},{"key":"1003_CR78","doi-asserted-by":"publisher","first-page":"370","DOI":"10.1016\/j.neucom.2021.07.045","volume":"461","author":"T Liang","year":"2021","unstructured":"Liang, T., Glossner, J., Wang, L., Shi, S., Zhang, X.: Pruning and quantization for deep neural network acceleration: a survey. Neurocomputing 461, 370\u2013403 (2021)","journal-title":"Neurocomputing"},{"key":"1003_CR79","first-page":"129","volume":"2","author":"D Blalock","year":"2020","unstructured":"Blalock, D., Gonzalez Ortiz, J.J., Frankle, J., Guttag, J.: What is the state of neural network pruning? Proc. Mach. Learn. Syst. 2, 129\u2013146 (2020)","journal-title":"Proc. Mach. Learn. Syst."},{"key":"1003_CR80","unstructured":"Liu, Z., Sun, M., Zhou, T., Huang, G., Darrell, T.: Rethinking the value of network pruning. In: Proceedings of the International Conference on Learning Representations. (2019)"},{"key":"1003_CR81","unstructured":"Krishnamoorthi, R.: Quantizing deep convolutional networks for efficient inference: A whitepaper. arXiv preprint arXiv:1806.08342 (2018)"},{"key":"1003_CR82","first-page":"598","volume":"2","author":"Y LeCun","year":"1989","unstructured":"LeCun, Y., Denker, J., Solla, S.: Optimal brain damage. Adv. Neural. Inf. Process. Syst. 2, 598\u2013605 (1989)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1003_CR83","unstructured":"Hassibi, B., Stork, D.G., Wolff, G.J.: Optimal brain surgeon and general network pruning. In: Proceedings of the IEEE International Conference on Neural Networks, pp. 293\u2013299 (1993)"},{"key":"1003_CR84","unstructured":"Molchanov, D., Ashukha, A., Vetrov, D.: Variational dropout sparsifies deep neural networks. In: Proceedings of the International Conference on Machine Learning, pp. 2498\u20132507 (2017)"},{"key":"1003_CR85","doi-asserted-by":"crossref","unstructured":"Luo, J.-H., Wu, J., Lin, W.: Thinet: A filter level pruning method for deep neural network compression. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 5058\u20135066 (2017)","DOI":"10.1109\/ICCV.2017.541"},{"key":"1003_CR86","unstructured":"Louizos, C., Welling, M., Kingma, D.P.: Learning sparse neural networks through L0 regularization. arXiv preprint arXiv:1712.01312 (2017)"},{"key":"1003_CR87","doi-asserted-by":"crossref","unstructured":"He, Y., Zhang, X., Sun, J.: Channel pruning for accelerating very deep neural networks. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 1389\u20131397 (2017)","DOI":"10.1109\/ICCV.2017.155"},{"key":"1003_CR88","doi-asserted-by":"crossref","unstructured":"Yu, R., Li, A., Chen, C.-F., Lai, J.-H., Morariu, V.I., Han, X., Gao, M., Lin, C.-Y., Davis, L.S.: Nisp: Pruning networks using neuron importance score propagation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9194\u20139203 (2018)","DOI":"10.1109\/CVPR.2018.00958"},{"key":"1003_CR89","doi-asserted-by":"crossref","unstructured":"Mittal, D., Bhardwaj, S., Khapra, M.M., Ravindran, B.: Recovering from random pruning: On the plasticity of deep convolutional neural networks. In: Proceedings of the 2018 IEEE Winter Conference on Applications of Computer Vision (WACV), pp. 848\u2013857 (2018)","DOI":"10.1109\/WACV.2018.00098"},{"key":"1003_CR90","unstructured":"Anwar, S., Sung, W.: Compact deep convolutional neural networks with coarse pruning. arXiv preprint arXiv:1610.09639 (2016)"},{"key":"1003_CR91","first-page":"14014","volume":"32","author":"P Michel","year":"2019","unstructured":"Michel, P., Levy, O., Neubig, G.: Are sixteen heads really better than one? Adv. Neural. Inf. Process. Syst. 32, 14014\u201314024\u00a0 (2019)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1003_CR92","first-page":"2181","volume":"30","author":"J Lin","year":"2017","unstructured":"Lin, J., Rao, Y., Lu, J., Zhou, J.: Runtime neural pruning. Adv. Neural Inf. Process. Syst. 30, 2181\u20132191 (2017)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"1003_CR93","unstructured":"Yu, J., Yang, L., Xu, N., Yang, J., Huang, T.: Slimmable neural networks. arXiv preprint arXiv:1812.08928 (2018)"},{"key":"1003_CR94","doi-asserted-by":"crossref","unstructured":"Ding, X., Zhang, X., Ma, N., Han, J., Ding, G., Sun, J.: Repvgg: Making vgg-style convnets great again. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13733\u201313742 (2021)","DOI":"10.1109\/CVPR46437.2021.01352"},{"key":"1003_CR95","doi-asserted-by":"crossref","unstructured":"Guo, Z., Zhang, X., Mu, H., Heng, W., Liu, Z., Wei, Y., Sun, J.: Single path one-shot neural architecture search with uniform sampling. In: Proceedings of the European Conference on Computer Vision, pp. 544\u2013560. Springer (2020)","DOI":"10.1007\/978-3-030-58517-4_32"},{"key":"1003_CR96","doi-asserted-by":"crossref","unstructured":"Yu, H., Han, Q., Li, J., Shi, J., Cheng, G., Fan, B.: Search what you want: Barrier panelty nas for mixed precision quantization. In: Proceedings of the European Conference on Computer Vision, pp. 1\u201316. Springer (2020)","DOI":"10.1007\/978-3-030-58545-7_1"},{"key":"1003_CR97","doi-asserted-by":"crossref","unstructured":"Zhang, B., Chen, H., Yang, L., Chen, C., Zhu, Y., Doermann, D.: Cp-nas: Child-parent neural architecture search for 1-bit cnns. In: Proceedings of the Twenty-Ninth International Joint Conference on Artificial Intelligence, pp. 1033\u20131039 (2021)","DOI":"10.24963\/ijcai.2020\/144"},{"key":"1003_CR98","doi-asserted-by":"crossref","unstructured":"Chen, M., Peng, H., Fu, J., & Ling, H.: Autoformer: Searching transformers for visual recognition. In Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 12270\u201312280 (2021)","DOI":"10.1109\/ICCV48922.2021.01205"},{"key":"1003_CR99","first-page":"8714","volume":"34","author":"M Chen","year":"2021","unstructured":"Chen, M., Wu, K., Ni, B., et al.: Searching the search space of vision transformer. Adv. Neural. Inf. Process. Syst. 34, 8714\u20138726 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1003_CR100","doi-asserted-by":"crossref","unstructured":"Tang, Y., Han, K., Wang, Y., Xu, C., Guo, J., Xu, C., Tao, D.: Patch slimming for efficient vision transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12165\u201312174 (2022)","DOI":"10.1109\/CVPR52688.2022.01185"},{"key":"1003_CR101","first-page":"1742","volume":"30","author":"U K\u00f6ster","year":"2017","unstructured":"K\u00f6ster, U., Webb, T., Wang, X., Nassar, M., Bansal, A.K., Constable, W., Elibol, O., Gray, S., Hall, S., Hornof, L.: Flexpoint: an adaptive numerical format for efficient training of deep neural networks. Adv. Neural. Inf. Process. Syst. 30, 1742\u20131752 (2017)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1003_CR102","unstructured":"Micikevicius, P., Narang, S., Alben, J., Diamos, G., Elsen, E., Garcia, D., Ginsburg, B., Houston, M., Kuchaiev, O., Venkatesh, G.: Mixed precision training. In: Proceedings of the International Conference on Learning Representations (2017)"},{"key":"1003_CR103","unstructured":"Das, D., Mellempudi, N., Mudigere, D., Kalamkar, D., Avancha, S., Banerjee, K., Sridharan, S., Vaidyanathan, K., Kaul, B., Georganas, E.: Mixed precision training of convolutional neural networks using integer operations. arXiv preprint arXiv:1802.00930 (2018)"},{"key":"1003_CR104","first-page":"5151","volume":"31","author":"R Banner","year":"2018","unstructured":"Banner, R., Hubara, I., Hoffer, E., Soudry, D.: Scalable methods for 8-bit training of neural networks. Adv. Neural. Inf. Process. Syst. 31, 5151\u20135159 (2018)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1003_CR105","unstructured":"Wu, S., Li, G., Chen, F., Shi, L.: Training and inference with integers in deep neural networks. In: Proceedings of the International Conference on Learning Representations (2018)"},{"key":"1003_CR106","doi-asserted-by":"publisher","first-page":"70","DOI":"10.1016\/j.neunet.2019.12.027","volume":"125","author":"Y Yang","year":"2020","unstructured":"Yang, Y., Deng, L., Wu, S., Yan, T., Xie, Y., Li, G.: Training high-performance and large-scale deep neural networks with full 8-bit integers. Neural Netw. 125, 70\u201382 (2020)","journal-title":"Neural Netw."},{"key":"1003_CR107","doi-asserted-by":"crossref","unstructured":"Zhu, F., Gong, R., Yu, F., Liu, X., Wang, Y., Li, Z., Yang, X., Yan, J.: Towards unified int8 training for convolutional neural network. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1969\u20131979 (2020)","DOI":"10.1109\/CVPR42600.2020.00204"},{"key":"1003_CR108","first-page":"28092","volume":"34","author":"Z Liu","year":"2021","unstructured":"Liu, Z., Wang, Y., Han, K., Zhang, W., Ma, S., Gao, W. : Post-training quantization for vision transformer. Adv. Neural. Inf. Process. Syst. 34, 28092\u201328103 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1003_CR109","unstructured":"Li, Z., Yang, T., Wang, P., Cheng, J.: Q-ViT: Fully Differentiable Quantization for Vision Transformer. arXiv preprint arXiv:2201.07703 (2022)"},{"key":"1003_CR110","unstructured":"Hinton, G., Vinyals, O., Dean, J.: Distilling the knowledge in a neural network. arXiv preprint arXiv:1503.02531 2, (2015)"},{"key":"1003_CR111","doi-asserted-by":"crossref","unstructured":"Bucilu\u01ce, C., Caruana, R., Niculescu-Mizil, A.: Model compression. In: Proceedings of the 12th ACM SIGKDD International Conference on Knowledge Discovery and Data Mining, pp. 535\u2013541 (2006)","DOI":"10.1145\/1150402.1150464"},{"key":"1003_CR112","unstructured":"Liu, S., Yin, L., Mocanu, D.C., Pechenizkiy, M.: Do we actually need dense over-parameterization? in-time over-parameterization in sparse training. In: Proceedings of the International Conference on Machine Learning, pp. 6989\u20137000 (2021)"},{"key":"1003_CR113","unstructured":"Jia, D., Han, K., Wang, Y., Tang, Y., Guo, J., Zhang, C., Tao, D.: Efficient vision transformers via fine-grained manifold distillation. arXiv preprint arXiv:2107.01378 (2021)"},{"key":"1003_CR114","unstructured":"Romero, A., Ballas, N., Kahou, S.E., Chassang, A., Gatta, C., Bengio, Y.: Fitnets: Hints for thin deep nets. In: Proceedings of the International Conference on Learning Representations. (2015)"},{"key":"1003_CR115","unstructured":"Zagoruyko, S., Komodakis, N.: Paying more attention to attention: Improving the performance of convolutional neural networks via attention transfer. In: Proceedings of the International Conference on Learning Representations (2017)"},{"key":"1003_CR116","unstructured":"Koratana, A., Kang, D., Bailis, P., Zaharia, M.: Lit: Learned intermediate representation training for model compression. In: Proceedings of the International Conference on Machine Learning, pp. 3509\u20133518. (2019)"},{"key":"1003_CR117","doi-asserted-by":"crossref","unstructured":"Liu, Z., Li, J., Shen, Z., Huang, G., Yan, S., Zhang, C.: Learning efficient convolutional networks through network slimming. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2736\u20132744 (2017)","DOI":"10.1109\/ICCV.2017.298"},{"key":"1003_CR118","doi-asserted-by":"crossref","unstructured":"Tung, F., Mori, G.: Similarity-preserving knowledge distillation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1365\u20131374 (2019)","DOI":"10.1109\/ICCV.2019.00145"},{"key":"1003_CR119","doi-asserted-by":"crossref","unstructured":"Tian, F., Gao, B., Cui, Q., Chen, E., Liu, T.-Y.: Learning deep representations for graph clustering. In: Proceedings of the AAAI Conference on Artificial Intelligence, pp.1293\u20131299 (2014)","DOI":"10.1609\/aaai.v28i1.8916"},{"key":"1003_CR120","unstructured":"Huang, Z., Wang, N.: Like what you like: Knowledge distill via neuron selectivity transfer. arXiv preprint arXiv:1707.01219 (2017)"},{"key":"1003_CR121","unstructured":"Wu, X., Wu, Y., Zhao, Y.: Binarized neural networks on the imagenet classification task. arXiv preprint arXiv:1604.03058 (2016)"},{"key":"1003_CR122","doi-asserted-by":"crossref","unstructured":"Park, W., Kim, D., Lu, Y., Cho, M.: Relational knowledge distillation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3967\u20133976 (2019)","DOI":"10.1109\/CVPR.2019.00409"},{"key":"1003_CR123","doi-asserted-by":"crossref","unstructured":"Guo, J., Han, K., Wu, H., Tang, Y., Chen, X., Wang, Y., Xu, C.: Cmt: Convolutional neural networks meet vision transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12175\u201312185 (2022)","DOI":"10.1109\/CVPR52688.2022.01186"},{"key":"1003_CR124","doi-asserted-by":"crossref","unstructured":"Sainath, T.N., Kingsbury, B., Sindhwani, V., Arisoy, E., Ramabhadran, B.: Low-rank matrix factorization for deep neural network training with high-dimensional output targets. In: Proceedings of the 2013 IEEE International Conference on Acoustics, Speech and Signal Processing, pp. 6655\u20136659 (2013)","DOI":"10.1109\/ICASSP.2013.6638949"},{"key":"1003_CR125","unstructured":"Lebedev, V., Ganin, Y., Rakhuba, M., Oseledets, I., Lempitsky, V.: Speeding-up convolutional neural networks using fine-tuned cp-decomposition. In: Proceedings of the International Conference on Learning Representations (2015)"},{"key":"1003_CR126","first-page":"442","volume":"22","author":"A Novikov","year":"2015","unstructured":"Novikov, A., Podoprikhin, D., Osokin, A., Vetrov, D.P.: Tensorizing neural networks. Adv. Neural. Inf. Process. Syst. 22, 442\u2013450 (2015)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1003_CR127","doi-asserted-by":"crossref","unstructured":"Jaderberg, M., Vedaldi, A., Zisserman, A.: Speeding up convolutional neural networks with low rank expansions. In: Proceedings of the British Machine Vision Conference (2014)","DOI":"10.5244\/C.28.88"},{"key":"1003_CR128","unstructured":"Chen, W., Wilson, J., Tyree, S., Weinberger, K., Chen, Y.: Compressing neural networks with the hashing trick. In: Proceedings of the International Conference on Machine Learning, pp. 2285\u20132294 (2015)"},{"key":"1003_CR129","doi-asserted-by":"publisher","first-page":"2889","DOI":"10.1109\/TPAMI.2018.2873305","volume":"41","author":"S Lin","year":"2018","unstructured":"Lin, S., Ji, R., Chen, C., Tao, D., Luo, J.: Holistic cnn compression via low-rank decomposition with knowledge transfer. IEEE Trans. Pattern Anal. Mach. Intell. 41, 2889\u20132905 (2018)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"1003_CR130","unstructured":"Child, R., Gray, S., Radford, A., Sutskever, I.: Generating long sequences with sparse transformers. arXiv preprint arXiv:1904.10509 (2019)"},{"key":"1003_CR131","doi-asserted-by":"crossref","unstructured":"Sandler, M., Howard, A., Zhu, M., Zhmoginov, A., Chen, L.-C.: Mobilenetv2: Inverted residuals and linear bottlenecks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4510\u20134520 (2018)","DOI":"10.1109\/CVPR.2018.00474"},{"key":"1003_CR132","doi-asserted-by":"crossref","unstructured":"Howard, A., Sandler, M., Chu, G., Chen, L.-C., Chen, B., Tan, M., Wang, W., Zhu, Y., Pang, R., Vasudevan, V.: Searching for mobilenetv3. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1314\u20131324 (2019)","DOI":"10.1109\/ICCV.2019.00140"},{"key":"1003_CR133","doi-asserted-by":"crossref","unstructured":"Ma, N., Zhang, X., Zheng, H.-T., Sun, J.: Shufflenet v2: Practical guidelines for efficient cnn architecture design. In: Proceedings of the European Conference on Computer Vision, pp. 116\u2013131 (2018)","DOI":"10.1007\/978-3-030-01264-9_8"},{"key":"1003_CR134","unstructured":"Sun, K., Li, M., Liu, D., Wang, J.: Igcv3: Interleaved low-rank group convolutions for efficient deep neural networks. In: Proceedings of the British Machine Vision Conference, 101, (2018)"},{"key":"1003_CR135","doi-asserted-by":"crossref","unstructured":"Zhang, X., Zhou, X., Lin, M., Sun, J.: Shufflenet: An extremely efficient convolutional neural network for mobile devices. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6848\u20136856 (2018)","DOI":"10.1109\/CVPR.2018.00716"},{"key":"1003_CR136","doi-asserted-by":"crossref","unstructured":"Zhang, T., Qi, G.-J., Xiao, B., Wang, J.: Interleaved group convolutions. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 4373\u20134382 (2017)","DOI":"10.1109\/ICCV.2017.469"},{"key":"1003_CR137","doi-asserted-by":"crossref","unstructured":"Huang, G., Liu, S., Van der Maaten, L., Weinberger, K.Q.: Condensenet: An efficient densenet using learned group convolutions. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2752\u20132761 (2018)","DOI":"10.1109\/CVPR.2018.00291"},{"key":"1003_CR138","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Li, J., Shao, W., Peng, Z., Zhang, R., Wang, X., Luo, P.: Differentiable learning-to-group channels via groupable convolutional neural networks. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3542\u20133551 (2019)","DOI":"10.1109\/ICCV.2019.00364"},{"key":"1003_CR139","first-page":"19974","volume":"34","author":"T Chen","year":"2021","unstructured":"Chen, T., Cheng, Y., Gan, Z., Yuan, L., Zhang, L., Wang, Z.: Chasing sparsity in vision transformers: an end-to-end exploration. Adv. Neural. Inf. Process. Syst. 34, 19974\u201319988 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1003_CR140","unstructured":"Evci, U., Gale, T., Menick, J., Castro, P.S., Elsen, E.: Rigging the lottery: Making all tickets winners. In: Proceedings of the International Conference on Machine Learning, pp. 2943\u20132952 (2020)"},{"key":"1003_CR141","doi-asserted-by":"publisher","first-page":"11357","DOI":"10.1109\/JIOT.2021.3052105","volume":"8","author":"L Wu","year":"2021","unstructured":"Wu, L., Lin, X., Chen, Z., Huang, J., Liu, H., Yang, Y.: An efficient binary convolutional neural network with numerous skip connections for fog computing. IEEE Internet Things J. 8, 11357\u201311367 (2021)","journal-title":"IEEE Internet Things J."},{"key":"1003_CR142","doi-asserted-by":"crossref","unstructured":"Singh, P., Namboodiri, V.P.: SkipConv: skip convolution for computationally efficient deep CNNs. In: Proceedings of the 2020 International Joint Conference on Neural Networks (IJCNN), pp. 1\u20138 (2020)","DOI":"10.1109\/IJCNN48605.2020.9207705"},{"key":"1003_CR143","unstructured":"Raghu, M., Unterthiner, T., Kornblith, S., Zhang, C., Dosovitskiy, A.: Do vision transformers see like convolutional neural networks? Adv. Neural Inf. Process. Syst. 34, 12116\u201312128 (2021)"},{"key":"1003_CR144","unstructured":"Dong, Y., Cordonnier, J.-B., Loukas, A.: Attention is not all you need: Pure attention loses rank doubly exponentially with depth. In: Proceedings of the International Conference on Machine Learning, pp. 2793\u20132803 (2021)"},{"key":"1003_CR145","doi-asserted-by":"crossref","unstructured":"Hosseini, H., Xiao, B., Poovendran, R.: Google's cloud vision api is not robust to noise. In: Proceedings of the 16th IEEE International Conference on Machine Learning and Applications (ICMLA), pp. 101\u2013105 (2017)","DOI":"10.1109\/ICMLA.2017.0-172"},{"key":"1003_CR146","unstructured":"Recht, B., Roelofs, R., Schmidt, L., Shankar, V.: Do CIFAR-10 classifiers generalize to CIFAR-10? arXiv preprint arXiv:1806.00451 (2018)"},{"key":"1003_CR147","unstructured":"Morrison, K., Gilby, B., Lipchak, C., Mattioli, A., Kovashka, A.: Exploring Corruption Robustness: Inductive Biases in Vision Transformers and MLP-Mixers. arXiv preprint arXiv:2106.13122 (2021)"},{"key":"1003_CR148","doi-asserted-by":"crossref","unstructured":"Cubuk, E.D., Zoph, B., Mane, D., Vasudevan, V., Le, Q.V.: Autoaugment: Learning augmentation strategies from data. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 113\u2013123 (2019)","DOI":"10.1109\/CVPR.2019.00020"},{"key":"1003_CR149","unstructured":"Hendrycks, D., Mu, N., Cubuk, E.D., Zoph, B., Gilmer, J., Lakshminarayanan, B.: Augmix: A simple data processing method to improve robustness and uncertainty. In: Proceedings of the International Conference on Learning Representations (2020)"},{"key":"1003_CR150","doi-asserted-by":"crossref","unstructured":"Hendrycks, D., Basart, S., Mu, N., Kadavath, S., Wang, F., Dorundo, E., Desai, R., Zhu, T., Parajuli, S., Guo, M.: The many faces of robustness: A critical analysis of out-of-distribution generalization. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8340\u20138349 (2021)","DOI":"10.1109\/ICCV48922.2021.00823"},{"key":"1003_CR151","unstructured":"DeVries, T., Taylor, G.W.: Improved regularization of convolutional neural networks with cutout. arXiv preprint arXiv:1708.04552 (2017)"},{"key":"1003_CR152","doi-asserted-by":"crossref","unstructured":"Yun, S., Han, D., Oh, S.J., Chun, S., Choe, J., Yoo, Y.: Cutmix: Regularization strategy to train strong classifiers with localizable features. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6023\u20136032 (2019)","DOI":"10.1109\/ICCV.2019.00612"},{"key":"1003_CR153","unstructured":"Zhang, H., Cisse, M., Dauphin, Y.N., Lopez-Paz, D.: mixup: Beyond empirical risk minimization. In: Proceedings of the International Conference on Learning Representations (2018)"},{"key":"1003_CR154","doi-asserted-by":"crossref","unstructured":"Cubuk, E.D., Zoph, B., Mane, D., Vasudevan, V., Le, Q.V.: Autoaugment: learning augmentation policies from data. arXiv preprint arXiv:1805.09501 (2018)","DOI":"10.1109\/CVPR.2019.00020"},{"key":"1003_CR155","unstructured":"Lopes, R.G., Yin, D., Poole, B., Gilmer, J., Cubuk, E.D.: Improving robustness without sacrificing accuracy with patch gaussian augmentation. arXiv preprint arXiv:1906.02611 (2019)"},{"key":"1003_CR156","unstructured":"Verma, V., Lamb, A., Beckham, C., Najafi, A., Mitliagkas, I., Lopez-Paz, D., Bengio, Y.: Manifold mixup: Better representations by interpolating hidden states. In: Proceedings of the International Conference on Machine Learning, pp. 6438\u20136447 (2019)"},{"key":"1003_CR157","unstructured":"Raghunathan, A., Xie, S.M., Yang, F., Duchi, J.C., Liang, P.: Adversarial training can hurt generalization. arXiv preprint arXiv:1906.06032 (2019)"},{"key":"1003_CR158","doi-asserted-by":"crossref","unstructured":"Modas, A., Rade, R., Ortiz-Jim\u00e9nez, G., Moosavi-Dezfooli, S.-M., Frossard, P.: PRIME: a few primitives can boost robustness to common corruptions. arXiv preprint arXiv:2112.13547 (2021)","DOI":"10.1007\/978-3-031-19806-9_36"},{"key":"1003_CR159","doi-asserted-by":"publisher","first-page":"1849","DOI":"10.1109\/LSP.2015.2438008","volume":"22","author":"J Chen","year":"2015","unstructured":"Chen, J., Kang, X., Liu, Y., Wang, Z.J.: Median filtering forensics based on convolutional neural networks. IEEE Signal Process. Lett. 22, 1849\u20131853 (2015)","journal-title":"IEEE Signal Process. Lett."},{"key":"1003_CR160","doi-asserted-by":"crossref","unstructured":"Bayar, B., Stamm, M.C.: A deep learning approach to universal image manipulation detection using a new convolutional layer. In: Proceedings of the 4th ACM Workshop on Information Hiding and Multimedia Security, pp. 5\u201310 (2016)","DOI":"10.1145\/2909827.2930786"},{"key":"1003_CR161","doi-asserted-by":"crossref","unstructured":"Rao, Y., Ni, J.: A deep learning approach to detection of splicing and copy-move forgeries in images. In: Proceedings of the 2016 IEEE International Workshop on Information Forensics and Security (WIFS), pp. 1\u20136 (2016)","DOI":"10.1109\/WIFS.2016.7823911"},{"key":"1003_CR162","doi-asserted-by":"crossref","unstructured":"Bappy, J.H., Roy-Chowdhury, A.K., Bunk, J., Nataraj, L., Manjunath, B.: Exploiting spatial structure for localizing manipulated image regions. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 4970\u20134979 (2017)","DOI":"10.1109\/ICCV.2017.532"},{"key":"1003_CR163","doi-asserted-by":"crossref","unstructured":"Wu, Y., AbdAlmageed, W., Natarajan, P.: Mantra-net: Manipulation tracing network for detection and localization of image forgeries with anomalous features. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9543\u20139552 (2019)","DOI":"10.1109\/CVPR.2019.00977"},{"key":"1003_CR164","doi-asserted-by":"crossref","unstructured":"Zhou, P., Han, X., Morariu, V.I., Davis, L.S.: Learning rich features for image manipulation detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1053\u20131061 (2018)","DOI":"10.1109\/CVPR.2018.00116"},{"key":"1003_CR165","doi-asserted-by":"crossref","unstructured":"Bi, X., Zhang, Z., Xiao, B.: Reality transform adversarial generators for image splicing forgery detection and localization. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 14294\u201314303 (2021)","DOI":"10.1109\/ICCV48922.2021.01403"},{"key":"1003_CR166","doi-asserted-by":"crossref","unstructured":"Hao, J., Zhang, Z., Yang, S., Xie, D., Pu, S.: Transforensics: image forgery localization with dense self-attention. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 15055\u201315064 (2021)","DOI":"10.1109\/ICCV48922.2021.01478"},{"key":"1003_CR167","doi-asserted-by":"crossref","unstructured":"Nguyen, E., Bui, T., Swaminathan, V., Collomosse, J.: OSCAR-Net: object-centric scene graph attention for image attribution. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 14499\u201314508 (2021)","DOI":"10.1109\/ICCV48922.2021.01423"},{"key":"1003_CR168","doi-asserted-by":"crossref","unstructured":"Bhojanapalli, S., Chakrabarti, A., Glasner, D., Li, D., Unterthiner, T., Veit, A.: Understanding robustness of transformers for image classification. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 10231\u201310241 (2021)","DOI":"10.1109\/ICCV48922.2021.01007"},{"key":"1003_CR169","unstructured":"Szegedy, C., Zaremba, W., Sutskever, I., Bruna, J., Erhan, D., Goodfellow, I., Fergus, R.: Intriguing properties of neural networks. In: Proceedings of the International Conference on Learning Representations (2013)"},{"key":"1003_CR170","unstructured":"Shaham, U., Yamada, Y., Negahban, S.: Understanding adversarial training: Increasing local stability of neural nets through robust optimization. arXiv preprint arXiv:1511.05432 (2015)"},{"key":"1003_CR171","doi-asserted-by":"crossref","unstructured":"Moosavi-Dezfooli, S.-M., Fawzi, A., Frossard, P.: Deepfool: a simple and accurate method to fool deep neural networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2574\u20132582 (2016)","DOI":"10.1109\/CVPR.2016.282"},{"key":"1003_CR172","unstructured":"Tram\u00e8r, F., Kurakin, A., Papernot, N., Goodfellow, I., Boneh, D., McDaniel, P.: Ensemble adversarial training: attacks and defenses. arXiv preprint arXiv:1705.07204 (2017)"},{"key":"1003_CR173","unstructured":"Wong, E., Rice, L., Kolter, J.Z.: Fast is better than free: revisiting adversarial training. In: Proceedings of the International Conference on Learning Representations (2020)"},{"key":"1003_CR174","unstructured":"Madry, A., Makelov, A., Schmidt, L., Tsipras, D., Vladu, A.: Towards deep learning models resistant to adversarial attacks. In: Proceedings of the International Conference on Learning Representations (2018)"},{"key":"1003_CR175","first-page":"227","volume":"32","author":"D Zhang","year":"2019","unstructured":"Zhang, D., Zhang, T., Lu, Y., Zhu, Z., Dong, B.: You only propagate once: accelerating adversarial training via maximal principle. Adv. Neural. Inf. Process. Syst. 32, 227\u2013238 (2019)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1003_CR176","doi-asserted-by":"crossref","unstructured":"Shell, K.: Applications of Pontryagin\u2019s maximum principle to economics. Mathematical Systems Theory and Economics I\/II, pp. 241\u2013292. Springer (1969)","DOI":"10.1007\/978-3-642-46196-5_12"},{"key":"1003_CR177","unstructured":"Zhang, H., Yu, Y., Jiao, J., Xing, E., El Ghaoui, L., Jordan, M.: Theoretically principled trade-off between robustness and accuracy. In: Proceedings of the International Conference on Machine Learning, pp. 7472\u20137482 (2019)"},{"key":"1003_CR178","unstructured":"Cisse, M., Bojanowski, P., Grave, E., Dauphin, Y., Usunier, N.: Parseval networks: improving robustness to adversarial examples. In: Proceedings of the International Conference on Machine Learning, pp. 854\u2013863 (2017)"},{"key":"1003_CR179","first-page":"417","volume":"31","author":"Z Yan","year":"2018","unstructured":"Yan, Z., Guo, Y., Zhang, C.: Deep defense: training dnns with improved adversarial robustness. Adv. Neural. Inf. Process. Syst. 31, 417\u2013426 (2018)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1003_CR180","first-page":"8322","volume":"31","author":"Y Song","year":"2018","unstructured":"Song, Y., Shu, R., Kushman, N., Ermon, S.: Constructing unrestricted adversarial examples with generative models. Adv. Neural. Inf. Process. Syst. 31, 8322\u20138333 (2018)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1003_CR181","unstructured":"Samangouei, P., Kabkab, M., Chellappa, R.: Defense-gan: Protecting classifiers against adversarial attacks using generative models. In: Proceedings of the International Conference on Learning Representations, pp. 1\u201317 (2018)"},{"key":"1003_CR182","unstructured":"Mu, N., Wagner, D.: Defending against Adversarial Patches with Robust Self-Attention. In: ICML 2021 Workshop on Uncertainty and Robustness in Deep Learning (2021)"},{"key":"1003_CR183","unstructured":"Wickramanayake, S., Hsu, W., Lee, M.L.: Towards fully interpretable deep neural networks: are we there yet? arXiv preprint arXiv:2106.13164 (2021)"},{"key":"1003_CR184","doi-asserted-by":"crossref","unstructured":"Zhou, B., Khosla, A., Lapedriza, A., Oliva, A., Torralba, A.: Learning deep features for discriminative localization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2921\u20132929 (2016)","DOI":"10.1109\/CVPR.2016.319"},{"key":"1003_CR185","doi-asserted-by":"crossref","unstructured":"Selvaraju, R.R., Cogswell, M., Das, A., Vedantam, R., Parikh, D., Batra, D.: Grad-cam: visual explanations from deep networks via gradient-based localization. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 618\u2013626 (2017)","DOI":"10.1109\/ICCV.2017.74"},{"key":"1003_CR186","unstructured":"Kim, B., Wattenberg, M., Gilmer, J., Cai, C., Wexler, J., Viegas, F.: Interpretability beyond feature attribution: quantitative testing with concept activation vectors (tcav). In: Proceedings of the International Conference on Machine Learning, pp. 2668\u20132677 (2018)"},{"key":"1003_CR187","doi-asserted-by":"crossref","unstructured":"Zheng, H., Fu, J., Mei, T., Luo, J.: Learning multi-attention convolutional neural network for fine-grained image recognition. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 5209\u20135217. (2017)","DOI":"10.1109\/ICCV.2017.557"},{"key":"1003_CR188","doi-asserted-by":"crossref","unstructured":"Huang, Z., Li, Y.: Interpretable and accurate fine-grained recognition via region grouping. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8662\u20138672 (2020)","DOI":"10.1109\/CVPR42600.2020.00869"},{"issue":"3","key":"1003_CR189","first-page":"2431","volume":"25","author":"V Pillai","year":"2021","unstructured":"Pillai, V., Pirsiavash, H.: Explainable models with consistent interpretations. UMBC Student Collection 25(3), 2431\u20132439 (2021)","journal-title":"UMBC Student Collection"},{"key":"1003_CR190","doi-asserted-by":"crossref","unstructured":"Chefer, H., Gur, S., Wolf, L.: Transformer interpretability beyond attention visualization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 782\u2013791 (2021)","DOI":"10.1109\/CVPR46437.2021.00084"},{"key":"1003_CR191","doi-asserted-by":"publisher","first-page":"1084","DOI":"10.1007\/s11263-017-1059-x","volume":"126","author":"J Zhang","year":"2018","unstructured":"Zhang, J., Bargal, S.A., Lin, Z., Brandt, J., Shen, X., Sclaroff, S.: Top-down neural attention by excitation backprop. Int. J. Comput. Vis. 126, 1084\u20131102 (2018)","journal-title":"Int. J. Comput. Vis."},{"key":"1003_CR192","doi-asserted-by":"crossref","unstructured":"Williford, J.R., May, B.B., Byrne, J.: Explainable face recognition. In: Proceedings of the European Conference on Computer Vision, pp. 248\u2013263. Springer (2020)","DOI":"10.1007\/978-3-030-58621-8_15"},{"key":"1003_CR193","unstructured":"Petsiuk, V., Das, A., Saenko, K.: Rise: Randomized input sampling for explanation of black-box models. arXiv preprint arXiv:1806.07421 (2018)"},{"key":"1003_CR194","unstructured":"Smilkov, D., Thorat, N., Kim, B., Vi\u00e9gas, F., Wattenberg, M.: Smoothgrad: removing noise by adding noise. arXiv preprint arXiv:1706.03825 (2017)"},{"key":"1003_CR195","unstructured":"Springenberg, J.T., Dosovitskiy, A., Brox, T., Riedmiller, M.: Striving for simplicity: the all convolutional net. In: Proceedings of the workshop on the International Conference on Learning Representations (2015)"},{"key":"1003_CR196","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pone.0130140","volume":"10","author":"S Bach","year":"2015","unstructured":"Bach, S., Binder, A., Montavon, G., Klauschen, F., M\u00fcller, K.-R., Samek, W.: On pixel-wise explanations for non-linear classifier decisions by layer-wise relevance propagation. PLoS\u00a0One 10, e0130140 (2015)","journal-title":"PLoS\u00a0One"},{"key":"1003_CR197","unstructured":"Shrikumar, A., Greenside, P., Kundaje, A.: Learning important features through propagating activation differences. In: Proceedings of the International Conference on Machine Learning, pp. 3145\u20133153. (2017)"},{"key":"1003_CR198","unstructured":"Sundararajan, M., Taly, A., Yan, Q.: Axiomatic attribution for deep networks. In: Proceedings of the International Conference on Machine Learning, pp. 3319\u20133328 (2017)"},{"key":"1003_CR199","doi-asserted-by":"crossref","unstructured":"Li, O., Liu, H., Chen, C., Rudin, C.: Deep learning for case-based reasoning through prototypes: A neural network that explains its predictions. In: Proceedings of the AAAI Conference on Artificial Intelligence, pp. 3530\u20133537 (2018)","DOI":"10.1609\/aaai.v32i1.11771"},{"key":"1003_CR200","doi-asserted-by":"crossref","unstructured":"Ancona, M., Ceolini, E., \u00d6ztireli, C., Gross, M.: Towards better understanding of gradient-based attribution methods for deep neural networks. In: Proceedings of the International Conference on Learning Representations (2018)","DOI":"10.1007\/978-3-030-28954-6_9"},{"key":"1003_CR201","doi-asserted-by":"crossref","unstructured":"Voita, E., Talbot, D., Moiseev, F., Sennrich, R., Titov, I.: Analyzing multi-head self-attention: specialized heads do the heavy lifting, the rest can be pruned. In: Proceedings of the Annual Meeting of the Association for Computational Linguistics, pp. 5797\u20135808 (2019)","DOI":"10.18653\/v1\/P19-1580"},{"key":"1003_CR202","first-page":"8928","volume":"32","author":"C Chen","year":"2019","unstructured":"Chen, C., Li, O., Tao, D., Barnett, A., Rudin, C., Su, J.K.: This looks like that: deep learning for interpretable image recognition. Adv. Neural. Inf. Process. Syst. 32, 8928\u20138939 (2019)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1003_CR203","unstructured":"Koh, P.W., Liang, P.: Understanding black-box predictions via influence functions. In: Proceedings of the International Conference on Machine Learning, pp. 1885\u20131894 (2017)"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-022-01003-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-022-01003-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-022-01003-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,6]],"date-time":"2024-10-06T03:01:21Z","timestamp":1728183681000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-022-01003-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,10,19]]},"references-count":203,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2023,4]]}},"alternative-id":["1003"],"URL":"https:\/\/doi.org\/10.1007\/s00530-022-01003-8","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,10,19]]},"assertion":[{"value":"2 May 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 September 2022","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"19 October 2022","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}