{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T16:50:57Z","timestamp":1777654257718,"version":"3.51.4"},"reference-count":114,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2024,12,11]],"date-time":"2024-12-11T00:00:00Z","timestamp":1733875200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,12,11]],"date-time":"2024-12-11T00:00:00Z","timestamp":1733875200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100012166","name":"National Key R&D Program of China","doi-asserted-by":"crossref","award":["2021ZD0140407"],"award-info":[{"award-number":["2021ZD0140407"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["42327901"],"award-info":[{"award-number":["42327901"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62276150"],"award-info":[{"award-number":["62276150"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62321005"],"award-info":[{"award-number":["62321005"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2025,5]]},"DOI":"10.1007\/s11263-024-02296-0","type":"journal-article","created":{"date-parts":[[2024,12,11]],"date-time":"2024-12-11T16:13:34Z","timestamp":1733933614000},"page":"2752-2782","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["InfoPro: Locally Supervised Deep Learning by Maximizing Information Propagation"],"prefix":"10.1007","volume":"133","author":[{"given":"Yulin","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zanlin","family":"Ni","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yifan","family":"Pu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Cai","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jixuan","family":"Ying","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shiji","family":"Song","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7251-0988","authenticated-orcid":false,"given":"Gao","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,12,11]]},"reference":[{"key":"2296_CR1","unstructured":"Krizhevsky, A., Sutskever, I., & Hinton, G.E. (2012). Imagenet classification with deep convolutional neural networks. In NeurIPS,, pp. 1097\u20131105."},{"key":"2296_CR2","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2016). Deep residual learning for image recognition. in CVPR, pp. 770\u2013778.","DOI":"10.1109\/CVPR.2016.90"},{"key":"2296_CR3","doi-asserted-by":"publisher","first-page":"8704","DOI":"10.1109\/TPAMI.2019.2918284","volume":"44","author":"G Huang","year":"2019","unstructured":"Huang, G., Liu, Z., Pleiss, G., Van Der Maaten, L., & Weinberger, K. (2019). Convolutional networks with dense connectivity. IEEE Transactions on Pattern Analysis and Machine Intelligence, 44, 8704\u20138716.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"issue":"4","key":"2296_CR4","doi-asserted-by":"publisher","first-page":"1050","DOI":"10.1007\/s11263-022-01575-y","volume":"130","author":"K Han","year":"2022","unstructured":"Han, K., Wang, Y., Chang, X., Guo, J., Chunjing, X., Enhua, W., & Tian, Q. (2022). Ghostnets on heterogeneous devices via cheap operations. International Journal of Computer Vision, 130(4), 1050\u20131069.","journal-title":"International Journal of Computer Vision"},{"key":"2296_CR5","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S., Uszkoreit, J., & Houlsby, N. (2021). An image is worth 16x16 words: Transformers for image recognition at scale. In ICLR."},{"key":"2296_CR6","unstructured":"Touvron, H., Cord, M., Douze, M., Massa, F., Sablayrolles, A., & J\u00e9gou, H. (2021). Training data-efficient image transformers & distillation through attention. In ICML, pp. 10347\u201310357."},{"issue":"5","key":"2296_CR7","doi-asserted-by":"publisher","first-page":"1141","DOI":"10.1007\/s11263-022-01739-w","volume":"131","author":"Q Zhang","year":"2023","unstructured":"Zhang, Q., Yufei, X., Zhang, J., & Tao, D. (2023). Vitaev2: Vision transformer advanced by exploring inductive bias for image recognition and beyond. International Journal of Computer Vision, 131(5), 1141\u20131162.","journal-title":"International Journal of Computer Vision"},{"issue":"12","key":"2296_CR8","doi-asserted-by":"publisher","first-page":"3136","DOI":"10.1007\/s11263-023-01861-3","volume":"131","author":"M Lin","year":"2023","unstructured":"Lin, M., Chen, M., Zhang, Y., Shen, C., Ji, R., & Cao, L. (2023). Super vision transformer. International Journal of Computer Vision, 131(12), 3136\u20133151.","journal-title":"International Journal of Computer Vision"},{"key":"2296_CR9","unstructured":"Jaderberg, M., Czarnecki, W., Marian, O., Simon, V., Oriol, G., Alex, S.D., & Kavukcuoglu, K. (2017). Decoupled neural interfaces using synthetic gradients. In ICML, pp. 1627\u20131635."},{"key":"2296_CR10","doi-asserted-by":"crossref","unstructured":"Touvron, H., Cord, M., Sablayrolles, A., Synnaeve, G., & J\u00e9gou, H. (2021). Going deeper with image transformers. In ICCV, pp. 32\u201342.","DOI":"10.1109\/ICCV48922.2021.00010"},{"key":"2296_CR11","doi-asserted-by":"crossref","unstructured":"Ni, Z., Wang, Y., Jiangwei, Y., Jiang, H., Cao, Y., & Huang, G. (2023). Deep incubation: Training large models by divide-and-conquering. In ICCV, pp. 17335\u201317345.","DOI":"10.1109\/ICCV51070.2023.01590"},{"key":"2296_CR12","doi-asserted-by":"crossref","unstructured":"Zhai, X., Kolesnikov, A., Houlsby, N., & Beyer, L. (2022). Scaling vision transformers. In CVPR, pp. 12104\u201312113.","DOI":"10.1109\/CVPR52688.2022.01179"},{"issue":"6203","key":"2296_CR13","doi-asserted-by":"publisher","first-page":"129","DOI":"10.1038\/337129a0","volume":"337","author":"F Crick","year":"1989","unstructured":"Crick, F. (1989). The recent excitement about neural networks. Nature, 337(6203), 129\u2013132.","journal-title":"Nature"},{"key":"2296_CR14","doi-asserted-by":"publisher","DOI":"10.3389\/fncom.2016.00094","volume":"10","author":"AH Marblestone","year":"2016","unstructured":"Marblestone, A. H., Wayne, G., & Kording, K. P. (2016). Toward an integration of deep learning and neuroscience. Frontiers in Computational Neuroscience, 10, 215943.","journal-title":"Frontiers in Computational Neuroscience"},{"key":"2296_CR15","unstructured":"L\u00f6we, S., O\u2019Connor, P., & Veeling, B. (2019). Putting an end to end-to-end: Gradient-isolated learning of representations. In NeurIPS, pp. 3039\u20133051."},{"issue":"1","key":"2296_CR16","doi-asserted-by":"publisher","first-page":"23","DOI":"10.1016\/j.neuron.2004.09.007","volume":"44","author":"Y Dan","year":"2004","unstructured":"Dan, Y., & Poo, M. (2004). Spike timing-dependent plasticity of neural circuits. Neuron, 44(1), 23\u201330.","journal-title":"Neuron"},{"key":"2296_CR17","doi-asserted-by":"publisher","first-page":"25","DOI":"10.1146\/annurev.neuro.31.060407.125639","volume":"31","author":"N Caporale","year":"2008","unstructured":"Caporale, N., & Dan, Y. (2008). Spike timing-dependent plasticity: A hebbian learning rule. Annual Review of Neuroscience, 31, 25\u201346.","journal-title":"Annual Review of Neuroscience"},{"key":"2296_CR18","unstructured":"Bengio, Y., Lee, D.H., Bornschein, J., Mesnard, T., & Lin, Z. (2015) Towards biologically plausible deep learning. arXiv:1502.04156."},{"key":"2296_CR19","unstructured":"Mosca, A., & Magoulas, G.D. (2017) Deep incremental boosting. arXiv:1708.03704."},{"key":"2296_CR20","doi-asserted-by":"publisher","first-page":"608","DOI":"10.3389\/fnins.2018.00608","volume":"12","author":"H Mostafa","year":"2018","unstructured":"Mostafa, H., Ramesh, V., & Cauwenberghs, G. (2018). Deep supervised learning using local errors. Frontiers in Neuroscience, 12, 608.","journal-title":"Frontiers in Neuroscience"},{"key":"2296_CR21","unstructured":"Huang, F., Ash, J., Langford, J., & Schapire, R. (2018). Learning deep resnet blocks sequentially using boosting theory. In ICML, pp. 2058\u20132067."},{"key":"2296_CR22","unstructured":"Belilovsky, E., Eickenberg, M., & Oyallon, E. (2019). Greedy layerwise learning can scale to imagenet. In ICML, pp. 583\u2013593."},{"key":"2296_CR23","first-page":"736","volume":"56","author":"E Belilovsky","year":"2020","unstructured":"Belilovsky, E., Eickenberg, M., & Oyallon, E. (2020). Decoupled greedy learning of cnns. ICML, 56, 736\u2013745.","journal-title":"Decoupled greedy learning of cnns. ICML"},{"key":"2296_CR24","unstructured":"N\u00f8kland, A., & Eidnes, L.H. (2019). Training neural networks with local error signals. In ICML, pp. 4839\u20134850."},{"key":"2296_CR25","unstructured":"Wang, Y., Ni, Z., Song, S., Yang, L., & Huang, G. (2021). Revisiting locally supervised learning: an alternative to end-to-end training. In ICLR."},{"key":"2296_CR26","doi-asserted-by":"crossref","unstructured":"Carreira, J. & Zisserman, A. (2017). Quo vadis, action recognition? a new model and the kinetics dataset. In CVPR, pp. 6299\u20136308.","DOI":"10.1109\/CVPR.2017.502"},{"key":"2296_CR27","doi-asserted-by":"crossref","unstructured":"Lin, J., Gan, C., & Han, S. (2019). Tsm: Temporal shift module for efficient video understanding. In ICCV, pp. 7083\u20137093.","DOI":"10.1109\/ICCV.2019.00718"},{"key":"2296_CR28","first-page":"6105","volume":"97","author":"M Tan","year":"2019","unstructured":"Tan, M., & Quoc, V. L. (2019). Efficientnet: Rethinking model scaling for convolutional neural networks. In ICML, 97, 6105\u20136114.","journal-title":"In ICML"},{"key":"2296_CR29","first-page":"3596","volume":"35","author":"P Ye","year":"2022","unstructured":"Ye, P., Tang, S., Li, B., Chen, T., & Ouyang, W. (2022). Stimulative training of residual networks: A social psychology perspective of loafing. In NeurIPS, 35, 3596\u20133608.","journal-title":"In NeurIPS"},{"key":"2296_CR30","unstructured":"Ye, P., He, T., Tang, S., Li, B., Chen, T., Bai, L., & Ouyang, W. (2023) Stimulative training++: Go beyond the performance limits of residual networks. arXiv:2305.02507."},{"key":"2296_CR31","doi-asserted-by":"publisher","first-page":"1527","DOI":"10.1162\/neco.2006.18.7.1527","volume":"18","author":"GE Hinton","year":"2006","unstructured":"Hinton, G. E., Osindero, S., & Teh, Y. W. (2006). A fast learning algorithm for deep belief nets. Neural Computation, 18, 1527\u20131554.","journal-title":"Neural Computation"},{"key":"2296_CR32","doi-asserted-by":"crossref","unstructured":"Bengio, Y., Lamblin, P., Popovici, D., & Larochelle, H. (2007). Greedy layer-wise training of deep networks. In NeurIPS, pp. 153\u2013160.","DOI":"10.7551\/mitpress\/7503.003.0024"},{"key":"2296_CR33","unstructured":"Ioffe, S. & Szegedy, C. (2015) Batch normalization: Accelerating deep network training by reducing internal covariate shift. arXiv:1502.03167."},{"key":"2296_CR34","unstructured":"Kulkarni, M. & Karande, S. (2017) Layer-wise training of deep networks using kernel similarity. arXiv:1703.07115."},{"key":"2296_CR35","unstructured":"Malach, E. & Shalev-Shwartz, S. (2018) A provably correct algorithm for deep learning that actually works. arXiv:1803.09522 ."},{"issue":"11","key":"2296_CR36","doi-asserted-by":"publisher","first-page":"5475","DOI":"10.1109\/TNNLS.2018.2805098","volume":"29","author":"ES Marquez","year":"2018","unstructured":"Marquez, E. S., Hare, J. S., & Niranjan, M. (2018). Deep cascade learning. IEEE Transactions on Neural Networks and Learning Systems, 29(11), 5475\u20135485.","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"2296_CR37","unstructured":"Fahlman, S.E. & Lebiere, C. (1990). The cascade-correlation learning architecture. In NeurIPS, pp. 524\u2013532."},{"key":"2296_CR38","unstructured":"Xiong, Y., Ren, M., & Urtasun, R. (2020). Loco: Local contrastive representation learning. In NeurIPS, pp. 11142\u201311153."},{"key":"2296_CR39","unstructured":"Laskin, M., Metz, L., Nabarro, S., Saroufim, M., Noune, B., Luschi, C., Sohl-Dickstein, J., & Abbeel, P. (2020) Parallel training of deep networks with local updates. arXiv:2012.03837."},{"issue":"1","key":"2296_CR40","first-page":"7714","volume":"23","author":"AN Gomez","year":"2022","unstructured":"Gomez, A. N., Key, O., Perlin, K., Gou, S., Frosst, N., Dean, J., & Gal, Y. (2022). Interlocking backpropagation: Improving depthwise model-parallelism. Journal of Machine Learning Research, 23(1), 7714\u20137741.","journal-title":"Journal of Machine Learning Research"},{"key":"2296_CR41","unstructured":"Chen, T., Xu, B., Zhang, C., & Guestrin, C. (2016) Training deep nets with sublinear memory cost. arXiv:1604.06174."},{"key":"2296_CR42","first-page":"4125","volume":"29","author":"A Gruslys","year":"2016","unstructured":"Gruslys, A., Munos, R., Danihelka, I., Lanctot, M., & Graves, A. (2016). Memory-efficient backpropagation through time. NeurIPS, 29, 4125\u20134133.","journal-title":"Memory-efficient backpropagation through time. NeurIPS"},{"key":"2296_CR43","unstructured":"Gomez, A.N., Ren, M., Urtasun, R., & Grosse, R.B. (2017). The reversible residual network: Backpropagation without storing activations. In NeurIPS, pp. 2214\u20132224."},{"key":"2296_CR44","unstructured":"Salimans, T. & Bulatov, Y. (2017) Gradient checkpointing."},{"key":"2296_CR45","unstructured":"Jacobsen, JH, Smeulders, A, & Oyallon, E. (2018) i-revnet: Deep invertible networks. arXiv:1802.07088."},{"key":"2296_CR46","doi-asserted-by":"crossref","unstructured":"Lee, D.-H., Zhang, S., Fischer, A., & Bengio, Y. (2015). Difference target propagation. in Joint european conference on machine learning and knowledge discovery in databases, pp. 498\u2013515.","DOI":"10.1007\/978-3-319-23528-8_31"},{"key":"2296_CR47","unstructured":"Bartunov, S., Santoro, A., Richards, B., Marris, L., Hinton, G.E., & Lillicrap, T. (2018). Assessing the scalability of biologically-motivated deep learning algorithms and architectures. In NeurIPS, pp. 9368\u20139378."},{"key":"2296_CR48","unstructured":", Timothy, P., Cownden, D., Tweed, D.B., & Akerman C.J. (2014) Random feedback weights support learning in deep neural networks. arXiv:1411.0247."},{"key":"2296_CR49","unstructured":"N\u00f8kland, A. (2016). Direct feedback alignment provides learning in deep neural networks. In NeurIPS, pp. 1037\u20131045."},{"key":"2296_CR50","unstructured":"Taylor, G., Burmeister, R., Zheng, X., Singh, B., Patel, A., & Goldstein, T. (2016). Training neural networks without gradients: A scalable admm approach. In ICML, pp. 2722\u20132731."},{"key":"2296_CR51","first-page":"24","volume":"1050","author":"ECENYU Anna Choromanska","year":"2018","unstructured":"Anna Choromanska, E. C. E. N. Y. U., Tandon, S. K., Luss, R., Rish, I., Kingsbury, B., Tejwani, R., & Bouneffouf, D. (2018). Beyond backprop: Alternating minimization with co-activation memory. Statistics, 1050, 24.","journal-title":"Statistics"},{"key":"2296_CR52","unstructured":"Huo, Z., Bin, G., Yang, Q., & Huang, H. (2018). Decoupled parallel backpropagation with convergence guarantee. In ICML, pp. 2098\u20132106."},{"key":"2296_CR53","unstructured":"Huo, Z., Bin, G., & Huang, H. (2018). Training neural networks using features replay. In NeurIPS, pp. 6659\u20136668."},{"key":"2296_CR54","unstructured":"Shwartz-Ziv, R., & Tishby, N. (2017) Opening the black box of deep neural networks via information. arXiv:1703.00810."},{"issue":"12","key":"2296_CR55","doi-asserted-by":"publisher","DOI":"10.1088\/1742-5468\/ab3985","volume":"2019","author":"AM Saxe","year":"2019","unstructured":"Saxe, A. M., Bansal, Y., Dapello, J., Advani, M., Kolchinsky, A., Tracey, B. D., & Cox, D. D. (2019). On the information bottleneck theory of deep learning. Journal of Statistical Mechanics: Theory and Experiment, 2019(12), 124020.","journal-title":"Journal of Statistical Mechanics: Theory and Experiment"},{"key":"2296_CR56","unstructured":"Tishby, N., Pereira, F.C., & Bialek, W. (2000) The information bottleneck method. arXiv:physics\/0004057."},{"issue":"1","key":"2296_CR57","first-page":"1947","volume":"19","author":"A Achille","year":"2018","unstructured":"Achille, A., & Soatto, S. (2018). Emergence of invariance and disentanglement in deep representations. Journal of Machine Learning Research, 19(1), 1947\u20131980.","journal-title":"Journal of Machine Learning Research"},{"key":"2296_CR58","unstructured":"Alemi, A.A., Fischer, I., Dillon, J.V., & Murphy, K. (2016) Deep variational information bottleneck. arXiv:1612.00410."},{"key":"2296_CR59","unstructured":"van den Oord, A., Li, Y, & Vinyals, O. (2018) Representation learning with contrastive predictive coding. arXiv:1807.03748."},{"key":"2296_CR60","unstructured":"Tian, Y., Sun, C., Poole, B., Krishnan, D., Schmid, C., & Phillip I. (2020) What makes for good views for contrastive learning. arXiv:2005.10243."},{"key":"2296_CR61","unstructured":"Hjelm, R.D., Fedorov, A., Lavoie-Marchildon, S., Grewal, K., Bachman, P., Trischler, A., & Bengio, Y. (2019) Learning deep representations by mutual information estimation and maximization. In ICLR."},{"issue":"9","key":"2296_CR62","doi-asserted-by":"publisher","first-page":"2205","DOI":"10.1007\/s11263-022-01639-z","volume":"130","author":"Y Li","year":"2022","unstructured":"Li, Y., Yang, M., Peng, D., Li, T., Huang, J., & Peng, X. (2022). Twin contrastive learning for online clustering. International Journal of Computer Vision, 130(9), 2205\u20132221.","journal-title":"International Journal of Computer Vision"},{"key":"2296_CR63","unstructured":"Chen, T., Kornblith, S., Norouzi, M., & Hinton, G. (2020). A simple framework for contrastive learning of visual representations. In ICML, pp. 1597\u20131607."},{"key":"2296_CR64","doi-asserted-by":"crossref","unstructured":"He, K., Fan, H., Yuxin, W., Xie, S., & Girshick, R. (2020). Momentum contrast for unsupervised visual representation learning. In CVPR, pp. 9729\u20139738.","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"2296_CR65","volume-title":"Learning multiple layers of features from tiny images","author":"A Krizhevsky","year":"2009","unstructured":"Krizhevsky, A., Hinton, G., et al. (2009). Learning multiple layers of features from tiny images. Citeseer: Technical report."},{"key":"2296_CR66","unstructured":"Netzer, Y., Wang, T., Coates, A., Bissacco, A., Wu, B., & Ng, A.Y. (2011) Reading digits in natural images with unsupervised feature learning."},{"key":"2296_CR67","unstructured":"Coates, A., Ng, A., & Lee, H. (2011). An analysis of single-layer networks in unsupervised feature learning. In AISTATS, pp. 215\u2013223."},{"key":"2296_CR68","unstructured":"Lee, C.Y., Xie, S., Gallagher, P., Zhang, Z., & Zhuowen, T. (2015). Deeply-supervised nets. In AISTATS, pp. 562\u2013570."},{"key":"2296_CR69","unstructured":"Tschannen, M., Djolonga, J., Rubenstein, P.K., Gelly, S., Lucic, M. (2020) On mutual information maximization for representation learning. In ICLR."},{"key":"2296_CR70","unstructured":"Mohamed I.B., Aristide, B., Sai, R., Sherjil, O., Yoshua, B., Aaron, C., & Devon, H. (2018). Mutual information neural estimation. In ICML , pp. 531\u2013540."},{"issue":"11","key":"2296_CR71","doi-asserted-by":"publisher","first-page":"2278","DOI":"10.1109\/5.726791","volume":"86","author":"Y LeCun","year":"1998","unstructured":"LeCun, Y., Bottou, L., Bengio, Y., & Haffner, P. (1998). Gradient-based learning applied to document recognition. Proceedings of the IEEE, 86(11), 2278\u20132324.","journal-title":"Proceedings of the IEEE"},{"key":"2296_CR72","doi-asserted-by":"crossref","unstructured":"Vincent, P., Larochelle, H., Bengio, Y., & Manzagol, P.A. (2008). Extracting and composing robust features with denoising autoencoders. In ICML, pp. 1096\u20131103.","DOI":"10.1145\/1390156.1390294"},{"key":"2296_CR73","doi-asserted-by":"crossref","unstructured":"Rifai, S., Bengio, Y., Courville, A., Vincent, P., & Mirza, M. (2012). Disentangling factors of variation for facial expression recognition. In ECCV, pp. 808\u2013822.","DOI":"10.1007\/978-3-642-33783-3_58"},{"key":"2296_CR74","unstructured":"Kingma, D.P., & Welling, M. (2013) Auto-encoding variational bayes. arXiv:1312.6114."},{"key":"2296_CR75","unstructured":"Makhzani, A., Shlens, J., Jaitly, N., Goodfellow, I., & Frey, B. (2015) Adversarial autoencoders. arXiv:1511.05644."},{"key":"2296_CR76","unstructured":"Khosla, P., Teterwak, P., Wang, C., Sarna, A., Tian, Y., Isola, P., Maschinot, A., Liu, C., & Krishnan, D. (2020). Supervised contrastive learning. In NeurIPS , pp. 18661\u201318673."},{"key":"2296_CR77","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L., Li, Kai, & Fei-Fei, Li. (2009). Imagenet: A large-scale hierarchical image database. In ICML, pp. 248\u2013255.","DOI":"10.1109\/CVPR.2009.5206848"},{"issue":"1","key":"2296_CR78","doi-asserted-by":"publisher","first-page":"98","DOI":"10.1007\/s11263-014-0733-5","volume":"111","author":"SM Mark Everingham","year":"2015","unstructured":"Mark Everingham, S. M., Eslami, A., Van Gool, L., Williams, C. K. I., Winn, J., & Zisserman, A. (2015). The pascal visual object classes challenge: A retrospective. International Journal of Computer Vision, 111(1), 98\u2013136.","journal-title":"International Journal of Computer Vision"},{"key":"2296_CR79","doi-asserted-by":"crossref","unstructured":"Cordts, M., Omran, M., Ramos, S., Rehfeld, T., Enzweiler, M., Benenson, R., Franke, U., Roth, S., & Schiele, B. (2016). The cityscapes dataset for semantic urban scene understanding. In CVPR, pp. 3213\u20133223.","DOI":"10.1109\/CVPR.2016.350"},{"key":"2296_CR80","doi-asserted-by":"crossref","unstructured":"Zhou, B., Zhao, H., Puig, X., Fidler, S., Barriuso, A., & Torralba, A. (2017). Scene parsing through ade20k dataset. In CVPR, pp. 633\u2013641.","DOI":"10.1109\/CVPR.2017.544"},{"key":"2296_CR81","doi-asserted-by":"crossref","unstructured":"Kuehne, H., Jhuang, H., Garrote, E., Poggio, T., & Serre, T. (2011). Hmdb: a large video database for human motion recognition. In ICCV, pp. 2556\u20132563.","DOI":"10.1109\/ICCV.2011.6126543"},{"key":"2296_CR82","unstructured":"Soomro, K., Zamir, A.R., Shah, M. (2012) Ucf101: A dataset of 101 human actions classes from videos in the wild. arXiv:1212.0402."},{"key":"2296_CR83","doi-asserted-by":"crossref","unstructured":"Goyal, R., Kahou, S.E., Michalski, V., Materzynska, J., Westphal, S., Kim, H., Haenel, V., Fruend, I., Yianilos, P., Mueller-Freitag, M., et al. (2017). The\" something something\" video database for learning and evaluating visual common sense. In ICCV, pp. 5842\u20135850.","DOI":"10.1109\/ICCV.2017.622"},{"key":"2296_CR84","doi-asserted-by":"crossref","unstructured":"Lin, T.Y., Maire, M., Belongie, S., Bourdev, L., Girshick, R., Hays, P., James, P., Ramanan, D., Lawrence Zitnick, C., & Doll\u00e1r, P. (2014) Microsoft coco: Common objects in context.","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"2296_CR85","unstructured":"Wang, Y., Pan, X., Song, S., Zhang, H., Huang, G., & Cheng, W. (2019). Implicit semantic data augmentation for deep networks. In NeurIPS, pp. 12635\u201312644."},{"key":"2296_CR86","unstructured":"Tarvainen, A., & Valpola, H. (2017). Mean teachers are better role models: Weight-averaged consistency targets improve semi-supervised deep learning results. In NeurIPS, pp. 1195\u20131204."},{"key":"2296_CR87","doi-asserted-by":"crossref","unstructured":"Hariharan, B., Arbel\u00e1ez, P., Bourdev, L., Maji, S., & Malik, J. (2011). Semantic contours from inverse detectors. In ICCV, pp. 991\u2013998.","DOI":"10.1109\/ICCV.2011.6126343"},{"key":"2296_CR88","unstructured":"Chen, L.C., Papandreou, G., Schroff, F., & Adam, H. (2017) Rethinking atrous convolution for semantic image segmentation. arXiv:1706.05587."},{"key":"2296_CR89","unstructured":"MMSegmentation Contributors. (2020) Mmsegmentation, an open source semantic segmentation toolbox. https:\/\/github.com\/open-mmlab\/mmsegmentation."},{"key":"2296_CR90","doi-asserted-by":"crossref","unstructured":"Wang, Y., Chen, Z., Jiang, H., Song, S., Han, Y., & Huang, G. (2021). Adaptive focus for efficient video recognition. In ICCV, pp. 16249\u201316258.","DOI":"10.1109\/ICCV48922.2021.01594"},{"key":"2296_CR91","doi-asserted-by":"crossref","unstructured":"Wang, Y., Yue, Y., Lin, Y., Jiang, H., Lai, Z., Kulikov, V., Orlov, N., Shi, H., & Huang, G. (2022a). Adafocus v2: End-to-end training of spatial dynamic networks for video recognition. In CVPR, pp. 20030\u201320040.","DOI":"10.1109\/CVPR52688.2022.01943"},{"key":"2296_CR92","doi-asserted-by":"crossref","unstructured":"Wang, Y., Yue, Y., Xinhong, X., Hassani, A., Kulikov, V., Orlov, N., Song, S., Shi, H., & Huang, G. (2022b). Adafocusv3: On unified spatial-temporal dynamic video recognition. In ECCV, pp. 226\u2013243.","DOI":"10.1007\/978-3-031-19772-7_14"},{"key":"2296_CR93","unstructured":"Chen, K., Wang, J., Pang, J., Yuhang Cao, Y., Xiong, X.L., Sun, S., Feng, W., Liu, Z., Jiarui, X., Zhang, Z., Cheng, D., Zhu, C., Cheng, T., Zhao, Q., Li, B., Xin, L., Zhu, R., Yue, W., Dai, J., Lin, D. (2019). MMDetection: Open mmlab detection toolbox and benchmark. arXiv:1906.07155."},{"key":"2296_CR94","unstructured":"Jacobsen, J\u00f6rn-Henrik, S., Arnold W.M., & Oyallon, Edouard. (2018). i-revnet: Deep invertible networks. In ICLR."},{"key":"2296_CR95","unstructured":"Cubuk, Ekin D., Zoph, B., Mane, D., Vasudevan, V., & Le, Q.\u00a0V. (2018). Autoaugment: Learning augmentation policies from data. arXiv preprintarXiv:1805.09501, ."},{"key":"2296_CR96","doi-asserted-by":"crossref","unstructured":"Cubuk, Ekin D., Zoph, B., Shlens, J., & Le, Q. V. (2020). Randaugment: Practical automated data augmentation with a reduced search space. In CVPRW, 702\u2013703.","DOI":"10.1109\/CVPRW50498.2020.00359"},{"key":"2296_CR97","doi-asserted-by":"crossref","unstructured":"Zhong, Z., Zheng, L., Kang, G., Li, S., & Yang, Y. (2020). Random erasing data augmentation. In AAAI, 34, 13001\u201313008.","DOI":"10.1609\/aaai.v34i07.7000"},{"key":"2296_CR98","unstructured":"Zhang, H., Cisse, M., Dauphin, Yann N., & Lopez-Paz, D. (2017). mixup: Beyond empirical risk minimization. arXiv preprintarXiv:1710.09412."},{"key":"2296_CR99","doi-asserted-by":"crossref","unstructured":"Yun, S., Han, D., Seong J., Oh., Chun, S., Choe, J., & Yoo, Y., (2019). Cutmix: Regularization strategy to train strong classifiers with localizable features. In ICCV, 6023\u20136032.","DOI":"10.1109\/ICCV.2019.00612"},{"key":"2296_CR100","doi-asserted-by":"crossref","unstructured":"Gao Huang, Y., Sun, Z. L., Sedra, D., & Weinberger, K. Q. (2016). Deep networks with stochastic depth. In ECCV, 646\u2013661.","DOI":"10.1007\/978-3-319-46493-0_39"},{"issue":"2","key":"2296_CR101","doi-asserted-by":"publisher","first-page":"157","DOI":"10.1109\/72.279181","volume":"5","author":"Y Bengio","year":"1994","unstructured":"Bengio, Y., Simard, P., & Frasconi, P. (1994). Learning long-term dependencies with gradient descent is difficult. IEEE Transactions on Neural Networks, 5(2), 157\u2013166.","journal-title":"IEEE Transactions on Neural Networks"},{"key":"2296_CR102","unstructured":"Ren, S., He, K., Girshick, R., & Sun, J. (2015). Faster r-cnn: Towards real-time object detection with region proposal networks. In NeurIPS, 91\u201399."},{"key":"2296_CR103","unstructured":"Chen, Z., Duan, Y., Wang, W., He, J., Lu, T., Dai, J., & Qiao, Y. (2023). Vision transformer adapter for dense predictions. In ICLR."},{"key":"2296_CR104","doi-asserted-by":"crossref","unstructured":"Zhao, H., Shi, J., Qi, X., Wang, X., & Jia, J. (2017). Pyramid scene parsing network. In CVPR, 2881\u20132890.","DOI":"10.1109\/CVPR.2017.660"},{"key":"2296_CR105","unstructured":"Jun, F., Liu, J., Tian, H., Li, Y., Bao, Y., Fang, Z., & Hanqing, Lu. (2019). Dual attention network for scene segmentation. In CVPR, 3146\u20133154."},{"key":"2296_CR106","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., & Girshick, R. (2017). Mask r-cnn. In ICCV, 2961\u20132969.","DOI":"10.1109\/ICCV.2017.322"},{"key":"2296_CR107","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., Doll\u00e1r, P., Girshick, R., He, K., Hariharan, B., & Belongie, S. (2017). Feature pyramid networks for object detection. In CVPR, 2117\u20132125.","DOI":"10.1109\/CVPR.2017.106"},{"key":"2296_CR108","unstructured":"Veit, A., Wilber, M. J., & Belongie, S. (2016). Residual networks behave like ensembles of relatively shallow networks. In NeurIPS, 550\u2013558."},{"key":"2296_CR109","unstructured":"Achiam, J., Adler, S., Agarwal, S., Ahmad, L., Akkaya, I., Aleman, F.\u00a0L., Almeida, D., Altenschmidt, J., Altman, S., Anadkat, S., et\u00a0al. (2023) Gpt-4 technical report. arXiv preprintarXiv:2303.08774."},{"key":"2296_CR110","unstructured":"Touvron, H., Martin, L., Stone, K., Albert, P., Almahairi, A., Babaei, Y., Bashlykov, N., Batra, S., Bhargava, P., Bhosale, S., et\u00a0al. (2023) Llama 2: Open foundation and fine-tuned chat models. arXiv preprintarXiv:2307.09288."},{"key":"2296_CR111","unstructured":"Radford, A., Kim, J. W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., et\u00a0al. (2021). Learning transferable visual models from natural language supervision. In ICML,56, 8748\u20138763, PMLR."},{"key":"2296_CR112","doi-asserted-by":"crossref","unstructured":"Sandler, M., Howard, A., Zhu, M., Zhmoginov, A., & Chen, L.-C. (2018). Mobilenetv 2: Inverted residuals and linear bottlenecks. In CVPR, 4510\u20134520.","DOI":"10.1109\/CVPR.2018.00474"},{"key":"2296_CR113","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1016\/j.neunet.2017.12.012","volume":"107","author":"S Elfwing","year":"2018","unstructured":"Elfwing, S., Uchibe, E., & Doya, K. (2018). Sigmoid-weighted linear units for neural network function approximation in reinforcement learning. Neural Networks, 107, 3\u201311.","journal-title":"Neural Networks"},{"key":"2296_CR114","unstructured":"Kingma, D. P., & Ba, J. (2014) Adam: A method for stochastic optimization. arXiv preprintarXiv:1412.6980, ."}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-024-02296-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-024-02296-0\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-024-02296-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,4,17]],"date-time":"2025-04-17T06:04:56Z","timestamp":1744869896000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-024-02296-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,11]]},"references-count":114,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2025,5]]}},"alternative-id":["2296"],"URL":"https:\/\/doi.org\/10.1007\/s11263-024-02296-0","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,12,11]]},"assertion":[{"value":"6 November 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"29 October 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 December 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}