{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,30]],"date-time":"2026-03-30T06:17:10Z","timestamp":1774851430626,"version":"3.50.1"},"reference-count":57,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T00:00:00Z","timestamp":1772323200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T00:00:00Z","timestamp":1772323200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Data Min Knowl Disc"],"published-print":{"date-parts":[[2026,3]]},"DOI":"10.1007\/s10618-025-01183-7","type":"journal-article","created":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T08:41:45Z","timestamp":1772786505000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Emp: enhance memory in data pruning"],"prefix":"10.1007","volume":"40","author":[{"given":"Jinying","family":"Xiao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ping","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jie","family":"Nie","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bin","family":"Ji","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shasha","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaodong","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jun","family":"Ma","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qingbo","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jie","family":"Yu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,3,6]]},"reference":[{"key":"1183_CR1","unstructured":"Achiam J, Adler S, Agarwal S, Ahmad L, Akkaya I, Aleman FL, Almeida D, Altenschmidt J, Altman S, Anadkat S, et al. (2023) Gpt-4 technical report. arXiv preprint arXiv:2303.08774"},{"issue":"50","key":"1183_CR2","first-page":"1","volume":"19","author":"A Achille","year":"2018","unstructured":"Achille A, Soatto S (2018) Emergence of invariance and disentanglement in deep representations. J Mach Learn Res 19(50):1\u201334","journal-title":"J Mach Learn Res"},{"key":"1183_CR3","doi-asserted-by":"crossref","unstructured":"Agarwal S, Arora H, Anand S, Arora C (2020) Contextual diversity for active learning. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XVI 16, pp. 137\u2013153 . Springer","DOI":"10.1007\/978-3-030-58517-4_9"},{"key":"1183_CR4","unstructured":"Aljundi R, Lin M, Goujaud B, Bengio Y (2019) Gradient based sample selection for online continual learning. Advances in neural information processing systems 32"},{"key":"1183_CR5","doi-asserted-by":"crossref","unstructured":"Amari S-i (1993) Backpropagation and stochastic gradient descent method. Neurocomputing 5(4-5), 185\u2013196","DOI":"10.1016\/0925-2312(93)90006-O"},{"key":"1183_CR6","unstructured":"Beyer L, H\u00e9naff OJ, Kolesnikov A, Zhai X, Oord Avd (2020) Are we done with imagenet? arXiv preprint arXiv:2006.07159"},{"key":"1183_CR7","doi-asserted-by":"crossref","unstructured":"Chen A, Yao Y, Chen P-Y, Zhang Y, Liu S (2023) Understanding and improving visual prompting: A label-mapping perspective. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 19133\u201319143","DOI":"10.1109\/CVPR52729.2023.01834"},{"key":"1183_CR8","unstructured":"Chen T, Kornblith S, Norouzi M, Hinton G (2020) A simple framework for contrastive learning of visual representations. In: International Conference on Machine Learning, pp. 1597\u20131607 . PMLR"},{"key":"1183_CR9","unstructured":"Coleman C, Yeh C, Mussmann S, Mirzasoleiman B, Bailis P, Liang P, Leskovec J, Zaharia M (2019) Selection via proxy: Efficient data selection for deep learning. In: International Conference on Learning Representations"},{"key":"1183_CR10","unstructured":"Dehghani M, Djolonga J, Mustafa B, Padlewski P, Heek J, Gilmer J, Steiner AP, Caron M, Geirhos R, Alabdulmohsin I (2023) Scaling vision transformers to 22 billion parameters. In: International Conference on Machine Learning, pp. 7480\u20137512 .PMLR"},{"key":"1183_CR11","doi-asserted-by":"crossref","unstructured":"Delobelle P, Winters T, Berendt B (2020) Robbert: a dutch roberta-based language model. In: Findings of the Association for Computational Linguistics: EMNLP 2020, pp. 3255\u20133265","DOI":"10.18653\/v1\/2020.findings-emnlp.292"},{"key":"1183_CR12","unstructured":"Devlin J (2018) Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805"},{"key":"1183_CR13","unstructured":"Ducoffe M, Precioso F (2018) Adversarial active learning for deep networks: a margin based approach. arXiv preprint arXiv:1802.09841"},{"issue":"10","key":"1183_CR14","doi-asserted-by":"publisher","first-page":"6648","DOI":"10.4249\/scholarpedia.6648","volume":"3","author":"RM Fano","year":"2008","unstructured":"Fano RM (2008) Fano inequality. Scholarpedia 3(10):6648","journal-title":"Scholarpedia"},{"key":"1183_CR15","first-page":"2881","volume":"33","author":"V Feldman","year":"2020","unstructured":"Feldman V, Zhang C (2020) What neural networks memorize and why: Discovering the long tail via influence estimation. Adv Neural Inf Process Syst 33:2881\u20132891","journal-title":"Adv Neural Inf Process Syst"},{"key":"1183_CR16","doi-asserted-by":"crossref","unstructured":"Feldman V (2020) Does learning require memorization? a short tale about a long tail. In: Proceedings of the 52nd Annual ACM SIGACT Symposium on Theory of Computing, pp. 954\u2013959","DOI":"10.1145\/3357713.3384290"},{"key":"1183_CR17","unstructured":"Goodfellow IJ, Vinyals O, Saxe AM (2014) Qualitatively characterizing neural network optimization problems. arXiv preprint arXiv:1412.6544"},{"key":"1183_CR18","unstructured":"Gu J, Tresp V (2019) Neural network memorization dissection. arXiv preprint arXiv:1911.09537"},{"key":"1183_CR19","doi-asserted-by":"crossref","unstructured":"Gui J, Chen T, Zhang J, Cao Q, Sun Z, Luo H, Tao D (2024) A survey on self-supervised learning: Algorithms, applications, and future trends. IEEE Trans Pattern Anal Mach Intell","DOI":"10.1109\/TPAMI.2024.3415112"},{"key":"1183_CR20","unstructured":"Harutyunyan H, Reing K, Ver\u00a0Steeg G, Galstyan A (2020) Improving generalization by controlling label-noise information in neural network weights. In: International Conference on Machine Learning, pp. 4071\u20134081 . PMLR"},{"key":"1183_CR21","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"issue":"1","key":"1183_CR22","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1162\/neco.1997.9.1.1","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter S, Schmidhuber J (1997) Flat minima. Neural Comput 9(1):1\u201342","journal-title":"Neural Comput"},{"key":"1183_CR23","unstructured":"Hochreiter S, Schmidhuber J (1994) Simplifying neural nets by discovering flat minima. Adv Neural Inform Process Syst 7"},{"key":"1183_CR24","doi-asserted-by":"crossref","unstructured":"Killamsetty K, Sivasubramanian D, Ramakrishnan G, Iyer R (2021) Glister: Generalization based data subset selection for efficient and robust learning. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 35, pp. 8110\u20138118","DOI":"10.1609\/aaai.v35i9.16988"},{"key":"1183_CR25","unstructured":"Kim W, Son B, Kim I (2021) Vilt: Vision-and-language transformer without convolution or region supervision. In: International Conference on Machine Learning, pp. 5583\u20135594 . PMLR"},{"key":"1183_CR26","unstructured":"Kingma DP, Ba J (2014) Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980"},{"issue":"1","key":"1183_CR27","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1214\/aos\/1015362183","volume":"30","author":"V Koltchinskii","year":"2002","unstructured":"Koltchinskii V, Panchenko D (2002) Empirical margin distributions and bounding the generalization error of combined classifiers. Ann Stat 30(1):1\u201350","journal-title":"Ann Stat"},{"key":"1183_CR28","unstructured":"Krizhevsky A, Nair V, Hinton G (2010) Cifar-10 (canadian institute for advanced research). URL http:\/\/www.cs.toronto.edu\/kriz\/cifar.html 5(4), 1"},{"key":"1183_CR29","unstructured":"Mirzasoleiman B, Bilmes J, Leskovec J (2020) Coresets for data-efficient training of machine learning models. In: International Conference on Machine Learning, pp. 6950\u20136960 . PMLR"},{"key":"1183_CR30","doi-asserted-by":"crossref","unstructured":"Patrini G, Rozza A, Krishna\u00a0Menon A, Nock R, Qu L (2017) Making deep neural networks robust to label noise: A loss correction approach. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1944\u20131952","DOI":"10.1109\/CVPR.2017.240"},{"key":"1183_CR31","first-page":"20596","volume":"34","author":"M Paul","year":"2021","unstructured":"Paul M, Ganguli S, Dziugaite GK (2021) Deep learning on a data diet: Finding important examples early in training. Adv Neural Inf Process Syst 34:20596\u201320607","journal-title":"Adv Neural Inf Process Syst"},{"key":"1183_CR32","unstructured":"Qin Z, Wang K, Zheng Z, Gu J, Peng X, Zhao Pan Zhou D, Shang L, Sun B, Xie X, You Y (2024) Infobatch: Lossless training speed up by unbiased dynamic data pruning. In: The Twelfth International Conference on Learning Representations . https:\/\/openreview.net\/forum?id=C61sk5LsK6"},{"key":"1183_CR33","unstructured":"Radford A, Kim JW, Hallacy C, Ramesh A, Goh G, Agarwal S, Sastry G, Askell A, Mishkin P, Clark J (2021) Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PMLR"},{"key":"1183_CR34","unstructured":"Raghu M, Gilmer J, Yosinski J, Sohl-Dickstein J (2017) Svcca: Singular vector canonical correlation analysis for deep learning dynamics and interpretability. Adv Neural Inform Process Syst 30"},{"key":"1183_CR35","unstructured":"Raju RS, Daruwalla K, Lipasti M (2021) Accelerating deep learning with dynamic data pruning. arXiv preprint arXiv:2111.12621"},{"key":"1183_CR36","unstructured":"Rolnick D, Veit A, Belongie S, Shavit N (2017) Deep learning is robust to massive label noise. arXiv preprint arXiv:1705.10694"},{"key":"1183_CR37","unstructured":"Sener O, Savarese S (2018) Active learning for convolutional neural networks: A core-set approach. In: International Conference on Learning Representations"},{"key":"1183_CR38","doi-asserted-by":"publisher","DOI":"10.1016\/j.media.2023.102802","volume":"88","author":"F Shamshad","year":"2023","unstructured":"Shamshad F, Khan S, Zamir SW, Khan MH, Hayat M, Khan FS, Fu H (2023) Transformers in medical imaging: A survey. Med Image Anal 88:102802","journal-title":"Med Image Anal"},{"key":"1183_CR39","unstructured":"Smith LN, Topin N (2017) Super-convergence: Very fast training of neural networks using large learning rates"},{"key":"1183_CR40","unstructured":"Sordoni A, Dziri N, Schulz H, Gordon G, Bachman P, Des\u00a0Combes RT (2021) Decomposed mutual information estimation for contrastive representation learning. In: International Conference on Machine Learning, pp. 9859\u20139869 . PMLR"},{"key":"1183_CR41","unstructured":"Stephenson C, Ganesh A, Hui Y, Tang H, Chung S, et al. (2021) On the geometry of generalization and memorization in deep neural networks. In: International Conference on Learning Representations"},{"key":"1183_CR42","unstructured":"Tan H, Wu S, Du F, Chen Y, Wang Z, Wang F, Qi X (2024) Data pruning via moving-one-sample-out. Advances in Neural Information Processing Systems 36"},{"key":"1183_CR43","unstructured":"Toneva M, Sordoni A, Combes RT, Trischler A, Bengio Y, Gordon GJ (2018) An empirical study of example forgetting during deep neural network learning. In: International Conference on Learning Representations"},{"key":"1183_CR44","unstructured":"Touvron H, Martin L, Stone K, Albert P, Almahairi A, Babaei Y, Bashlykov, N, Batra S, Bhargava P, Bhosale S, et al. (2023) Llama 2: Open foundation and fine-tuned chat models. arXiv preprint arXiv:2307.09288"},{"key":"1183_CR45","unstructured":"Vysogorets AM, Ahuja K, Kempe J (2024) Drop: Distributionally robust data pruning. In: The Thirteenth International Conference on Learning Representations"},{"key":"1183_CR46","doi-asserted-by":"crossref","unstructured":"Wang A (2018) Glue: A multi-task benchmark and analysis platform for natural language understanding. arXiv preprint arXiv:1804.07461","DOI":"10.18653\/v1\/W18-5446"},{"key":"1183_CR47","unstructured":"Wei J, Zhang Y, Zhang LY, Ding M, Chen C, Ong K-L, Zhang J, Xiang Y (2024) Memorization in deep learning: A survey. arXiv preprint arXiv:2406.03880"},{"key":"1183_CR48","doi-asserted-by":"crossref","unstructured":"Welling M (2009) Herding dynamical weights to learn. In: Proceedings of the 26th Annual International Conference on Machine Learning, pp. 1121\u20131128","DOI":"10.1145\/1553374.1553517"},{"key":"1183_CR49","doi-asserted-by":"crossref","unstructured":"Wu J, Chen J, Wu J, Shi W, Wang X, He X (2024) Understanding contrastive learning via distributionally robust optimization. Adv Neural Inform Process Syst 36","DOI":"10.52202\/075280-1010"},{"key":"1183_CR50","unstructured":"Xia X, Liu J, Yu J, Shen X, Han B, Liu T (2022) Moderate coreset: A universal method of data selection for real-world data-efficient deep learning. In: The Eleventh International Conference on Learning Representations"},{"key":"1183_CR51","doi-asserted-by":"crossref","unstructured":"Xiao J, Li P, Nie J (2024) Ted: Accelerate model training by internal generalization. arXiv preprint arXiv:2405.03228","DOI":"10.3233\/FAIA240823"},{"key":"1183_CR52","unstructured":"Yang S, Xie Z, Peng H, Xu M, Sun M, Li P (2022) Dataset pruning: Reducing training data by examining generalization influence. In: The Eleventh International Conference on Learning Representations"},{"key":"1183_CR53","unstructured":"Yang Z (2019) Xlnet: Generalized autoregressive pretraining for language understanding. arXiv preprint arXiv:1906.08237"},{"key":"1183_CR54","unstructured":"Yosinski J, Clune J, Bengio Y, Lipson H (2014) How transferable are features in deep neural networks? Adv Neural Inform Process Syst 27"},{"key":"1183_CR55","unstructured":"You Y, Gitman I, Ginsburg B (2017) Large batch training of convolutional networks. arXiv preprint arXiv:1708.03888"},{"issue":"3","key":"1183_CR56","doi-asserted-by":"publisher","first-page":"107","DOI":"10.1145\/3446776","volume":"64","author":"C Zhang","year":"2021","unstructured":"Zhang C, Bengio S, Hardt M, Recht B, Vinyals O (2021) Understanding deep learning (still) requires rethinking generalization. Commun ACM 64(3):107\u2013115","journal-title":"Commun ACM"},{"key":"1183_CR57","first-page":"39321","volume":"36","author":"C Zhang","year":"2023","unstructured":"Zhang C, Ippolito D, Lee K, Jagielski M, Tram\u00e8r F, Carlini N (2023) Counterfactual memorization in neural language models. Adv Neural Inf Process Syst 36:39321\u201339362","journal-title":"Adv Neural Inf Process Syst"}],"container-title":["Data Mining and Knowledge Discovery"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10618-025-01183-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10618-025-01183-7","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10618-025-01183-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,30]],"date-time":"2026-03-30T04:54:58Z","timestamp":1774846498000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10618-025-01183-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3]]},"references-count":57,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2026,3]]}},"alternative-id":["1183"],"URL":"https:\/\/doi.org\/10.1007\/s10618-025-01183-7","relation":{},"ISSN":["1384-5810","1573-756X"],"issn-type":[{"value":"1384-5810","type":"print"},{"value":"1573-756X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,3]]},"assertion":[{"value":"18 December 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 December 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 March 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval and consent to participate"}},{"value":"All authors consent to the publication of this work.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for publication"}}],"article-number":"24"}}