{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T15:44:48Z","timestamp":1783784688544,"version":"3.55.0"},"reference-count":26,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2024,3,8]],"date-time":"2024-03-08T00:00:00Z","timestamp":1709856000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,3,8]],"date-time":"2024-03-08T00:00:00Z","timestamp":1709856000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"Major national statistical science research projects of China","award":["2020LD02"],"award-info":[{"award-number":["2020LD02"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Comput Stat"],"published-print":{"date-parts":[[2025,1]]},"DOI":"10.1007\/s00180-024-01471-8","type":"journal-article","created":{"date-parts":[[2024,3,8]],"date-time":"2024-03-08T06:29:26Z","timestamp":1709879366000},"page":"27-64","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Imbalanced data sampling design based on grid boundary domain for big data"],"prefix":"10.1007","volume":"40","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4201-6479","authenticated-orcid":false,"given":"Hanji","family":"He","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jianfeng","family":"He","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Liwei","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,3,8]]},"reference":[{"issue":"1","key":"1471_CR1","doi-asserted-by":"publisher","first-page":"321","DOI":"10.1613\/jair.953","volume":"16","author":"N Chawla","year":"2002","unstructured":"Chawla N, Bowyer K, Hall L, Kegelmeyer W (2002) SMOTE: synthetic minority oversampling technique. J Artif Intell Res 16(1):321\u2013357. https:\/\/doi.org\/10.1613\/jair.953","journal-title":"J Artif Intell Res"},{"key":"1471_CR2","doi-asserted-by":"publisher","first-page":"112","DOI":"10.1016\/j.jspi.2020.03.004","volume":"209","author":"Q Cheng","year":"2020","unstructured":"Cheng Q, Wang HY, Yang M (2020) Information-based optimal subdata selection for big data logistic regression. J Stat Plan Inference 209:112\u2013122. https:\/\/doi.org\/10.1016\/j.jspi.2020.03.004","journal-title":"J Stat Plan Inference"},{"key":"1471_CR3","unstructured":"Clough RW (1960) The finite element method in plane stress analysis. In: proceedings of the 2nd ASCE conference on electronic computation. American Society of Civil Engineers, pp 345\u2013378. http:\/\/refhub.elsevier.com\/S1674-7755(18)30451-7\/sref20"},{"key":"1471_CR4","doi-asserted-by":"publisher","first-page":"2505","DOI":"10.48550\/arXiv.1802.06749","volume":"2018","author":"M Derezinski","year":"2018","unstructured":"Derezinski M, Warmuth MKK, Hsu DJ (2018) Leveraged volume sampling for linear regression. Adv Neural Inform Process Syst 2018:2505\u20132514. https:\/\/doi.org\/10.48550\/arXiv.1802.06749","journal-title":"Adv Neural Inform Process Syst"},{"key":"1471_CR5","volume-title":"Subspace sampling and relative-error matrix approximation: column-row-based methods","author":"P Drineas","year":"2006","unstructured":"Drineas P, Mahoney M, Muthukrishnan S (2006) Subspace sampling and relative-error matrix approximation: column-row-based methods. Springer, Berlin"},{"key":"1471_CR6","doi-asserted-by":"publisher","first-page":"878","DOI":"10.1007\/11538059_91","volume":"91","author":"H Han","year":"2005","unstructured":"Han H, Wang WY, Mao BH (2005) Borderline-SMOTE: a new over-sampling method in imbalanced data sets learning. Lect Notes Comput Sci 91:878\u2013887. https:\/\/doi.org\/10.1007\/11538059_91","journal-title":"Lect Notes Comput Sci"},{"key":"1471_CR7","unstructured":"Kub\u00e1t M, Matwin S (1997) Addressing the curse of imbalanced training sets: one-sided selection. In: proceedings of the fourteenth international conference on machine learning, pp 179\u2013186"},{"key":"1471_CR8","doi-asserted-by":"publisher","first-page":"17","DOI":"10.1016\/j.ins.2017.05.008","volume":"409\u2013410","author":"WC Lin","year":"2017","unstructured":"Lin WC, Tsai CF, Hu YH, Jhang JS (2017) Clustering-based undersampling in class-imbalanced data. Inform Sci 409\u2013410:17\u201326. https:\/\/doi.org\/10.1016\/j.ins.2017.05.008","journal-title":"Inform Sci"},{"issue":"2","key":"1471_CR9","doi-asserted-by":"publisher","first-page":"539","DOI":"10.1109\/TSMCB.2008.2007853","volume":"39","author":"XY Liu","year":"2009","unstructured":"Liu XY, Wu J, Zhou ZH (2009) Exploratory undersampling for class-imbalance learning. IEEE Trans Syst Man Cybern Part B 39(2):539\u2013550. https:\/\/doi.org\/10.1109\/TSMCB.2008.2007853","journal-title":"IEEE Trans Syst Man Cybern Part B"},{"key":"1471_CR10","doi-asserted-by":"publisher","first-page":"861","DOI":"10.48550\/arXiv.1306.5362","volume":"16","author":"P Ma","year":"2015","unstructured":"Ma P, Mahoney MW, Yu B (2015) A statistical perspective on algorithmic leveraging. J Mach Learn Res 16:861\u2013911. https:\/\/doi.org\/10.48550\/arXiv.1306.5362","journal-title":"J Mach Learn Res"},{"issue":"177","key":"1471_CR11","first-page":"1","volume":"23","author":"P Ma","year":"2022","unstructured":"Ma P et al (2022) Asymptotic analysis of sampling estimators for randomized numerical linear algebra algorithms. JMLR 23(177):1\u201345","journal-title":"JMLR"},{"issue":"2","key":"1471_CR12","doi-asserted-by":"publisher","first-page":"647","DOI":"10.1561\/2200000035","volume":"3","author":"MW Mahoney","year":"2011","unstructured":"Mahoney MW (2011) Randomized algorithms for matrices and data. Adv Mach Learn Data Min Astron 3(2):647\u2013672. https:\/\/doi.org\/10.1561\/2200000035","journal-title":"Adv Mach Learn Data Min Astron"},{"issue":"3","key":"1471_CR13","doi-asserted-by":"publisher","first-page":"213","DOI":"10.1007\/s00500-008-0319-7","volume":"13","author":"A Orriols-Puig","year":"2009","unstructured":"Orriols-Puig A, Bernado-Mansilla E (2009) Evolutionary rule based systems for imbalanced data sets. Soft Comput 13(3):213\u2013225. https:\/\/doi.org\/10.1007\/s00500-008-0319-7","journal-title":"Soft Comput"},{"key":"1471_CR14","doi-asserted-by":"publisher","first-page":"1214","DOI":"10.1016\/j.ins.2019.10.048","volume":"512","author":"T Pan","year":"2020","unstructured":"Pan T, Zhao J, Wu W (2020) Learning imbalanced datasets based on SMOTE and Gaussian distribution. Inf Sci 512:1214\u20131233. https:\/\/doi.org\/10.1016\/j.ins.2019.10.048","journal-title":"Inf Sci"},{"issue":"1","key":"1471_CR15","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s00607-020-00854-1","volume":"103","author":"S Park","year":"2021","unstructured":"Park S, Park H (2021) Combined oversampling and undersampling method based on slow-start algorithm for imbalanced network traffic. Computing 103(1):1\u201324. https:\/\/doi.org\/10.1007\/s00607-020-00854-1","journal-title":"Computing"},{"key":"1471_CR16","doi-asserted-by":"publisher","first-page":"245","DOI":"10.1007\/s10115-011-0465-6","volume":"33","author":"E Ramentol","year":"2012","unstructured":"Ramentol E, Caballero Y, Bello R et al (2012) SMOTE-RSB *: a hybrid preprocessing approach based on oversampling and undersampling for high imbalanced data-sets using SMOTE and rough sets theory. Knowl Inf Syst 33:245\u2013265. https:\/\/doi.org\/10.1007\/s10115-011-0465-6","journal-title":"Knowl Inf Syst"},{"key":"1471_CR17","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1109\/ACCESS.2020.2983003","volume":"8","author":"AS Tarawneh","year":"2020","unstructured":"Tarawneh AS, Hassanat A, Almohammadi K (2020) SMOTEFUNA: synthetic minority over-sampling technique based on furthest neighbour algorithm. IEEE Access 8:1\u201315. https:\/\/doi.org\/10.1109\/ACCESS.2020.2983003","journal-title":"IEEE Access"},{"key":"1471_CR18","doi-asserted-by":"publisher","DOI":"10.1007\/s42519-019-0048-5","author":"H Wang","year":"2019","unstructured":"Wang H (2019) Divide-and-conquer information-based optimal subdata selection algorithm. J Stat Theory Pract. https:\/\/doi.org\/10.1007\/s42519-019-0048-5","journal-title":"J Stat Theory Pract"},{"issue":"1","key":"1471_CR19","doi-asserted-by":"publisher","first-page":"99","DOI":"10.1093\/biomet\/asaa043","volume":"108","author":"H Wang","year":"2021","unstructured":"Wang H, Ma Y (2021) Optimal subsampling for quantile regression in big data. Biometrika 108(1):99\u2013112. https:\/\/doi.org\/10.1093\/biomet\/asaa043","journal-title":"Biometrika"},{"issue":"522","key":"1471_CR20","doi-asserted-by":"publisher","first-page":"829","DOI":"10.1080\/01621459.2017.1292914","volume":"113","author":"H Wang","year":"2018","unstructured":"Wang H, Zhu R, Ma P (2018) Optimal subsampling for large sample logistic regression. J Am Stat Assoc 113(522):829\u2013844. https:\/\/doi.org\/10.1080\/01621459.2017.1292914","journal-title":"J Am Stat Assoc"},{"issue":"525","key":"1471_CR21","doi-asserted-by":"publisher","first-page":"393","DOI":"10.1080\/01621459.2017.1408468","volume":"114","author":"H Wang","year":"2019","unstructured":"Wang H, Yang M, Stufken J (2019) Information-based optimal subdata selection for big data linear regression. J Am Stat Assoc 114(525):393\u2013405. https:\/\/doi.org\/10.1080\/01621459.2017.1408468","journal-title":"J Am Stat Assoc"},{"key":"1471_CR22","doi-asserted-by":"publisher","DOI":"10.1016\/j.jbi.2020.103465","volume":"107","author":"Z Xu","year":"2020","unstructured":"Xu Z, Shen D, Nie T (2020) A hybrid sampling algorithm combining M-SMOTE and ENN based on random forest for medical imbalanced data. J Biomed Inform 107:103465. https:\/\/doi.org\/10.1016\/j.jbi.2020.103465","journal-title":"J Biomed Inform"},{"issue":"2","key":"1471_CR23","doi-asserted-by":"publisher","first-page":"731","DOI":"10.1007\/978-3-540-37256-1_89","volume":"344","author":"SJ Yen","year":"2006","unstructured":"Yen SJ, Lee YS (2006) Under-sampling approaches for improving prediction of the minority class in an imbalanced dataset. Lect Notes Control Inform Sci 344(2):731\u2013740. https:\/\/doi.org\/10.1007\/978-3-540-37256-1_89","journal-title":"Lect Notes Control Inform Sci"},{"issue":"3","key":"1471_CR24","doi-asserted-by":"publisher","first-page":"5718","DOI":"10.1016\/j.eswa.2008.06.108","volume":"36","author":"SJ Yen","year":"2009","unstructured":"Yen SJ, Lee YS (2009) Cluster-based under-sampling approaches for imbalanced data distributions. Expert Syst Appl 36(3):5718\u20135727. https:\/\/doi.org\/10.1016\/j.eswa.2008.06.108","journal-title":"Expert Syst Appl"},{"key":"1471_CR25","doi-asserted-by":"publisher","first-page":"1883","DOI":"10.1007\/s00362-022-01299-8","volume":"63","author":"J Yu","year":"2022","unstructured":"Yu J, Wang H (2022) Subdata selection algorithm for linear model discrimination. Stat Pap 63:1883\u20131906. https:\/\/doi.org\/10.1007\/s00362-022-01299-8","journal-title":"Stat Pap"},{"key":"1471_CR26","doi-asserted-by":"publisher","first-page":"2535","DOI":"10.1007\/s00180-021-01089-0","volume":"36","author":"L Zuo","year":"2021","unstructured":"Zuo L, Zhang H, Wang H et al (2021) Optimal subsample selection for massive logistic regression with distributed data. Comput Stat 36:2535\u20132562. https:\/\/doi.org\/10.1007\/s00180-021-01089-0","journal-title":"Comput Stat"}],"container-title":["Computational Statistics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00180-024-01471-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00180-024-01471-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00180-024-01471-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,2,10]],"date-time":"2025-02-10T06:49:10Z","timestamp":1739170150000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00180-024-01471-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,3,8]]},"references-count":26,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2025,1]]}},"alternative-id":["1471"],"URL":"https:\/\/doi.org\/10.1007\/s00180-024-01471-8","relation":{},"ISSN":["0943-4062","1613-9658"],"issn-type":[{"value":"0943-4062","type":"print"},{"value":"1613-9658","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,3,8]]},"assertion":[{"value":"10 April 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"29 January 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 March 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that there is no conflict of interest in the publication of this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}