{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,14]],"date-time":"2026-04-14T01:00:12Z","timestamp":1776128412027,"version":"3.50.1"},"reference-count":45,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2013,11,9]],"date-time":"2013-11-09T00:00:00Z","timestamp":1383955200000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by\/2.0"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["BMC Med Inform Decis Mak"],"published-print":{"date-parts":[[2013,12]]},"DOI":"10.1186\/1472-6947-13-124","type":"journal-article","created":{"date-parts":[[2013,11,9]],"date-time":"2013-11-09T18:01:04Z","timestamp":1384020064000},"source":"Crossref","is-referenced-by-count":38,"title":["An improved survivability prognosis of breast cancer by using sampling and feature selection technique to solve imbalanced patient classification data"],"prefix":"10.1186","volume":"13","author":[{"given":"Kung-Jeng","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bunjira","family":"Makond","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kung-Min","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2013,11,9]]},"reference":[{"key":"740_CR1","volume-title":"Quick cancer facts. Retrieved September 22","author":"World Health Organization","year":"2010","unstructured":"World Health Organization: Quick cancer facts. Retrieved September 22. 2010,\n                    http:\/\/www.who.int\/cancer\/en\/\n                    \n                  ,"},{"key":"740_CR2","doi-asserted-by":"publisher","first-page":"409","DOI":"10.3322\/caac.20134","volume":"61","author":"C DeSantis","year":"2011","unstructured":"DeSantis C, Siegel R, Bandi P, Jemal A: Breast Cancer Statistics, 2011. Cancer J Clin. 2011, 61: 409-418.","journal-title":"Cancer J Clin"},{"key":"740_CR3","volume-title":"Cancer trends progress report-2009\/2010 update. Retrieved June 22","author":"National Cancer Institute","year":"2009","unstructured":"National Cancer Institute: Cancer trends progress report-2009\/2010 update. Retrieved June 22. 2009,\n                    http:\/\/progressreport.cancer.gov\/highlights.asp\n                    \n                  ,"},{"key":"740_CR4","doi-asserted-by":"publisher","first-page":"281","DOI":"10.1159\/000012061","volume":"57","author":"M Lundin","year":"1999","unstructured":"Lundin M, Lundin J, Burke HB, Toikkanen S, Pylkk\u00e4nen L, Joensuu H: Artificial neural networks applied to survival prediction in breast cancer. Oncology. 1999, 57: 281-286. 10.1159\/000012061.","journal-title":"Oncology"},{"key":"740_CR5","first-page":"619","volume-title":"Proceedings of the seventh international conference IEEE","author":"D Soria","year":"2008","unstructured":"Soria D, Garibaldi JM, Biganzoli E, Ellis IO: A comparison of three different methods for classification of breast cancer data. Proceedings of the seventh international conference IEEE. 2008, San Diego: IEEE, 619-624."},{"key":"740_CR6","first-page":"5148","volume-title":"Proceedings of 30th Annual International IEEE EMBS Conference","author":"MU Khan","year":"2008","unstructured":"Khan MU, Choi JP, Shin H, Kim M: Predicting breast cancer survivability using fuzzy decision trees for personalized healthcare. Proceedings of 30th Annual International IEEE EMBS Conference. 2008, Vancouver: IEEE, 5148-5151."},{"key":"740_CR7","first-page":"1","volume":"9","author":"WP Chang","year":"2008","unstructured":"Chang WP, Liou DM: Comparison of three data mining techniques with genetic algorithm in the analysis of breast cancer data. J Telemed Telecare. 2008, 9: 1-26.","journal-title":"J Telemed Telecare"},{"key":"740_CR8","doi-asserted-by":"publisher","first-page":"113","DOI":"10.1016\/j.artmed.2004.07.002","volume":"34","author":"D Delen","year":"2005","unstructured":"Delen D, Walker G, Kadam A: Predicting breast cancer survivability: a comparison of three data mining methods. Artif Intell Med. 2005, 34: 113-127. 10.1016\/j.artmed.2004.07.002.","journal-title":"Artif Intell Med"},{"key":"740_CR9","first-page":"10","volume":"58","author":"A Bellaachia","year":"2006","unstructured":"Bellaachia A, Guven E: Predicting breast cancer survivability using data mining techniques. Age. 2006, 58: 10-110.","journal-title":"Age"},{"key":"740_CR10","first-page":"11","volume":"13","author":"A Endo","year":"2008","unstructured":"Endo A, Shibata T, Tanaka H: Comparison of seven algorithms to predict breast cancer survival. Int J Biomed Soft Comput Hum Sci. 2008, 13: 11-16.","journal-title":"Int J Biomed Soft Comput Hum Sci"},{"key":"740_CR11","first-page":"1","volume-title":"Proceedings of International Conference on Bioinformatics and Biomedical Engineering","author":"Y Liu","year":"2009","unstructured":"Liu Y, Cheng W, Lu Z: Decision tree based predictive models for breast cancer survivability on imbalance data. Proceedings of International Conference on Bioinformatics and Biomedical Engineering. 2009, Beijing: IEEE, 1-4."},{"key":"740_CR12","first-page":"107","volume-title":"Proceedings of the 7th European conference on principles and practice of knowledge discovery in database","author":"NV Chawla","year":"2003","unstructured":"Chawla NV, Lazarevic A, Hall LO, Bowyer KW: SMOTEBoost: Improving prediction of the minority class in boosting. Proceedings of the 7th European conference on principles and practice of knowledge discovery in database. 2003, Berlin: Springer, 107-119."},{"issue":"9","key":"740_CR13","doi-asserted-by":"publisher","first-page":"1263","DOI":"10.1109\/TKDE.2008.239","volume":"21","author":"H He","year":"2009","unstructured":"He H, Garcia E: Learning from imbalanced data. IEEE Trans Knowl Data Eng. 2009, 21 (9): 1263-1284.","journal-title":"IEEE Trans Knowl Data Eng"},{"key":"740_CR14","doi-asserted-by":"publisher","first-page":"287","DOI":"10.1007\/978-3-642-04843-2_31","volume":"5821","author":"Q Gu","year":"2009","unstructured":"Gu Q, Cai Z, Ziu L: Classification of imbalanced data sets by using the hybrid re-sampling algorithm based on isomap. In LNCS, Adv Comput Intelligence. 2009, 5821: 287-296. 10.1007\/978-3-642-04843-2_31.","journal-title":"In LNCS, Adv Comput Intelligence"},{"key":"740_CR15","first-page":"104","volume-title":"Proceeding of the IEEE symposium on computational intelligence and data mining","author":"T Maciejewski","year":"2011","unstructured":"Maciejewski T, Stefanowski J: Local neighbourhood extension of SMOTE for mining imbalanced data. Proceeding of the IEEE symposium on computational intelligence and data mining. 2011, Paris: IEEE, 104-111."},{"key":"740_CR16","doi-asserted-by":"publisher","first-page":"51","DOI":"10.1186\/1472-6947-11-51","volume":"11","author":"M Khalilia","year":"2011","unstructured":"Khalilia M, Chakraborty S, Popescu M: Predicting disease risks from highly imbalanced data using random forest. BMC Med Inform Decis Mak. 2011, 11: 51-10.1186\/1472-6947-11-51.","journal-title":"BMC Med Inform Decis Mak"},{"key":"740_CR17","doi-asserted-by":"publisher","first-page":"30","DOI":"10.1186\/1472-6947-13-30","volume":"13","author":"Z Afzal","year":"2013","unstructured":"Afzal Z, Schuemie MJ, van Blijderveen JC, Sen EF, Sturkenboom MCJM, Kors JA: Improving sensitivity of machine learning methods for automated case identification from free-text electronic medical records. BMC Med Inform Decis Mak. 2013, 13: 30-10.1186\/1472-6947-13-30.","journal-title":"BMC Med Inform Decis Mak"},{"key":"740_CR18","first-page":"179","volume-title":"Proceedings of the Fourteenth International Conference on Machine Learning","author":"M Kubat","year":"1997","unstructured":"Kubat M, Matwin S: Addressing the course of imbalanced training-sets: one-sided selection. Proceedings of the Fourteenth International Conference on Machine Learning. 1997, San Francisco: Morgan Kaufmann, 179-186."},{"key":"740_CR19","doi-asserted-by":"publisher","first-page":"853","DOI":"10.1007\/0-387-25465-X_40","volume-title":"An Overview. In Data Mining and Knowledge Discovery Handbook","author":"NV Chawla","year":"2005","unstructured":"Chawla NV: Data Mining for Imbalanced Datasets. An Overview. In Data Mining and Knowledge Discovery Handbook. 2005, USA: Springer, 853-867."},{"key":"740_CR20","unstructured":"Lopez V, Fern\u00e1ndez A, Garc\u00eda S, Palade V, Herrera F: An insight into classification with imbalanced data: Empirical results and current trends on using data intrinsic characteristics. Inform Sci. \u2009-in press"},{"key":"740_CR21","first-page":"1","volume-title":"Proceeding of Workshop on Learning from Imbalanced Datasets II, ICML","author":"C Drummond","year":"2003","unstructured":"Drummond C, Holte RC: C4.5, class imbalance, and cost sensitivity: Why under-sampling beats over-sampling. Proceeding of Workshop on Learning from Imbalanced Datasets II, ICML. 2003, 1-8."},{"key":"740_CR22","doi-asserted-by":"crossref","first-page":"321","DOI":"10.1613\/jair.953","volume":"16","author":"NV Chawla","year":"2002","unstructured":"Chawla NV, Bowyer KW, Hall LO, Kegelmeyer WP: SMOTE: Synthetic minority over-sampling technique. J Artif Intell Res. 2002, 16: 321-357.","journal-title":"J Artif Intell Res"},{"issue":"4","key":"740_CR23","doi-asserted-by":"publisher","first-page":"1125","DOI":"10.1002\/prot.21870","volume":"70","author":"XM Zhao","year":"2007","unstructured":"Zhao XM, Li X, Chen L, Aihara K: Protein classification with imbalanced data. Proteins. 2007, 70 (4): 1125-1132. 10.1002\/prot.21870.","journal-title":"Proteins"},{"key":"740_CR24","first-page":"69","volume-title":"Proceedings of the annual meeting of the North American fuzzy information processing society","author":"L Pelayo","year":"2007","unstructured":"Pelayo L, Dick S: Applying novel resampling strategies to software defect prediction. Proceedings of the annual meeting of the North American fuzzy information processing society. 2007, San Diego: IEEE, 69-72."},{"key":"740_CR25","doi-asserted-by":"publisher","first-page":"196","DOI":"10.1109\/ESEM.2007.28","volume-title":"Proceedings of First International Symposium on Empirical Software Engineering and Measurement","author":"Y Kamei","year":"2007","unstructured":"Kamei Y, Monden A, Matsumoto S, Kakimoto T, Matsumoto K: The effects of over and under sampling on fault-prone module detection. Proceedings of First International Symposium on Empirical Software Engineering and Measurement. 2007, Madrid: IEEE, 196-204."},{"key":"740_CR26","volume-title":"Encyclopedia of Machine Learning","author":"CX Ling","year":"2008","unstructured":"Ling CX, Sheng VS: Cost-Sensitive Learning and the Class Imbalance Problem. Encyclopedia of Machine Learning. Edited by: Sammut C. 2008, New York: Springer"},{"key":"740_CR27","unstructured":"Surveillance, Epidemiology, and End Results (SEER) Program, Research Data (1973\u20132007), National Cancer Institute, DCCPS, Surveillance Research Program, Cancer Statistics Branch, released. 2010,\n                    http:\/\/www.seer.cancer.gov\n                    \n                  , April , based on the November 2009 submission,"},{"key":"740_CR28","first-page":"29","volume":"20","author":"A Agrawal","year":"2012","unstructured":"Agrawal A, Misra S, Narayanan R, Polepeddi L, Choudhary A: Lung cancer survival prediction using ensemble data mining on SEER data. Sci Program. 2012, 20: 29-42.","journal-title":"Sci Program"},{"key":"740_CR29","volume-title":"Data mining: Concepts and techniques","author":"J Han","year":"2006","unstructured":"Han J, Kamber M: Data mining: Concepts and techniques. 2006, San Francisco: Morgan Kaufmann, Elsevier Science"},{"key":"740_CR30","first-page":"181","volume-title":"Proceedings of Australasian Computer Science Conference","author":"MA Hall","year":"1998","unstructured":"Hall MA, Smith LA: Practical feature subset selection for machine learning. Proceedings of Australasian Computer Science Conference. 1998, Berlin: Springer, 181-191."},{"key":"740_CR31","volume-title":"Correlation-based feature selection for machine learning. PhD Thesis","author":"MA Hall","year":"1999","unstructured":"Hall MA: Correlation-based feature selection for machine learning. PhD Thesis. 1999, New Zealand: Department of Computer Science, Waikato University"},{"key":"740_CR32","first-page":"1157","volume":"3","author":"I Guyon","year":"2003","unstructured":"Guyon I, Elisseeff A: An introduction to variable and feature selection. J Mach Learn Res. 2003, 3: 1157-1182.","journal-title":"J Mach Learn Res"},{"key":"740_CR33","volume-title":"Proceeding of Pacific-Asia Conference Knowledge Discovery and Data Mining","author":"A Lazarevic","year":"2004","unstructured":"Lazarevic A, Srivastava J, Kumar V: Tutorial: Data mining for analysis of rare events: a case study in security, financial and medical applications. Proceeding of Pacific-Asia Conference Knowledge Discovery and Data Mining. 2004"},{"key":"740_CR34","volume-title":"Data mining: practical machine learning tools and techniques","author":"IH Witten","year":"2005","unstructured":"Witten IH, Frank E: Data mining: practical machine learning tools and techniques. 2005, San Francisco, CA: Morgan Kaufmann"},{"key":"740_CR35","first-page":"724","volume-title":"Proceedings of the 18th European Conference on Machine Learning","author":"VS Sheng","year":"2007","unstructured":"Sheng VS, Ling CX: Roulette sampling for cost-sensitive learning. Proceedings of the 18th European Conference on Machine Learning. 2007, Berlin: Springer, 724-731."},{"key":"740_CR36","doi-asserted-by":"publisher","first-page":"10","DOI":"10.1145\/1656274.1656278","volume":"11","author":"M Hall","year":"2009","unstructured":"Hall M, Frank E, Holmes G, Pfahringer B, Reutemann P, Witten IH: The WEKA Data Mining Software: An Update. ACM SIGKDD Explorations Newsletter. 2009, 11: 10-18. 10.1145\/1656274.1656278.","journal-title":"ACM SIGKDD Explorations Newsletter"},{"key":"740_CR37","doi-asserted-by":"publisher","first-page":"12","DOI":"10.1186\/1758-2946-1-21","volume":"1","author":"AC Schierz","year":"2009","unstructured":"Schierz AC: Virtual screening of bioassay data. J Cheminformatics. 2009, 1: 12-10.1186\/1758-2946-1-12.","journal-title":"J Cheminformatics"},{"key":"740_CR38","doi-asserted-by":"publisher","DOI":"10.1002\/0471722146","volume-title":"Applied logistic regression (2nd Ed.)","author":"DW Hosmer","year":"2000","unstructured":"Hosmer DW, Lemeshow S: Applied logistic regression (2nd Ed.). 2000, New York, USA: A Wiley-Interscience Publication, John Wiley & Sons Inc."},{"key":"740_CR39","doi-asserted-by":"publisher","first-page":"1431","DOI":"10.1002\/sim.680","volume":"20","author":"L Barker","year":"2001","unstructured":"Barker L, Brown C: Logistic regression when binary predictor variables are highly correlated. Stat Med. 2001, 20: 1431-1442. 10.1002\/sim.680.","journal-title":"Stat Med"},{"key":"740_CR40","first-page":"229","volume-title":"Using Decision Trees for the Semi-automatic Development of Medical Data Patterns: A Computer-Supported Framework","author":"A Fountoulaki","year":"2010","unstructured":"Fountoulaki A, Karacapilidis M, Manatakis N: Using Decision Trees for the Semi-automatic Development of Medical Data Patterns: A Computer-Supported Framework. 2010, Biomedicine: Web-Based Applications in Healthcare and, 229-242."},{"key":"740_CR41","volume-title":"Department of Computer Science, Iowa State University","author":"Y Chen","year":"2009","unstructured":"Chen Y: Learning classifiers from imbalanced, only positive and unlabeled data set. Department of Computer Science, Iowa State University. 2009"},{"key":"740_CR42","doi-asserted-by":"publisher","first-page":"6585","DOI":"10.1016\/j.eswa.2011.12.043","volume":"39","author":"V Lopez","year":"2012","unstructured":"Lopez V, Fern\u00e1ndez A, Moreno-Torres JG, Herrera F: Analysis of preprocessing vs. cost-sensitive learning for imbalanced classification. Open problems on intrinsic data characteristics. Expert Syst Appl. 2012, 39: 6585-6608. 10.1016\/j.eswa.2011.12.043.","journal-title":"Expert Syst Appl"},{"key":"740_CR43","doi-asserted-by":"publisher","first-page":"69","DOI":"10.1145\/1089827.1089836","volume-title":"Proceedings of the 1st international workshop on Utility-based data mining - UBDM \u201905","author":"K McCarthy","year":"2005","unstructured":"McCarthy K, Zabar B, Weiss G: Does cost-sensitive learning beat sampling for classifying rare classes?. Proceedings of the 1st international workshop on Utility-based data mining - UBDM \u201905. 2005, USA: ACM Press, 69-77."},{"key":"740_CR44","first-page":"116","volume":"8","author":"S Palaniappan","year":"2008","unstructured":"Palaniappan S, Hong TK: Discretization of continuous valued dimensions in OLAP data cubes. Int J Comput Sci Network Secur. 2008, 8: 116-126.","journal-title":"Int J Comput Sci Network Secur"},{"key":"740_CR45","first-page":"80","volume-title":"Proceeding of the Business Intelligence and Data Mining Conference","author":"A Ali","year":"2010","unstructured":"Ali A, An Y, Kim D, Park K, Shin H, Kim M: Prediction of breast cancer survivability: to alleviate oncologists in decision making. Proceeding of the Business Intelligence and Data Mining Conference. 2010, Seoul, Korea: Seoul, Korea, 80-92."}],"container-title":["BMC Medical Informatics and Decision Making"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/article\/10.1186\/1472-6947-13-124\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1186\/1472-6947-13-124.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1186\/1472-6947-13-124","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1186\/1472-6947-13-124.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,1,22]],"date-time":"2019-01-22T16:02:44Z","timestamp":1548172964000},"score":1,"resource":{"primary":{"URL":"https:\/\/bmcmedinformdecismak.biomedcentral.com\/articles\/10.1186\/1472-6947-13-124"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2013,11,9]]},"references-count":45,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2013,12]]}},"alternative-id":["740"],"URL":"https:\/\/doi.org\/10.1186\/1472-6947-13-124","relation":{},"ISSN":["1472-6947"],"issn-type":[{"value":"1472-6947","type":"electronic"}],"subject":[],"published":{"date-parts":[[2013,11,9]]},"article-number":"124"}}