{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,24]],"date-time":"2026-04-24T01:23:19Z","timestamp":1776993799306,"version":"3.51.4"},"reference-count":34,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2020,11,13]],"date-time":"2020-11-13T00:00:00Z","timestamp":1605225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2020,11,13]],"date-time":"2020-11-13T00:00:00Z","timestamp":1605225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"funder":[{"name":"National Key Research and Development Program of China","award":["2016YFB0200902"],"award-info":[{"award-number":["2016YFB0200902"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"published-print":{"date-parts":[[2021,6]]},"DOI":"10.1007\/s11227-020-03506-5","type":"journal-article","created":{"date-parts":[[2020,11,13]],"date-time":"2020-11-13T11:03:22Z","timestamp":1605265402000},"page":"5960-5983","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":26,"title":["OKCM: improving parallel task scheduling in high-performance computing systems using online learning"],"prefix":"10.1007","volume":"77","author":[{"given":"Jingbo","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1434-7016","authenticated-orcid":false,"given":"Xingjun","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Li","family":"Han","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zeyu","family":"Ji","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaoshe","family":"Dong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chenglong","family":"Hu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2020,11,13]]},"reference":[{"issue":"10","key":"3506_CR1","doi-asserted-by":"publisher","first-page":"2304","DOI":"10.1109\/TPDS.2018.2820699","volume":"29","author":"\u00c9 Gaussier","year":"2018","unstructured":"Gaussier \u00c9, Lelong J, Reis V, Trystram D (2018) Online tuning of easy-backfilling using queue reordering policies. IEEE Trans Parallel Distrib Syst 29(10):2304\u20132316","journal-title":"IEEE Trans Parallel Distrib Syst"},{"key":"3506_CR2","doi-asserted-by":"publisher","first-page":"985","DOI":"10.1016\/j.future.2017.03.024","volume":"105","author":"J Yang","year":"2020","unstructured":"Yang J, Jiang B, Lv Z, Choo KR (2020) A task scheduling algorithm considering game theory designed for energy management in cloud computing. Future Gener Comput Syst 105:985\u2013992","journal-title":"Future Gener Comput Syst"},{"key":"3506_CR3","doi-asserted-by":"crossref","unstructured":"Liang S, Yang Z, Jin F, Chen Y (2020) Data centers job scheduling with deep reinforcement learning. In: Lauw HW, Wong RC, Ntoulas A, Lim E, Ng S, Pan SJ (eds) Advances in Knowledge Discovery and Data Mining - 24th Pacific-Asia Conference, PAKDD 2020, Singapore, May 11-14, 2020, Proceedings, Part II, Springer, Lecture Notes in Computer Science, vol 12085, pp 906\u2013917","DOI":"10.1007\/978-3-030-47436-2_68"},{"key":"3506_CR4","doi-asserted-by":"crossref","unstructured":"Fan Y, Rich P, Allcock WE, Papka ME, Lan Z (2017) Trade-off between prediction accuracy and underestimation rate in job runtime estimates. In: 2017 IEEE International Conference on Cluster Computing, CLUSTER 2017, Honolulu, HI, USA, September 5-8, 2017, IEEE Computer Society, pp 530\u2013540","DOI":"10.1109\/CLUSTER.2017.11"},{"key":"3506_CR5","doi-asserted-by":"crossref","unstructured":"Carastan-Santos D, de\u00a0Camargo RY (2017) Obtaining dynamic scheduling policies with simulation and machine learning. In: Mohr B, Raghavan P (eds) Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis, SC 2017, Denver, CO, USA, November 12 - 17, 2017, ACM, pp 32:1\u201332:13","DOI":"10.1145\/3126908.3126955"},{"key":"3506_CR6","doi-asserted-by":"crossref","unstructured":"Tang W, Lan Z, Desai N, Buettner D (2009) Fault-aware, utility-based job scheduling on blue, gene\/p systems. In: Proceedings of the 2009 IEEE International Conference on Cluster Computing, August 31 - September 4, 2009, New Orleans, Louisiana, USA, IEEE Computer Society, pp 1\u201310","DOI":"10.1109\/CLUSTR.2009.5289206"},{"key":"3506_CR7","doi-asserted-by":"crossref","unstructured":"Zhang D, Dai D, He Y, Bao FS (2019) Rlscheduler: Learn to schedule HPC batch jobs using deep reinforcement learning. arXiv:1910.08925","DOI":"10.1109\/SC41405.2020.00035"},{"key":"3506_CR8","doi-asserted-by":"crossref","unstructured":"Gaussier \u00c9, Glesser D, Reis V, Trystram D (2015) Improving backfilling by using machine learning to predict running times. In: Kern J, Vetter JS (eds) Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis, SC 2015, Austin, TX, USA, November 15-20, 2015, ACM, pp 64:1\u201364:10","DOI":"10.1145\/2807591.2807646"},{"issue":"6","key":"3506_CR9","doi-asserted-by":"publisher","first-page":"789","DOI":"10.1109\/TPDS.2007.70606","volume":"18","author":"D Tsafrir","year":"2007","unstructured":"Tsafrir D, Etsion Y, Feitelson DG (2007) Backfilling using system-generated predictions rather than user runtime estimates. IEEE Trans Parallel Distrib Syst 18(6):789\u2013803","journal-title":"IEEE Trans Parallel Distrib Syst"},{"key":"3506_CR10","doi-asserted-by":"crossref","unstructured":"Obaida MA, Liu J (2017) Simulation of HPC job scheduling and large-scale parallel workloads. In: 2017 Winter Simulation Conference, WSC 2017, Las Vegas, NV, USA, December 3-6, 2017, IEEE, pp 920\u2013931","DOI":"10.1109\/WSC.2017.8247843"},{"key":"3506_CR11","doi-asserted-by":"crossref","unstructured":"Tsafrir D, Etsion Y, Feitelson DG (2005) Modeling user runtime estimates. In: Feitelson DG, Frachtenberg E, Rudolph L, Schwiegelshohn U (eds) Job Scheduling Strategies for Parallel Processing, 11th International Workshop, JSSPP 2005, Cambridge, MA, USA, June 19, 2005, Revised Selected Papers, Springer, Lecture Notes in Computer Science, vol 3834, pp 1\u201335","DOI":"10.1007\/11605300_1"},{"issue":"11","key":"3506_CR12","doi-asserted-by":"publisher","first-page":"4635","DOI":"10.1007\/s11227-017-2038-2","volume":"73","author":"J Park","year":"2017","unstructured":"Park J, Kim E (2017) Runtime prediction of parallel applications with workload-aware clustering. J Supercomput 73(11):4635\u20134651","journal-title":"J Supercomput"},{"issue":"10","key":"3506_CR13","doi-asserted-by":"publisher","first-page":"2967","DOI":"10.1016\/j.jpdc.2014.06.013","volume":"74","author":"DG Feitelson","year":"2014","unstructured":"Feitelson DG, Tsafrir D, Krakov D (2014) Experience with using the parallel workloads archive. J Parallel Distributed Comput 74(10):2967\u20132982","journal-title":"J Parallel Distributed Comput"},{"issue":"15","key":"3506_CR14","doi-asserted-by":"publisher","first-page":"11285","DOI":"10.1007\/s00521-019-04625-8","volume":"32","author":"S Balasundaram","year":"2020","unstructured":"Balasundaram S, Prasad SC (2020) Robust twin support vector regression based on huber loss function. Neural Comput Appl 32(15):11285\u201311309","journal-title":"Neural Comput Appl"},{"key":"3506_CR15","doi-asserted-by":"crossref","unstructured":"Thonglek K, Ichikawa K, Takahashi K, Iida H, Nakasan C (2019) Improving resource utilization in data centers using an lstm-based prediction model. In: 2019 IEEE International Conference on Cluster Computing, CLUSTER 2019, Albuquerque, NM, USA, September 23-26, 2019, IEEE, pp 1\u20138","DOI":"10.1109\/CLUSTER.2019.8891022"},{"key":"3506_CR16","doi-asserted-by":"publisher","first-page":"307","DOI":"10.1016\/j.future.2019.08.012","volume":"102","author":"G Ismayilov","year":"2020","unstructured":"Ismayilov G, Topcuoglu HR (2020) Neural network based multi-objective evolutionary algorithm for dynamic workflow scheduling in cloud computing. Future Gener Comput Syst 102:307\u2013322","journal-title":"Future Gener Comput Syst"},{"key":"3506_CR17","doi-asserted-by":"crossref","unstructured":"Esmaili A, Pedram M (2020) Energy-aware scheduling of jobs in heterogeneous cluster systems using deep reinforcement learning. In: 21st International Symposium on Quality Electronic Design, ISQED 2020, Santa Clara, CA, USA, March 25-26, 2020, IEEE, pp 426\u2013431","DOI":"10.1109\/ISQED48828.2020.9137025"},{"key":"3506_CR18","doi-asserted-by":"crossref","unstructured":"Zhang L, Qi Q, Wang J, Sun H, Liao J (2019) Multi-task deep reinforcement learning for scalable parallel task scheduling. In: 2019 IEEE International Conference on Big Data (Big Data), Los Angeles, CA, USA, December 9-12, 2019, IEEE, pp 2992\u20133001","DOI":"10.1109\/BigData47090.2019.9006027"},{"key":"3506_CR19","doi-asserted-by":"crossref","unstructured":"Chen L, Wu I, Chang Y (2019) Reinforcement learning based fragment-aware scheduling for high utilization HPC platforms. In: 2019 International Conference on Technologies and Applications of Artificial Intelligence, TAAI 2019, Kaohsiung, Taiwan, November 21-23, 2019, IEEE, pp 1\u20137","DOI":"10.1109\/TAAI48200.2019.8959932"},{"key":"3506_CR20","doi-asserted-by":"crossref","unstructured":"Klus\u00e1cek D, Chlumsk\u00fd V (2018) Evaluating the impact of soft walltimes on job scheduling performance. In: Klus\u00e1cek D, Cirne W, Desai N (eds) Job Scheduling Strategies for Parallel Processing - 22nd International Workshop, JSSPP 2018, Vancouver, BC, Canada, May 25, 2018, Revised Selected Papers, Springer, Lecture Notes in Computer Science, vol 11332, pp 15\u201338","DOI":"10.1007\/978-3-030-10632-4_2"},{"key":"3506_CR21","doi-asserted-by":"crossref","unstructured":"Skovira J, Chan W, Zhou H, Lifka DA (1996) The EASY - loadleveler API project. In: Feitelson DG, Rudolph L (eds) Job Scheduling Strategies for Parallel Processing, IPPS\u201996 Workshop, Honolulu, Haiwai, USA, April 16, 1996, Proceedings, Springer, Lecture Notes in Computer Science, vol 1162, pp 41\u201347","DOI":"10.1007\/BFb0022286"},{"key":"3506_CR22","doi-asserted-by":"crossref","unstructured":"Yoo AB, Jette MA, Grondona M (2003) SLURM: simple linux utility for resource management. In: Feitelson DG, Rudolph L, Schwiegelshohn U (eds) Job Scheduling Strategies for Parallel Processing, 9th International Workshop, JSSPP 2003, Seattle, WA, USA, June 24, 2003, Revised Papers, Springer, Lecture Notes in Computer Science, vol 2862, pp 44\u201360","DOI":"10.1007\/10968987_3"},{"key":"3506_CR23","doi-asserted-by":"crossref","unstructured":"Talby D, Feitelson DG (1999) Supporting priorities and improving utilization of the IBM SP scheduler using slack-based backfilling. In: 13th International Parallel Processing Symposium \/ 10th Symposium on Parallel and Distributed Processing (IPPS \/ SPDP \u201999), 12-16 April 1999, San Juan, Puerto Rico, Proceedings, IEEE Computer Society, pp 513\u2013517","DOI":"10.1109\/IPPS.1999.760525"},{"issue":"6","key":"3506_CR24","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1109\/71.932708","volume":"12","author":"AM Weil","year":"2001","unstructured":"Weil AM, Feitelson DG (2001) Utilization, predictability, workloads, and user runtime estimates in scheduling the IBM SP2 with backfilling. IEEE Trans Parallel Distrib Syst 12(6):529\u2013543","journal-title":"IEEE Trans Parallel Distrib Syst"},{"key":"#cr-split#-3506_CR25.1","doi-asserted-by":"crossref","unstructured":"Perkovic D, Keleher PJ, (2000) Randomization, speculation, and adaptation in batch schedulers. In: Donnelley J","DOI":"10.1109\/SC.2000.10041"},{"key":"#cr-split#-3506_CR25.2","unstructured":"(ed) Proceedings Supercomputing 2000(November), pp. 4-10, (2000) Dallas. IEEE Computer Society, CD-ROM, IEEE Computer Society, Texas, USA, p 7"},{"key":"3506_CR26","doi-asserted-by":"crossref","unstructured":"McKenna R, Herbein S, Moody A, Gamblin T, Taufer M (2016) Machine learning predictions of runtime and IO traffic on high-end clusters. In: 2016 IEEE International Conference on Cluster Computing, CLUSTER 2016, Taipei, Taiwan, September 12-16, 2016, IEEE Computer Society, pp 255\u2013258","DOI":"10.1109\/CLUSTER.2016.58"},{"key":"3506_CR27","unstructured":"Ross S, Mineiro P, Langford J (2013) Normalized online learning. In: Nicholson A, Smyth P (eds) Proceedings of the Twenty-Ninth Conference on Uncertainty in Artificial Intelligence, UAI 2013, Bellevue, WA, USA, August 11-15, 2013, AUAI Press"},{"issue":"1","key":"3506_CR28","doi-asserted-by":"publisher","first-page":"72","DOI":"10.3390\/app10010072","volume":"10","author":"J Li","year":"2020","unstructured":"Li J, Zhang X, Zhou J, Dong X, Zhang C (2020) swHPFM: Refactoring and optimizing the structured grid fluid mechanical algorithm on the sunway taihulight supercomputer. Appl Sci 10(1):72","journal-title":"Appl Sci"},{"key":"3506_CR29","doi-asserted-by":"crossref","unstructured":"Duan X, Gao P, Zhang M, Zhang T, Meng H, Li Y, Schmidt B, Fu H, Gan L, Xue W, Yang G, Liu W (2020) Neighbor-list-free molecular dynamics on sunway taihulight supercomputer. In: Gupta R, Shen X (eds) PPoPP \u201920: 25th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, San Diego, California, USA, February 22-26, 2020, ACM, pp 413\u2013414","DOI":"10.1145\/3332466.3374532"},{"key":"3506_CR30","doi-asserted-by":"crossref","unstructured":"Niu S, Zhai J, Ma X, Liu M, Zhai Y, Chen W, Zheng W (2012) Employing checkpoint to improve job scheduling in large-scale systems. In: Cirne W, Desai N, Frachtenberg E, Schwiegelshohn U (eds) Job Scheduling Strategies for Parallel Processing, 16th International Workshop, JSSPP 2012, Shanghai, China, May 25, 2012. Revised Selected Papers, Springer, Lecture Notes in Computer Science, vol 7698, pp 36\u201355","DOI":"10.1007\/978-3-642-35867-8_3"},{"issue":"10","key":"3506_CR31","doi-asserted-by":"publisher","first-page":"2899","DOI":"10.1016\/j.jpdc.2014.06.008","volume":"74","author":"H Casanova","year":"2014","unstructured":"Casanova H, Giersch A, Legrand A, Quinson M, Suter F (2014) Versatile, scalable, and accurate simulation of distributed applications and platforms. J Parallel Distrib Comput 74(10):2899\u20132917","journal-title":"J Parallel Distrib Comput"},{"issue":"4","key":"3506_CR32","doi-asserted-by":"publisher","first-page":"3043","DOI":"10.1007\/s11227-019-03091-2","volume":"76","author":"CT Do","year":"2020","unstructured":"Do CT, Choi HJ, Chung SW, Kim CH (2020) A novel warp scheduling scheme considering long-latency operations for high-performance gpus. J Supercomput 76(4):3043\u20133062","journal-title":"J Supercomput"},{"key":"3506_CR33","doi-asserted-by":"publisher","first-page":"99","DOI":"10.1016\/j.jpdc.2020.03.009","volume":"141","author":"M Wei","year":"2020","unstructured":"Wei M, Zhao W, Chen Q, Dai H, Leng J, Li C, Zheng W, Guo M (2020) Predicting and reining in application-level slowdown on spatial multitasking gpus. J Parallel Distrib Comput 141:99\u2013114","journal-title":"J Parallel Distrib Comput"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-020-03506-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-020-03506-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-020-03506-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,5,11]],"date-time":"2021-05-11T12:12:29Z","timestamp":1620735149000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-020-03506-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,11,13]]},"references-count":34,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2021,6]]}},"alternative-id":["3506"],"URL":"https:\/\/doi.org\/10.1007\/s11227-020-03506-5","relation":{},"ISSN":["0920-8542","1573-0484"],"issn-type":[{"value":"0920-8542","type":"print"},{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2020,11,13]]},"assertion":[{"value":"29 October 2020","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 November 2020","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Compliance with ethical standards"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}