{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,5,21]],"date-time":"2025-05-21T06:12:10Z","timestamp":1747807930844,"version":"3.38.0"},"publisher-location":"Berlin, Heidelberg","reference-count":44,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"type":"print","value":"9783540253303"},{"type":"electronic","value":"9783540317951"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2005]]},"DOI":"10.1007\/11407522_13","type":"book-chapter","created":{"date-parts":[[2010,9,23]],"date-time":"2010-09-23T20:24:37Z","timestamp":1285273477000},"page":"233-252","source":"Crossref","is-referenced-by-count":39,"title":["Performance Implications of Failures in Large-Scale Cluster Scheduling"],"prefix":"10.1007","author":[{"given":"Yanyong","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mark S.","family":"Squillante","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Anand","family":"Sivasubramaniam","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ramendra K.","family":"Sahoo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"issue":"2-3","key":"13_CR1","doi-asserted-by":"publisher","first-page":"85","DOI":"10.1016\/S0166-218X(00)00266-3","volume":"110","author":"S. Albers","year":"2001","unstructured":"Albers, S., Schmidt, G.: Scheduling with unexpected machine breakdowns. Discrete Applied Mathematics\u00a0110(2-3), 85\u201399 (2001)","journal-title":"Discrete Applied Mathematics"},{"key":"13_CR2","unstructured":"S.M., Andrews, D.: On the reliability of the ibm mvs\/xa operating system. IEEE Trans. Software Engineering (October 1987)"},{"key":"13_CR3","unstructured":"Arlitt, M., Jin, T.: Workload Characterization of the 1998 World Cup E-Commerce Site. Technical Report Technical Report HPL-1999-62, HP (May 1999)"},{"key":"13_CR4","doi-asserted-by":"publisher","first-page":"881","DOI":"10.1007\/s002360050110","volume":"34","author":"J.L. Bruno","year":"1997","unstructured":"Bruno, J.L., Coffman, E.G.: Optimal Fault-Tolerant Computing onMultiprocess Systems. Acta Informatica\u00a034, 881\u2013904 (1997)","journal-title":"Acta Informatica"},{"key":"13_CR5","doi-asserted-by":"crossref","unstructured":"Buckley, M.F., Siewiorek, D.P.: Vax\/vms event monitoring and analysis. In: FTCS-25, Computing Digest of Papers, June 1995, pp. 414\u2013423 (1995)","DOI":"10.1109\/FTCS.1995.466958"},{"key":"13_CR6","doi-asserted-by":"crossref","unstructured":"Buckley, M.F., Siewiorek, D.P.: Comparative analysis of event tupling schemes. In: FTCS-26, Computing Digest of Papers, June 1996, pp. 294\u2013303 (1996)","DOI":"10.1109\/FTCS.1996.534614"},{"key":"13_CR7","unstructured":"Castillo, X., Siewiorek, D.P.: A workload dependent software reliability prediction model. In: Proc. 12th. Intl. Symp. Fault-Tolerant Computing, June 1982, pp. 279\u2013286 (1982)"},{"key":"13_CR8","unstructured":"Feitelson, D.: A survey of scheduling in multiprogrammed parallel systems. IBM Research Technical Report, RC 19790 (1994)"},{"key":"13_CR9","doi-asserted-by":"crossref","unstructured":"Flautner, K., Kim, N., Martin, S., Blaauw, D., Mudge, T.: Drowsy Caches: Simple Techniques for Reducing Leakage Power. In: Proceedings of the International Symposium on Computer Architecture (ISCA), pp. 148\u2013157 (2002)","DOI":"10.1145\/545214.545232"},{"key":"13_CR10","doi-asserted-by":"crossref","unstructured":"Franke, H., Jann, J., Moreira, J.E., Pattnaik, P.: An evaluation of parallel job scheduling for asci blue-pacific. In: Proc. of SC 1999. Portland OR, IBM Research Report RC 21559, IBM TJ Watson Research Center (November 1999)","DOI":"10.1145\/331532.331577"},{"key":"13_CR11","doi-asserted-by":"crossref","unstructured":"Franke, H., Jann, J., Moreira, J.E., Pattnaik, P., Jette, M.A.: Evaluation of Parallel Job Scheduling for ASCI Blue-Pacific. In: Proceedings of Supercomputing (November 1999)","DOI":"10.1145\/331532.331577"},{"key":"13_CR12","unstructured":"Gorda, B., Wolski, R.: Time sharing massively parallel machines. In: Proc. of ICPP 1995. Portland OR, pp. 214\u2013217 (August 1995)"},{"key":"13_CR13","doi-asserted-by":"crossref","unstructured":"Heath, T., Martin, R.P., Nguyen, T.D.: Improving cluster availability using workstation validation. In: Proceedings of the ACM SIGMETRICS 2002 Conference on Measurement and Modeling of Computer Systems, pp. 217\u2013227 (2002)","DOI":"10.1145\/511334.511362"},{"key":"13_CR14","unstructured":"Hsueh, M.C., Iyer, R.K., Trivedi, K.S.: A measurement-based performability model for a multiprocessor system. In: Computer Performance and Reliability, pp. 337\u2013352 (1987)"},{"key":"13_CR15","doi-asserted-by":"publisher","first-page":"1438","DOI":"10.1109\/TSE.1985.232180","volume":"SE-11","author":"R.K. Iyer","year":"1985","unstructured":"Iyer, R.K., Rossetti, D.J.: Effect of system workload on operating system reliability: A study on ibm 3081. IEEE Trans. Software Engineering\u00a0SE-11, 1438\u20131448 (1985)","journal-title":"IEEE Trans. Software Engineering"},{"key":"13_CR16","doi-asserted-by":"crossref","unstructured":"B. Kalyanasundaram and K. R. Pruhs. Fault-tolerant scheduling. In 26th Annual ACM Symposium on Theory of Computing, pages 115\u2013124, 1994.","DOI":"10.1145\/195058.195115"},{"key":"13_CR17","doi-asserted-by":"publisher","first-page":"719","DOI":"10.1109\/12.600888","volume":"46","author":"S. Kartik","year":"1997","unstructured":"Kartik, S., Murthy, C.S.R.: Task allocation algorithms for maximizing reliability of distributed computing systems. IEEE Transactions on Computer Systems\u00a046, 719\u2013724 (1997)","journal-title":"IEEE Transactions on Computer Systems"},{"key":"13_CR18","doi-asserted-by":"crossref","unstructured":"Krevat, E., Castanos, J.G., Moreira, J.E.: Job scheduling for the bluegene\/l system. In: JSPP (2003)","DOI":"10.1007\/3-540-36180-4_3"},{"key":"13_CR19","unstructured":"Lee, I., Iyer, R.K.: Analysis of software halts in tandem system. In: Proceedings 3rd Intl. Software Reliability Engineering, October 1992, pp. 227\u2013236 (1992)"},{"issue":"4","key":"13_CR20","doi-asserted-by":"publisher","first-page":"419","DOI":"10.1109\/24.58720","volume":"39","author":"T.Y. Lin","year":"1990","unstructured":"Lin, T.Y., Siewiorek, D.P.: Error log analysis: Statistical modelling and heuristic trend analysis. IEEE Trans. on Reliability\u00a039(4), 419\u2013432 (1990)","journal-title":"IEEE Trans. on Reliability"},{"issue":"7","key":"13_CR21","doi-asserted-by":"publisher","first-page":"699","DOI":"10.1109\/12.936236","volume":"50","author":"Y. Ling","year":"2001","unstructured":"Ling, Y., Mi, J., Lin, X.: A Variational Calculus Approach to Optimal Checkpoint Placement. IEEE Transactions on Computer Systems\u00a050(7), 699\u2013708 (2001)","journal-title":"IEEE Transactions on Computer Systems"},{"issue":"3","key":"13_CR22","doi-asserted-by":"publisher","first-page":"209","DOI":"10.1145\/320557.320558","volume":"2","author":"G.M. Lohman","year":"1977","unstructured":"Lohman, G.M., Muckstadt, J.A.: Optimal Policy for Batch Operations: Backup, Checkpointing, Reorganization, and Updating. ACM Transactions on Database Systems\u00a02(3), 209\u2013222 (1977)","journal-title":"ACM Transactions on Database Systems"},{"key":"13_CR23","doi-asserted-by":"crossref","unstructured":"Lyu, M., Mendiratta, V.: Software Fault Tolerance in a Clustered Architecture: Techniques and Reliability Modeling. In: Proceedings 1999 IEEE Aerospace Conference, pp. 141\u2013150 (1999)","DOI":"10.1109\/AERO.1999.790197"},{"key":"13_CR24","doi-asserted-by":"crossref","unstructured":"Meyer, J., Wei, L.: Analysis of workload influence on dependability. In: Proceedings of the International Symposium on Fault-Tolerant Computing, pp. 84\u201389 (1988)","DOI":"10.1109\/FTCS.1988.5301"},{"key":"13_CR25","doi-asserted-by":"crossref","unstructured":"Mukherjee, S., Weaver, C., Emer, J., Reinhardt, S., Austin, T.: A Systematic Methodology to Compute the Architectural Vulnerabilityi Factors for a High-Performance Microprocessor. In: Proceedings of the International Symposium on Microarchitecture (MICRO), pp. 29\u201340 (2003)","DOI":"10.1109\/MICRO.2003.1253181"},{"issue":"11","key":"13_CR26","doi-asserted-by":"publisher","first-page":"1570","DOI":"10.1006\/jpdc.2001.1757","volume":"61","author":"J.S. Plank","year":"2001","unstructured":"Plank, J.S., Thomason, M.G.: Processor allocation and checkpoint interval selection in cluster computing systems. Journal of Parallel and Distributed Computing\u00a061(11), 1570\u20131590 (2001)","journal-title":"Journal of Parallel and Distributed Computing"},{"key":"13_CR27","unstructured":"Qin, X., Jiang, H., Swanson, D.R.: An efficient fault-tolerant scheduling algorithm for real-time tasks with precedence constraints in heterogeneous systems, citeseer.nj.nec.com\/qin02efficient.html"},{"key":"13_CR28","doi-asserted-by":"crossref","unstructured":"Sahoo, R., Sivasubramaniam, A., Squillante, M., Zhang, Y.: Failure Data Analysis of a Large-Scale Heterogeneous Server Environment. In: Proceedings of the 2004 International Conference on Dependable Systems and Networks, pp. 389\u2013398 (2004) (to appear)","DOI":"10.1109\/DSN.2004.1311948"},{"key":"13_CR29","doi-asserted-by":"crossref","unstructured":"Sahoo, R.K., Oliner, A.J., Rish, I., Gupta, M., Moreira, J.E., Ma, S., Vilalta, R., Sivasubramaniam, A.: Critical event prediction for proactive management in large-scale computer clusters. In: KDD, August 2003, pp. 426\u2013435 (2003)","DOI":"10.1145\/956750.956799"},{"key":"13_CR30","doi-asserted-by":"publisher","first-page":"1156","DOI":"10.1109\/12.165396","volume":"41","author":"S.M. Shaltz","year":"1992","unstructured":"Shaltz, S.M., Wang, J.P., Goto, M.: Task allocation for maximizing reliability of distributed computer systems. IEEE Transactions on Computer Systems\u00a041, 1156\u20131168 (1992)","journal-title":"IEEE Transactions on Computer Systems"},{"key":"13_CR31","doi-asserted-by":"crossref","unstructured":"Shivakumar, P., Kistler, M., Keckler, S., Burger, D., Alvisi, L.: Modeling the effect of technology trends on soft error rate of combinational logic. In: Proceedings of the 2002 International Conference on Dependable Systems and Networks, pp. 389\u2013398 (2002)","DOI":"10.1109\/DSN.2002.1028924"},{"key":"13_CR32","unstructured":"Squillante, M.S.: Matrix-Analytic Methods in Stochastic Parallel-Server Scheduling Models. Advances in Matrix-Analytic Methods for Stochastic Models. Notable Publications (1998)"},{"key":"13_CR33","doi-asserted-by":"crossref","unstructured":"Squillante, M.S., Wang, F., Papaefthymiou, M.: Stochastic Analysis of Gang Scheduling in Parallel and Distributed Systems. Technical Report, IBM Research Division (1996)","DOI":"10.1016\/S0166-5316(96)90031-0"},{"issue":"1","key":"13_CR34","doi-asserted-by":"publisher","first-page":"43","DOI":"10.1145\/511399.511341","volume":"30","author":"M.S. Squillante","year":"2002","unstructured":"Squillante, M.S., Zhang, Y., Sivasubramanian, A., Gautam, N., Moreira, J.E., Franke, H.: Modeling and analysis of dynamic coscheduling in parallel and distributed environments. Performance Evaluation Review\u00a030(1), 43\u201354 (2002)","journal-title":"Performance Evaluation Review"},{"key":"13_CR35","doi-asserted-by":"crossref","unstructured":"Sullivan, M., Chillarege, R.: Software Defects and Their Impact on System Availability - A Study of Field Failures in Operating Systems. In: Proceedings of The 21st International Symposium on Fault Tolerant Computer Systems (FTCS), pp. 2\u20139 (1991)","DOI":"10.1109\/FTCS.1991.146625"},{"key":"13_CR36","doi-asserted-by":"crossref","unstructured":"Tang, D., Iyer, R.K.: Impact of correlated failures on dependability in a vaxcluster system. In: IFIP Working Conference on Dependable Computing for Critical Applications (1991)","DOI":"10.1007\/978-3-7091-9198-9_9"},{"key":"13_CR37","doi-asserted-by":"crossref","unstructured":"Tang, D., Iyer, R.K., Subramani, S.S.: Failure analysis and modelling of a vaxcluster system. In: Proceedings 20th. Intl. Symposium on Fault-tolerant Computing, pp. 244\u2013251 (1990)","DOI":"10.1109\/FTCS.1990.89372"},{"key":"13_CR38","doi-asserted-by":"crossref","unstructured":"Vaidyanathan, K., Harper, R.E., Hunter, S.W., Trivedi, K.S.: Analysis and Implementation of Software Rejuvenation in Cluster Systems. In: Proceedings of the ACM SIGMETRICS 2001 Conference on Measurement and Modeling of Computer Systems, June 2001, pp. 62\u201371 (2001)","DOI":"10.1145\/378420.378434"},{"key":"13_CR39","doi-asserted-by":"crossref","unstructured":"Vaidyanathan, K., Harper, R.E., Hunter, S.W., Trivedi, K.S.: Analysis and implementation of software rejuvenation in cluster systems. In: SIGMETRICS 2001, pp. 62\u201371 (2001)","DOI":"10.1145\/378420.378434"},{"key":"13_CR40","unstructured":"Xu, J., Kallbarczyk, Z., Iyer, R.K.: Networked windows nt system field failure data analysis. Technical Report CRHC 9808 University of Illinois at Urbana-Champaign (1999)"},{"issue":"1","key":"13_CR41","doi-asserted-by":"publisher","first-page":"19","DOI":"10.1147\/rd.401.0019","volume":"40","author":"J. Zeigler","year":"1996","unstructured":"Zeigler, J.: Terrestrial Cosmic Rays. IBM Journal of Research and Development\u00a040(1), 19\u201339 (1996)","journal-title":"IBM Journal of Research and Development"},{"key":"13_CR42","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"245","DOI":"10.1007\/3-540-44520-X_33","volume-title":"Euro-Par 2000 Parallel Processing","author":"Y. Zhang","year":"2000","unstructured":"Zhang, Y., Franke, H., Moreira, J., Sivasubramaniam, A.: The Impact of Migration on Parallel Job Scheduling for Distributed Systems. In: Bode, A., Ludwig, T., Karl, W.C., Wism\u00fcller, R. (eds.) Euro-Par 2000. LNCS, vol.\u00a01900, pp. 245\u2013251. Springer, Heidelberg (2000)"},{"key":"13_CR43","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Franke, H., Moreira, J., Sivasubramaniam, A.: Improving parallel job scheduling by combining gang scheduling and backfilling techniques. In: Proceedings of the International Parallel and Distributed Processing Symposium, May 2000, pp. 133\u2013142 (2000)","DOI":"10.1109\/IPDPS.2000.845975"},{"issue":"3","key":"13_CR44","doi-asserted-by":"publisher","first-page":"236","DOI":"10.1109\/TPDS.2003.1189582","volume":"14","author":"Y. Zhang","year":"2003","unstructured":"Zhang, Y., Franke, H., Moreira, J., Sivasubramaniam, A.: An integrated approach to parallel scheduling using gang-scheduling backfilling and migration. IEEE Transactions on Parallel and Distributed System\u00a014(3), 236\u2013247 (2003)","journal-title":"IEEE Transactions on Parallel and Distributed System"}],"container-title":["Lecture Notes in Computer Science","Job Scheduling Strategies for Parallel Processing"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/11407522_13.pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,2,26]],"date-time":"2025-02-26T01:01:12Z","timestamp":1740531672000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/11407522_13"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2005]]},"ISBN":["9783540253303","9783540317951"],"references-count":44,"URL":"https:\/\/doi.org\/10.1007\/11407522_13","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2005]]}}}