{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,18]],"date-time":"2025-11-18T11:48:18Z","timestamp":1763466498492,"version":"3.37.3"},"publisher-location":"Berlin, Heidelberg","reference-count":34,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"type":"print","value":"9783642309601"},{"type":"electronic","value":"9783642309618"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2012]]},"DOI":"10.1007\/978-3-642-30961-8_8","type":"book-chapter","created":{"date-parts":[[2012,5,22]],"date-time":"2012-05-22T17:44:50Z","timestamp":1337708690000},"page":"102-115","source":"Crossref","is-referenced-by-count":30,"title":["libKOMP, an Efficient OpenMP Runtime System for Both Fork-Join and Data Flow Paradigms"],"prefix":"10.1007","author":[{"given":"Fran\u00e7ois","family":"Broquedis","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Thierry","family":"Gautier","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Vincent","family":"Danjean","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"unstructured":"Agathos, S.N., Hadjidoukas, P.E., Dimakopoulos, V.V.: Design and implementation of openmp tasks in the ompi compiler. In: Angelidis, P., Michalas, A. (eds.) Panhellenic Conference on Informatics, pp. 265\u2013269. IEEE (2011), \n                      \n                        http:\/\/dblp.uni-trier.de\/db\/conf\/pci\/pci2011.html#AgathosHD11","key":"8_CR1"},{"unstructured":"Agullo, E., Augonnet, C., Dongarra, J., Ltaief, H., Namyst, R., Roman, J., Thibault, S., Tomov, S.: Dynamically scheduled Cholesky factorization on multicore architectures with GPU accelerators. In: Symposium on Application Accelerators in High Performance Computing (SAAHPC), Knoxville, USA (July 2010)","key":"8_CR2"},{"issue":"2","key":"8_CR3","doi-asserted-by":"publisher","first-page":"115","DOI":"10.1007\/s002240011004","volume":"34","author":"N.S. Arora","year":"2001","unstructured":"Arora, N.S., Blumofe, R.D., Plaxton, C.G.: Thread scheduling for multiprogrammed multiprocessors. Theor. Comp. Sys.\u00a034(2), 115\u2013144 (2001)","journal-title":"Theor. Comp. Sys."},{"key":"8_CR4","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"154","DOI":"10.1007\/978-3-642-02303-3_13","volume-title":"Evolving OpenMP in an Age of Extreme Parallelism","author":"E. Ayguade","year":"2009","unstructured":"Ayguade, E., Badia, R.M., Cabrera, D., Duran, A., Gonzalez, M., Igual, F., Jimenez, D., Labarta, J., Martorell, X., Mayo, R., Perez, J.M., Quintana-Ort\u00ed, E.S.: A Proposal to Extend the OpenMP Tasking Model for Heterogeneous Architectures. In: M\u00fcller, M.S., de Supinski, B.R., Chapman, B.M. (eds.) IWOMP 2009. LNCS, vol.\u00a05568, pp. 154\u2013167. Springer, Heidelberg (2009), \n                      \n                        http:\/\/dx.doi.org\/10.1007\/978-3-642-02303-3_13"},{"key":"8_CR5","doi-asserted-by":"publisher","first-page":"2438","DOI":"10.1002\/cpe.1463","volume":"21","author":"R.M. Badia","year":"2009","unstructured":"Badia, R.M., Herrero, J.R., Labarta, J., P\u00e9rez, J.M., Quintana-Ort\u00ed, E.S., Quintana-Ort\u00ed, G.: Parallelizing dense and banded linear algebra libraries using smpss. Concurr. Comput.: Pract. Exper.\u00a021, 2438\u20132456 (2009)","journal-title":"Concurr. Comput.: Pract. Exper."},{"issue":"1","key":"8_CR6","doi-asserted-by":"publisher","first-page":"55","DOI":"10.1006\/jpdc.1996.0107","volume":"37","author":"R. Blumofe","year":"1996","unstructured":"Blumofe, R., Joerg, C., Kuszmaul, B., Leiserson, C., Randall, K., Zhou, Y.: Cilk: An efficient multithreaded runtime system. Journal of Parallel and Distributed Computing\u00a037(1), 55\u201369 (1996), \n                      \n                        citeseer.nj.nec.com\/article\/blumofe95cilk.html","journal-title":"Journal of Parallel and Distributed Computing"},{"key":"8_CR7","doi-asserted-by":"publisher","first-page":"38","DOI":"10.1016\/j.parco.2008.10.002","volume":"35","author":"A. Buttari","year":"2009","unstructured":"Buttari, A., Langou, J., Kurzak, J., Dongarra, J.: A class of parallel tiled linear algebra algorithms for multicore architectures. Parallel Comput.\u00a035, 38\u201353 (2009)","journal-title":"Parallel Comput."},{"key":"8_CR8","doi-asserted-by":"publisher","first-page":"291","DOI":"10.1177\/1094342007078442","volume":"21","author":"B. Chamberlain","year":"2007","unstructured":"Chamberlain, B., Callahan, D., Zima, H.: Parallel programmability and the chapel language. Int. J. High Perform. Comput. Appl.\u00a021, 291\u2013312 (2007), \n                      \n                        http:\/\/dl.acm.org\/citation.cfm?id=1286120.1286123","journal-title":"Int. J. High Perform. Comput. Appl."},{"key":"8_CR9","doi-asserted-by":"publisher","first-page":"519","DOI":"10.1145\/1103845.1094852","volume":"40","author":"P. Charles","year":"2005","unstructured":"Charles, P., Grothoff, C., Saraswat, V., Donawa, C., Kielstra, A., Ebcioglu, K., von Praun, C., Sarkar, V.: X10: an object-oriented approach to non-uniform cluster computing. SIGPLAN Not.\u00a040, 519\u2013538 (2005)","journal-title":"SIGPLAN Not."},{"key":"8_CR10","doi-asserted-by":"publisher","first-page":"17","DOI":"10.1023\/A:1019122726788","volume":"16","author":"B. Dumitrescu","year":"1997","unstructured":"Dumitrescu, B., Doreille, M., Roch, J.L., Trystram, D.: Two-dimensional block partitionings for the parallel sparse cholesky factorization. Numerical Algorithms\u00a016, 17\u201338 (1997)","journal-title":"Numerical Algorithms"},{"key":"8_CR11","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"100","DOI":"10.1007\/978-3-540-79561-2_9","volume-title":"OpenMP in a New Era of Parallelism","author":"A. Duran","year":"2008","unstructured":"Duran, A., Corbal\u00e1n, J., Ayguad\u00e9, E.: Evaluation of OpenMP Task Scheduling Strategies. In: Eigenmann, R., de Supinski, B.R. (eds.) IWOMP 2008. LNCS, vol.\u00a05004, pp. 100\u2013110. Springer, Heidelberg (2008)"},{"key":"8_CR12","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"111","DOI":"10.1007\/978-3-540-79561-2_10","volume-title":"OpenMP in a New Era of Parallelism","author":"A. Duran","year":"2008","unstructured":"Duran, A., Perez, J.M., Ayguad\u00e9, E., Badia, R.M., Labarta, J.: Extending the OpenMP Tasking Model to Allow Dependent Tasks. In: Eigenmann, R., de Supinski, B.R. (eds.) IWOMP 2008. LNCS, vol.\u00a05004, pp. 111\u2013122. Springer, Heidelberg (2008)"},{"doi-asserted-by":"crossref","unstructured":"Duran, A., Teruel, X., Ferrer, R., Martorell, X., Ayguade, E.: Barcelona openmp tasks suite: A set of benchmarks targeting the exploitation of task parallelism in openmp. In: International Conference on Parallel Processing, ICPP 2009, pp. 124\u2013131. IEEE (2009)","key":"8_CR13","DOI":"10.1109\/ICPP.2009.64"},{"unstructured":"Faucher, V.: Advanced Parallel Computing for Explosive Fluid-Structure Interaction. In: COMPDYN 2011, Corfu, Greece (May 2011)","key":"8_CR14"},{"key":"8_CR15","doi-asserted-by":"publisher","first-page":"212","DOI":"10.1145\/277650.277725","volume-title":"Proceedings of the ACM SIGPLAN 1998 Conference on Programming Language Design and Implementation, PLDI 1998","author":"M. Frigo","year":"1998","unstructured":"Frigo, M., Leiserson, C.E., Randall, K.H.: The implementation of the cilk-5 multithreaded language. In: Proceedings of the ACM SIGPLAN 1998 Conference on Programming Language Design and Implementation, PLDI 1998, pp. 212\u2013223. ACM, New York (1998)"},{"key":"8_CR16","first-page":"88","volume-title":"Proceedings of PACT 1998","author":"F. Galil\u00e9e","year":"1998","unstructured":"Galil\u00e9e, F., Roch, J.L., Cavalheiro, G.G.H., Doreille, M.: Athapascan-1: On-line building data flow graph in a parallel language. In: Proceedings of PACT 1998, p. 88. IEEE Computer Society, Washington, DC (1998)"},{"doi-asserted-by":"crossref","unstructured":"Gautier, T., Besseron, X., Pigeon, L.: Kaapi: a thread scheduling runtime system for data flow computations on cluster of multi-processors. In: PASCO 2007 (2007)","key":"8_CR17","DOI":"10.1145\/1278177.1278182"},{"key":"8_CR18","volume-title":"Workshop PAPP 2007 - Practical Aspects of High-Level Parallel Programming in (ICCS2007)","author":"T. Gautier","year":"2007","unstructured":"Gautier, T., Roch, J.L., Wagner, F.: Fine grain distributed implementation of a dataflow language with provable performances. In: Workshop PAPP 2007 - Practical Aspects of High-Level Parallel Programming in (ICCS2007). IEEE, Beijing (2007)"},{"key":"8_CR19","doi-asserted-by":"publisher","first-page":"355","DOI":"10.1145\/1810479.1810540","volume-title":"Proceedings of the 22nd ACM Symposium on Parallelism in Algorithms and Architectures, SPAA 2010","author":"D. Hendler","year":"2010","unstructured":"Hendler, D., Incze, I., Shavit, N., Tzafrir, M.: Flat combining and the synchronization-parallelism tradeoff. In: Proceedings of the 22nd ACM Symposium on Parallelism in Algorithms and Architectures, SPAA 2010, pp. 355\u2013364. ACM, New York (2010)"},{"key":"8_CR20","doi-asserted-by":"publisher","first-page":"280","DOI":"10.1145\/571825.571876","volume-title":"PODC 2002: Proceedings of the Twenty-First Annual Symposium on Principles of Distributed Computing","author":"D. Hendler","year":"2002","unstructured":"Hendler, D., Shavit, N.: Non-blocking steal-half work queues. In: PODC 2002: Proceedings of the Twenty-First Annual Symposium on Principles of Distributed Computing, pp. 280\u2013289. ACM, New York (2002)"},{"key":"8_CR21","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"235","DOI":"10.1007\/978-3-642-15291-7_23","volume-title":"Euro-Par 2010 - Parallel Processing","author":"E. Hermann","year":"2010","unstructured":"Hermann, E., Raffin, B., Faure, F., Gautier, T., Allard, J.: Multi-GPU and Multi-CPU Parallelization for Interactive Physics Simulations. In: D\u2019Ambra, P., Guarracino, M., Talia, D. (eds.) Euro-Par 2010. LNCS, vol.\u00a06272, pp. 235\u2013246. Springer, Heidelberg (2010)"},{"key":"8_CR22","doi-asserted-by":"publisher","first-page":"15","DOI":"10.1002\/cpe.1467","volume":"22","author":"J. Kurzak","year":"2010","unstructured":"Kurzak, J., Ltaief, H., Dongarra, J., Badia, R.M.: Scheduling dense linear algebra operations on multicore processors. Concurr. Comput.: Pract. Exper.\u00a022, 15\u201344 (2010)","journal-title":"Concurr. Comput. : Pract. Exper."},{"key":"8_CR23","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"165","DOI":"10.1007\/978-3-642-21487-5_13","volume-title":"OpenMP in the Petascale Era","author":"J. LaGrone","year":"2011","unstructured":"LaGrone, J., Aribuki, A., Addison, C., Chapman, B.: A Runtime Implementation of OpenMP Tasks. In: Chapman, B.M., Gropp, W.D., Kumaran, K., M\u00fcller, M.S. (eds.) IWOMP 2011. LNCS, vol.\u00a06665, pp. 165\u2013178. Springer, Heidelberg (2011), \n                      \n                        http:\/\/dl.acm.org\/citation.cfm?id=2023025.2023042"},{"unstructured":"Le Mentec, F., Danjean, V., Gautier, T.: X-Kaapi C programming interface. Tech. Rep. RT-0417, INRIA (December 2011)","key":"8_CR24"},{"unstructured":"Le Mentec, F., Gautier, T., Danjean, V.: The X-Kaapi\u2019s Application Programming Interface. Part I: Data Flow Programming. Tech. Rep. RT-0418, INRIA (December 2011)","key":"8_CR25"},{"key":"8_CR26","doi-asserted-by":"publisher","first-page":"45","DOI":"10.1145\/1594835.1504186","volume":"44","author":"M.M. Michael","year":"2009","unstructured":"Michael, M.M., Vechev, M.T., Saraswat, V.A.: Idempotent work stealing. SIGPLAN Not.\u00a044, 45\u201354 (2009)","journal-title":"SIGPLAN Not."},{"key":"8_CR27","doi-asserted-by":"publisher","first-page":"49","DOI":"10.1145\/1988796.1988804","volume-title":"Proceedings of the 1st International Workshop on Runtime and Operating Systems for Supercomputers, ROSS 2011","author":"S.L. Olivier","year":"2011","unstructured":"Olivier, S.L., Porterfield, A.K., Wheeler, K.B., Prins, J.F.: Scheduling task parallelism on multi-socket multicore systems. In: Proceedings of the 1st International Workshop on Runtime and Operating Systems for Supercomputers, ROSS 2011, pp. 49\u201356. ACM, New York (2011), \n                      \n                        http:\/\/doi.acm.org\/10.1145\/1988796.1988804"},{"doi-asserted-by":"crossref","unstructured":"Olivier, S.L., Porterfield, A.K., Wheeler, K.B., Spiegel, M., Prins, J.F.: Openmp task scheduling strategies for multicore numa systems. International Journal of High Performance Computing Applications (2012)","key":"8_CR28","DOI":"10.1177\/1094342011434065"},{"unstructured":"OpenMP Architecture Review Board (1997-2008), \n                      \n                        http:\/\/www.openmp.org","key":"8_CR29"},{"doi-asserted-by":"crossref","unstructured":"Robison, A., Voss, M., Kukanov, A.: Optimization via reflection on work stealing in TBB. In: IPDPS (2008)","key":"8_CR30","DOI":"10.1109\/IPDPS.2008.4536188"},{"key":"8_CR31","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"99","DOI":"10.1007\/978-3-642-21878-1_13","volume-title":"Euro-Par 2010 Parallel Processing Workshops","author":"M. Tchiboukdjian","year":"2011","unstructured":"Tchiboukdjian, M., Danjean, V., Gautier, T., Le Mentec, F., Raffin, B.: A Work Stealing Scheduler for Parallel Loops on Shared Cache Multicores. In: Guarracino, M.R., Vivien, F., Tr\u00e4ff, J.L., Cannatoro, M., Danelutto, M., Hast, A., Perla, F., Kn\u00fcpfer, A., Di Martino, B., Alexander, M. (eds.) Euro-Par-Workshop 2010. LNCS, vol.\u00a06586, pp. 99\u2013107. Springer, Heidelberg (2011)"},{"key":"8_CR32","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"291","DOI":"10.1007\/978-3-642-17514-5_25","volume-title":"Algorithms and Computation","author":"M. Tchiboukdjian","year":"2010","unstructured":"Tchiboukdjian, M., Gast, N., Trystram, D., Roch, J.-L., Bernard, J.: A Tighter Analysis of Work Stealing. In: Cheong, O., Chwa, K.-Y., Park, K. (eds.) ISAAC 2010, Part II. LNCS, vol.\u00a06507, pp. 291\u2013302. Springer, Heidelberg (2010)"},{"key":"8_CR33","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"887","DOI":"10.1007\/978-3-540-85451-7_95","volume-title":"Euro-Par 2008 Parallel Processing","author":"D. Traor\u00e9","year":"2008","unstructured":"Traor\u00e9, D., Roch, J.-L., Maillard, N., Gautier, T., Bernard, J.: Deque-Free Work-Optimal Parallel STL Algorithms. In: Luque, E., Margalef, T., Ben\u00edtez, D. (eds.) Euro-Par 2008. LNCS, vol.\u00a05168, pp. 887\u2013897. Springer, Heidelberg (2008), \n                      \n                        http:\/\/www.caos.uab.es\/europar2008\/"},{"unstructured":"YarKhan, A., Kurzak, J., Dongarra, J.: Quark users\u2019 guide: Queueing and runtime for kernels. Tech. Rep. ICL-UT-11-02. University of Tennessee (2011)","key":"8_CR34"}],"container-title":["Lecture Notes in Computer Science","OpenMP in a Heterogeneous World"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-642-30961-8_8.pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,5,4]],"date-time":"2021-05-04T11:36:43Z","timestamp":1620128203000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-642-30961-8_8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2012]]},"ISBN":["9783642309601","9783642309618"],"references-count":34,"URL":"https:\/\/doi.org\/10.1007\/978-3-642-30961-8_8","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2012]]}}}