{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,6,16]],"date-time":"2024-06-16T15:51:44Z","timestamp":1718553104521},"reference-count":30,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2014,9,26]],"date-time":"2014-09-26T00:00:00Z","timestamp":1411689600000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"published-print":{"date-parts":[[2014,11]]},"DOI":"10.1007\/s11227-014-1302-y","type":"journal-article","created":{"date-parts":[[2014,9,25]],"date-time":"2014-09-25T13:19:16Z","timestamp":1411651156000},"page":"830-844","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Improving an autotuning engine for 3D Fast Wavelet Transform on manycore systems"],"prefix":"10.1007","volume":"70","author":[{"given":"Gregorio","family":"Bernab\u00e9","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Javier","family":"Cuenca","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Luis Pedro","family":"Garc\u00eda","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Domingo","family":"Gim\u00e9nez","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2014,9,26]]},"reference":[{"issue":"8","key":"1302_CR1","doi-asserted-by":"crossref","first-page":"85","DOI":"10.1109\/MC.2005.261","volume":"38","author":"D Manocha","year":"2005","unstructured":"Manocha D (2005) General-purpose computation using graphic processors. IEEE Comput 38(8):85\u201388","journal-title":"IEEE Comput"},{"issue":"1","key":"1302_CR2","doi-asserted-by":"crossref","first-page":"80","DOI":"10.1111\/j.1467-8659.2007.01012.x","volume":"26","author":"JD Owens","year":"2007","unstructured":"Owens JD, Luebke D, Govindaraju N, Harris M, Kr\u00fcger J, Lefohn AE, Purcell TJ (2007) A survey of general-purpose computation on graphics hardware. Comput Graph Forum 26(1):80\u2013113","journal-title":"Comput Graph Forum"},{"key":"1302_CR3","unstructured":"CUDA Zone maintained by NVIDIA. http:\/\/www.nvidia.com\/object\/cuda.html (2009)"},{"key":"1302_CR4","unstructured":"NVIDIA, Whitepaper NVIDIA\u2019s Next Generation CUDA Compute Architecture: Kepler GK110. http:\/\/www.nvidia.com\/content\/pdf\/kepler\/nvidia-kepler-gk110-architecture-whitepaper.pdf (2012)"},{"key":"1302_CR5","unstructured":"Intel Corporation, An Overview of Programming for Intel Xeon processors and Intel Xeon Phi. coprocessors, https:\/\/software.intel.com\/en-us\/articles\/an-overview-of-programming-for-intel-xeon-processors-and-intel-xeon-phi-coprocessors (2013)"},{"key":"1302_CR6","doi-asserted-by":"crossref","unstructured":"Bernab\u00e9 G, Cuenca J, Gim\u00e9nez D (2013) Optimizing a 3D-FWT code in heterogeneous cluster of multicore CPUs and manycore GPUs. In: 25th international symposium on computer architecture and high performance computing (2013)","DOI":"10.1109\/SBAC-PAD.2013.26"},{"key":"1302_CR7","doi-asserted-by":"crossref","unstructured":"Carvalho E, Calazans N, Moraes F (2007) Heuristics for dynamic task mapping in NoC-based heterogeneous MPSoCs. In: Proceedings of 18th IEEE\/IFIP international workshop on rapid system prototyping, pp 34\u201340","DOI":"10.1109\/RSP.2007.26"},{"key":"1302_CR8","doi-asserted-by":"crossref","first-page":"105","DOI":"10.1016\/j.sysarc.2004.10.005","volume":"52","author":"F Almeida","year":"2006","unstructured":"Almeida F, Gonz\u00e1lez D, Moreno L (2006) The master-slave paradigm on heterogeneous systems: a dynamic programming approach for the optimal mapping. J Syst Architect 52:105\u2013116","journal-title":"J Syst Architect"},{"key":"1302_CR9","doi-asserted-by":"crossref","first-page":"88","DOI":"10.1016\/j.sysarc.2004.10.008","volume":"52","author":"A Giersch","year":"2006","unstructured":"Giersch A, Robert Y, Vivien F (2006) Scheduling tasks sharing files on heterogeneous master-slave platforms. J Syst Archit 52:88\u2013104","journal-title":"J Syst Archit"},{"key":"1302_CR10","doi-asserted-by":"crossref","first-page":"569","DOI":"10.1016\/j.future.2006.09.007","volume":"23","author":"C Hsu","year":"2007","unstructured":"Hsu C, Chen T, Li K (2007) Performance effective pre-scheduling strategy for heterogeneous grid systems in the master slave paradigm. Future Gener Comput Syst 23:569\u2013579","journal-title":"Future Gener Comput Syst"},{"key":"1302_CR11","doi-asserted-by":"crossref","first-page":"319","DOI":"10.1109\/TPDS.2004.1271181","volume":"15","author":"C Banino","year":"2004","unstructured":"Banino C, Beaumont O, Carter L, Ferrante J, Legrand A, Robert Y (2004) Scheduling strategies for master-slave tasking on heterogeneous processor platforms. IEEE Trans Parallel Distrib Syst 15:319\u2013330","journal-title":"IEEE Trans Parallel Distrib Syst"},{"key":"1302_CR12","doi-asserted-by":"crossref","unstructured":"Volkov V, Demmel JW (2008) Benchmarking GPUs to tune dense linear algebra. In: Proceedings of 2008 ACM\/IEEE conference on supercomputing SC\u201908","DOI":"10.1109\/SC.2008.5214359"},{"key":"1302_CR13","doi-asserted-by":"crossref","unstructured":"Yinan L, Dongarra J, Tomov S (2009) A note on auto-tuning GEMM for GPUs. In: Proceedings of 9th international conference on computational science: part I, pp 884\u2013892 (2009)","DOI":"10.1007\/978-3-642-01970-8_89"},{"key":"1302_CR14","doi-asserted-by":"crossref","first-page":"110","DOI":"10.1007\/978-3-642-28145-7_11","volume":"7134","author":"A Davidson","year":"2012","unstructured":"Davidson A, Owens J (2012) Toward techniques for auto-tuning GPU algorithms. Appl Parallel Sci Comput Lect Notes Comput Sci 7134:110\u2013119","journal-title":"Appl Parallel Sci Comput Lect Notes Comput Sci"},{"key":"1302_CR15","doi-asserted-by":"crossref","unstructured":"Fatica M (2009) Accelerating linpack with CUDA on heterogenous clusters. In: Proceedings of 2nd workshop on general purpose processing on graphics processing units, GPGPU-2, pp 46\u201351","DOI":"10.1145\/1513895.1513901"},{"key":"1302_CR16","unstructured":"Spiga F, Girotto I (2008) phiGEMM: a CPU-GPU library for porting quantum ESPRESSO on hybrid systems. In: Proceedings of 16th Euromicro conference on parallel, distributed and network-based processing, pp 368\u2013375"},{"key":"1302_CR17","doi-asserted-by":"crossref","first-page":"854","DOI":"10.1007\/s11390-011-0184-1","volume":"26","author":"F Wang","year":"2011","unstructured":"Wang F, Yang C, Du Y, Chen HYJ, Xu W (2011) Optimizing LINPACK benchmark on GPU-accelerated petascale supercomputer. J Comput Sci Technol 26:854\u2013865","journal-title":"J Comput Sci Technol"},{"key":"1302_CR18","doi-asserted-by":"crossref","unstructured":"Tsai Y, Wang W, Chen R (2012) Tuning block size for QR factorization on CPU-GPU hybrid systems. In: Proceedings of IEEE 6th international symposium on embedded multicore socs (MCSoC), pp 205\u2013211","DOI":"10.1109\/MCSoC.2012.32"},{"key":"1302_CR19","first-page":"187","volume":"23","author":"C Augonnet","year":"2011","unstructured":"Augonnet C, Thibault S, Namyst R, Wacrenier P (2011) StarPU: a unified platform for task scheduling on heterogeneous multicore architectures. J Comput Sci Technol 23:187\u2013198","journal-title":"J Comput Sci Technol"},{"key":"1302_CR20","unstructured":"Intel Corporation, Intel MKL web page. http:\/\/software.intel.com\/en-us\/intel-mkl\/ (2013)"},{"key":"1302_CR21","doi-asserted-by":"crossref","unstructured":"Dongarra J, Gates M, Haidar A, Jia Y, Kabir K, Luszczek P, Tomov S (2013) Portable HPC programming on intel many-integrated-core hardware with MAGMA Port to Xeon Phi. In: Parallel processing and applied mathematics (2013)","DOI":"10.1007\/978-3-642-55224-3_53"},{"issue":"7","key":"1302_CR22","doi-asserted-by":"crossref","first-page":"674","DOI":"10.1109\/34.192463","volume":"11","author":"S Mallat","year":"1989","unstructured":"Mallat S (1989) A theory for multiresolution signal descomposition: the wavelet representation. IEEE Trans Patt Anal Mach Intell 11(7):674\u2013693","journal-title":"IEEE Trans Patt Anal Mach Intell"},{"issue":"3","key":"1302_CR23","doi-asserted-by":"crossref","first-page":"526","DOI":"10.1016\/j.jss.2008.09.034","volume":"82","author":"G Bernab\u00e9","year":"2009","unstructured":"Bernab\u00e9 G, Garc\u00eda JM, Gonz\u00e1lez J (2009) A lossy 3D wavelet transform for high-quality compression of medical video. J Syst Softw 82(3):526\u2013534","journal-title":"J Syst Softw"},{"key":"1302_CR24","doi-asserted-by":"crossref","unstructured":"Daubechies I (1992) Ten lectures on wavelets. Society for Industrial and Applied Mathematics","DOI":"10.1137\/1.9781611970104"},{"key":"1302_CR25","unstructured":"The Khronos Group, The OpenCL core API specification, http:\/\/www.khronos.org\/registry\/cl (2011)"},{"key":"1302_CR26","doi-asserted-by":"crossref","unstructured":"Franco J, Bernab\u00e9 G, Fern\u00e1ndez J, Ujald\u00f3n M (2010) Parallel 3D fast wavelet transform on manycore GPUs and multicore CPUs. In: 10 international conference on computational science (2010)","DOI":"10.1016\/j.procs.2010.04.122"},{"key":"1302_CR27","doi-asserted-by":"crossref","unstructured":"Bernab\u00e9 G, Cuenca J, Gim\u00e9nez D (2013) Optimization techniques for 3D-FWT on systems with manycore GPUs and multicore CPUs. In: International conference on computational science (2013)","DOI":"10.1016\/j.procs.2013.05.195"},{"key":"1302_CR28","doi-asserted-by":"crossref","first-page":"408","DOI":"10.1007\/s10766-013-0249-6","volume":"42","author":"J C\u00e1mara","year":"2014","unstructured":"C\u00e1mara J, Cuenca J, Gim\u00e9nez D, Garc\u00eda LP, Vidal A (2014) Empirical installation of linear algebra shared-memory subroutines for auto-tuning. Int J Parallel Program 42:408\u2013434","journal-title":"Int J Parallel Program"},{"key":"1302_CR29","doi-asserted-by":"crossref","unstructured":"Franco J, Bernab\u00e9 G, Fern\u00e1ndez J, Acacio ME, Parallel A (2009) Implementation of the 2D wavelet transform using CUDA. In: 17 Euromicro international conference on parallel, distributed, and network-based processing (2009)","DOI":"10.1109\/PDP.2009.40"},{"key":"1302_CR30","unstructured":"NVIDIA Tutorial at PDP\u201908, CUDA: A New Architecture for Computing on the GPU (February 2008)"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-014-1302-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11227-014-1302-y\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-014-1302-y","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,8,15]],"date-time":"2019-08-15T11:17:02Z","timestamp":1565867822000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11227-014-1302-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2014,9,26]]},"references-count":30,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2014,11]]}},"alternative-id":["1302"],"URL":"https:\/\/doi.org\/10.1007\/s11227-014-1302-y","relation":{},"ISSN":["0920-8542","1573-0484"],"issn-type":[{"value":"0920-8542","type":"print"},{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2014,9,26]]}}}