{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,27]],"date-time":"2025-08-27T16:18:53Z","timestamp":1756311533855,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":32,"publisher":"ACM","license":[{"start":{"date-parts":[[2016,3,12]],"date-time":"2016-03-12T00:00:00Z","timestamp":1457740800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2016,3,12]]},"DOI":"10.1145\/2884045.2884052","type":"proceedings-article","created":{"date-parts":[[2016,3,4]],"date-time":"2016-03-04T20:57:50Z","timestamp":1457125070000},"page":"53-62","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":17,"title":["Implementing directed acyclic graphs with the heterogeneous system architecture"],"prefix":"10.1145","author":[{"given":"Sooraj","family":"Puthoor","sequence":"first","affiliation":[{"name":"AMD Research"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ashwin M.","family":"Aji","sequence":"additional","affiliation":[{"name":"AMD Research"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shuai","family":"Che","sequence":"additional","affiliation":[{"name":"AMD Research"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mayank","family":"Daga","sequence":"additional","affiliation":[{"name":"AMD Research"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei","family":"Wu","sequence":"additional","affiliation":[{"name":"The University of Tennessee, Knoxville"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bradford M.","family":"Beckmann","sequence":"additional","affiliation":[{"name":"AMD Research"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Gregory","family":"Rodgers","sequence":"additional","affiliation":[{"name":"AMD Research"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2016,3,12]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/AICCSA.2011.6126599"},{"key":"e_1_3_2_1_2_1","first-page":"012037","volume-title":"IOP Publishing","author":"Agullo E.","year":"2009","unstructured":"E. Agullo , J. Demmel , J. Dongarra , B. Hadri , J. Kurzak , J. Langou , H. Ltaief , P. Luszczek , and S. Tomov , \" Numerical linear algebra on emerging architectures: The PLASMA and MAGMA projects,\" in Journal of Physics: Conference Series, vol. 180, no. 1 . IOP Publishing , 2009 , p. 012037 . E. Agullo, J. Demmel, J. Dongarra, B. Hadri, J. Kurzak, J. Langou, H. Ltaief, P. Luszczek, and S. Tomov, \"Numerical linear algebra on emerging architectures: The PLASMA and MAGMA projects,\" in Journal of Physics: Conference Series, vol. 180, no. 1. IOP Publishing, 2009, p. 012037."},{"key":"e_1_3_2_1_3_1","volume-title":"Marina del Rey","author":"Amarasinghe S.","year":"2011","unstructured":"S. Amarasinghe , M. Hall , R. Lethin , K. Pingali , D. Quinlan , V. Sarkar , J. Shalf , R. Lucas , K. Yelick , P. Balanji , P. C. Diniz , A. Koniges , and M. Snir , \" Exascale Programming Challenges.\" Proceedings of the DOE Workshop on Exascale Programming Challenges , Marina del Rey , CA , USA. Jul 2011 . http:\/\/science.energy.gov\/\/media\/ascr\/pdf\/programdocuments\/docs\/ProgrammingChallengesWorkshopReport.pdf S. Amarasinghe, M. Hall, R. Lethin, K. Pingali, D. Quinlan, V. Sarkar, J. Shalf, R. Lucas, K. Yelick, P. Balanji, P. C. Diniz, A. Koniges, and M. Snir, \"Exascale Programming Challenges.\" Proceedings of the DOE Workshop on Exascale Programming Challenges, Marina del Rey, CA, USA. Jul 2011. http:\/\/science.energy.gov\/\/media\/ascr\/pdf\/programdocuments\/docs\/ProgrammingChallengesWorkshopReport.pdf"},{"key":"e_1_3_2_1_4_1","unstructured":"AMD Accelerated Processing Units (APUs). http:\/\/www.amd.com\/en-us\/innovations\/software-technologies\/apu  AMD Accelerated Processing Units (APUs). http:\/\/www.amd.com\/en-us\/innovations\/software-technologies\/apu"},{"key":"e_1_3_2_1_5_1","volume-title":"LAPACK Users' Guide.\" SIAM","author":"Anderson E.","year":"1992","unstructured":"E. Anderson , Z. Bai , C. Bischof , L. S. Blackford , J. W. Demmel , J. Dongarra , J. Du Croz , A. Greenbaum , S. Hammarling , A. McKenney , and D. Sorensen , \" LAPACK Users' Guide.\" SIAM , 1992 . E. Anderson, Z. Bai, C. Bischof, L. S. Blackford, J. W. Demmel, J. Dongarra, J. Du Croz, A. Greenbaum, S. Hammarling, A. McKenney, and D. Sorensen, \"LAPACK Users' Guide.\" SIAM, 1992."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1002\/cpe.1631"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/2555243.2555258"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/209936.209958"},{"key":"e_1_3_2_1_10_1","volume-title":"Applying AMD's \"Kaveri\" APU for Heterogeneous Computing.\" Hotchips","author":"Bouvier D.","year":"2014","unstructured":"D. Bouvier and B. Sander , \" Applying AMD's \"Kaveri\" APU for Heterogeneous Computing.\" Hotchips 26, August 2014 . http:\/\/hotchips.org\/wp-content\/uploads\/hc_archives\/hc26\/HC26-11-day1-epub\/HC26.11-2-Mobile-Processors-epub\/HC26.11.220-Bouvier-Kaveri-AMD-Final.pdf. D. Bouvier and B. Sander, \"Applying AMD's \"Kaveri\" APU for Heterogeneous Computing.\" Hotchips 26, August 2014. http:\/\/hotchips.org\/wp-content\/uploads\/hc_archives\/hc26\/HC26-11-day1-epub\/HC26.11-2-Mobile-Processors-epub\/HC26.11.220-Bouvier-Kaveri-AMD-Final.pdf."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC.2009.5306797"},{"key":"e_1_3_2_1_12_1","volume-title":"An algorithmic Approach","author":"Christofides N.","year":"1975","unstructured":"N. Christofides , \" Graph Theory : An algorithmic Approach .\" 1975 . N. Christofides, \"Graph Theory: An algorithmic Approach.\" 1975."},{"key":"e_1_3_2_1_13_1","unstructured":"clBLAS library https:\/\/github.com\/clMathLibraries\/clBLAS.  clBLAS library https:\/\/github.com\/clMathLibraries\/clBLAS."},{"volume-title":"Rudolf Eigenmann and Bronis R. De Supinski (Eds.)","author":"Duran A.","key":"e_1_3_2_1_14_1","unstructured":"A. Duran , J. Perez , E. Ayguad\u00e9 , R. Badia , J. Labarta , \" Extending the OpenMP tasking model to allow dependent tasks.\" In Proceedings of the 4th International Conference on OpenMP in a New Era of Parallelism (IWOMP'08) , Rudolf Eigenmann and Bronis R. De Supinski (Eds.) . Springer-Verlag , Berlin, Heidelberg , 111-122. A. Duran, J. Perez, E. Ayguad\u00e9, R. Badia, J. Labarta, \"Extending the OpenMP tasking model to allow dependent tasks.\" In Proceedings of the 4th International Conference on OpenMP in a New Era of Parallelism (IWOMP'08), Rudolf Eigenmann and Bronis R. De Supinski (Eds.). Springer-Verlag, Berlin, Heidelberg, 111-122."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-14313-2_51"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/277650.277725"},{"key":"e_1_3_2_1_17_1","volume-title":"HSA Platform System Architecture Specification. Version 1.0 (Jan","author":"Foundation HSA","year":"2015","unstructured":"HSA Foundation . 2015. HSA Platform System Architecture Specification. Version 1.0 (Jan . 2015 ). http:\/\/www.hsafoundation.com\/standards HSA Foundation. 2015. HSA Platform System Architecture Specification. Version 1.0 (Jan. 2015). http:\/\/www.hsafoundation.com\/standards"},{"key":"e_1_3_2_1_18_1","volume-title":"Compiler Writer, and Object Format (BRIG). Version 1.0 (Feb.","author":"Foundation HSA","year":"2015","unstructured":"HSA Foundation . 2015. HSA Programmer's Reference Manual: HSAIL Virtual ISA and Programming Model , Compiler Writer, and Object Format (BRIG). Version 1.0 (Feb. 2015 ). http:\/\/www.hsafoundation.com\/standards HSA Foundation. 2015. HSA Programmer's Reference Manual: HSAIL Virtual ISA and Programming Model, Compiler Writer, and Object Format (BRIG). Version 1.0 (Feb. 2015). http:\/\/www.hsafoundation.com\/standards"},{"key":"e_1_3_2_1_19_1","volume-title":"HSA Runtime Programmers Reference Manual. Version 1.0 (Feb","author":"Foundation HSA","year":"2015","unstructured":"HSA Foundation . 2015. HSA Runtime Programmers Reference Manual. Version 1.0 (Feb . 2015 ). http:\/\/www.hsafoundation.com\/standards HSA Foundation. 2015. HSA Runtime Programmers Reference Manual. Version 1.0 (Feb. 2015). http:\/\/www.hsafoundation.com\/standards"},{"key":"e_1_3_2_1_20_1","unstructured":"\"Intel Threading Building Blocks.\" Available: http:\/\/www.threadingbuildingblocks.org\/  \"Intel Threading Building Blocks.\" Available: http:\/\/www.threadingbuildingblocks.org\/"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/2676870.2676883"},{"key":"e_1_3_2_1_22_1","volume-title":"VECPAR 2010","author":"Ltaief H.","year":"1932","unstructured":"H. Ltaief , S. Tomov , R. Nath , P. Du , J. Dongarra , \" A Scalable High Performant Cholesky Factorization for Multicore with GPU Accelerators .\" In High Performance Comptuting for Computational Science -- VECPAR 2010 . DOI= http:\/\/dx.doi.org\/10.1007%2f978-3-642- 1932 8-6_11 H. Ltaief, S. Tomov, R. Nath, P. Du, J. Dongarra, \"A Scalable High Performant Cholesky Factorization for Multicore with GPU Accelerators.\" In High Performance Comptuting for Computational Science -- VECPAR 2010. DOI= http:\/\/dx.doi.org\/10.1007%2f978-3-642-19328-6_11"},{"key":"e_1_3_2_1_23_1","volume-title":"Expressing Locality and Independence with Logical Regions.\" In the International Conference on Supercomputing","author":"Bauer M.","year":"2012","unstructured":"M. Bauer , S. Treichler , E. Slaughter , A. Aiken , \"Legion : Expressing Locality and Independence with Logical Regions.\" In the International Conference on Supercomputing , 2012 M. Bauer, S. Treichler, E. Slaughter, A. Aiken, \"Legion: Expressing Locality and Independence with Logical Regions.\" In the International Conference on Supercomputing, 2012"},{"volume-title":"Stanford University","author":"Saunders M. A.","key":"e_1_3_2_1_24_1","unstructured":"M. A. Saunders , \" Large-Scale Linear Programming Using the Cholesky Factorization .\" Technical Report . Stanford University , Stanford, CA, USA . M. A. Saunders, \"Large-Scale Linear Programming Using the Cholesky Factorization.\" Technical Report. Stanford University, Stanford, CA, USA."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.1998.658762"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1016\/0022-2836(70)90057-4"},{"key":"e_1_3_2_1_27_1","unstructured":"OpenMP4.5 Specification. 2015. The OpenMP Architecture Review Board. http:\/\/www.openmp.org\/mp-documents\/openmp-4.5.pdf  OpenMP4.5 Specification. 2015. The OpenMP Architecture Review Board. http:\/\/www.openmp.org\/mp-documents\/openmp-4.5.pdf"},{"volume-title":"Portable Operating System Interface (POSIX)","year":"2008","key":"e_1_3_2_1_28_1","unstructured":"POSIX.1003.1-2008. IEEE Standard , Portable Operating System Interface (POSIX) . The Open Group Standard Base Specification . http:\/\/standards.ieee.org\/findstds\/standard\/1003.1- 2008 .html POSIX.1003.1-2008. IEEE Standard, Portable Operating System Interface (POSIX). The Open Group Standard Base Specification. http:\/\/standards.ieee.org\/findstds\/standard\/1003.1-2008.html"},{"key":"e_1_3_2_1_29_1","volume-title":"CLOC: C Language Offline Compiler. Version 0.9 (April","author":"Rodgers G.","year":"2015","unstructured":"G. Rodgers and S. Ramalingam , \" CLOC: C Language Offline Compiler. Version 0.9 (April . 2015 ).\" Github repository. http:\/\/github.com\/HSAFoundation\/CLOCs. G. Rodgers and S. Ramalingam, \"CLOC: C Language Offline Compiler. Version 0.9 (April. 2015).\" Github repository. http:\/\/github.com\/HSAFoundation\/CLOCs."},{"key":"e_1_3_2_1_30_1","unstructured":"Thread Building Blocks (TBB). https:\/\/www.threadingbuildingblocks.org\/  Thread Building Blocks (TBB). https:\/\/www.threadingbuildingblocks.org\/"},{"key":"e_1_3_2_1_31_1","unstructured":"S. Tomov J. Dongarra V. Volkov and J. Demmel. MAGMA Library version 0.1. http:\/\/icl.cs.utk.edu\/magma 08\/2009.  S. Tomov J. Dongarra V. Volkov and J. Demmel. MAGMA Library version 0.1. http:\/\/icl.cs.utk.edu\/magma 08\/2009."},{"key":"e_1_3_2_1_32_1","volume-title":"High Performance Graphics","author":"Tzeng S.","year":"2010","unstructured":"S. Tzeng , A. Patney , and J. D. Owens . Task Management for Irregular-Parallel Workloads on the GPU , High Performance Graphics , 2010 S. Tzeng, A. Patney, and J. D. Owens. Task Management for Irregular-Parallel Workloads on the GPU, High Performance Graphics, 2010"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/79173.79181"}],"event":{"name":"PPoPP '16: 21st ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming","acronym":"PPoPP '16","location":"Barcelona Spain"},"container-title":["Proceedings of the 9th Annual Workshop on General Purpose Processing using Graphics Processing Unit"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/2884045.2884052","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/2884045.2884052","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T04:54:08Z","timestamp":1750222448000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/2884045.2884052"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2016,3,12]]},"references-count":32,"alternative-id":["10.1145\/2884045.2884052","10.1145\/2884045"],"URL":"https:\/\/doi.org\/10.1145\/2884045.2884052","relation":{},"subject":[],"published":{"date-parts":[[2016,3,12]]},"assertion":[{"value":"2016-03-12","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}