{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,8]],"date-time":"2025-11-08T12:50:34Z","timestamp":1762606234249,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":37,"publisher":"ACM","license":[{"start":{"date-parts":[[2013,3,16]],"date-time":"2013-03-16T00:00:00Z","timestamp":1363392000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100004963","name":"Seventh Framework Programme","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100004963","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003329","name":"Ministerio de Econom\u00eda y Competitividad","doi-asserted-by":"publisher","award":["TIN2007-60625, TIN2012-34557"],"award-info":[{"award-number":["TIN2007-60625, TIN2012-34557"]}],"id":[{"id":"10.13039\/501100003329","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100005302","name":"University of Illinois at Urbana-Champaign","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100005302","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100007065","name":"Nvidia","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100007065","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2013,3,16]]},"DOI":"10.1145\/2458523.2458524","type":"proceedings-article","created":{"date-parts":[[2013,4,1]],"date-time":"2013-04-01T19:39:39Z","timestamp":1364845179000},"page":"1-11","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":14,"title":["Comparison based sorting for systems with multiple GPUs"],"prefix":"10.1145","author":[{"given":"Ivan","family":"Tanasic","sequence":"first","affiliation":[{"name":"Barcelona Supercomputing Center"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Llu\u00eds","family":"Vilanova","sequence":"additional","affiliation":[{"name":"Barcelona Supercomputing Center"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Marc","family":"Jord\u00e0","sequence":"additional","affiliation":[{"name":"Barcelona Supercomputing Center"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Javier","family":"Cabezas","sequence":"additional","affiliation":[{"name":"Barcelona Supercomputing Center"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Isaac","family":"Gelado","sequence":"additional","affiliation":[{"name":"Barcelona Supercomputing Center"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Nacho","family":"Navarro","sequence":"additional","affiliation":[{"name":"Barcelona Supercomputing Center and Universitat Politecnica de Catalunya"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wen-mei","family":"Hwu","sequence":"additional","affiliation":[{"name":"University of Illinois"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2013,3,16]]},"reference":[{"key":"e_1_3_2_1_1_1","first-page":"1","volume-title":"IPDPS 2008. IEEE International Symposium on, april","author":"Jetley P.","year":"2008","unstructured":"P. Jetley , F. Gioachin , C. Mendes , L. Kale , and T. Quinn , \" Massively parallel cosmological simulations with changa,\" in Parallel and Distributed Processing, 2008 . IPDPS 2008. IEEE International Symposium on, april 2008 , pp. 1 -- 12 . P. Jetley, F. Gioachin, C. Mendes, L. Kale, and T. Quinn, \"Massively parallel cosmological simulations with changa,\" in Parallel and Distributed Processing, 2008. IPDPS 2008. IEEE International Symposium on, april 2008, pp. 1--12."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/1654059.1654078"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1111\/j.1467-8659.2009.01377.x"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/1132960.1132964"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/1531666.1531668"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/1454115.1454152"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2011.102"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2008.917757"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jpdc.2008.05.012"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2009.5161005"},{"key":"e_1_3_2_1_11_1","first-page":"1","volume-title":"2010 IEEE International Symposium on, april","author":"Ye X.","year":"2010","unstructured":"X. Ye , D. Fan , W. Lin , N. Yuan , and P. Ienne , \" High performance comparison-based sorting algorithm on many-core gpus,\" in Parallel Distributed Processing (IPDPS) , 2010 IEEE International Symposium on, april 2010 , pp. 1 -- 10 . X. Ye, D. Fan, W. Lin, N. Yuan, and P. Ienne, \"High performance comparison-based sorting algorithm on many-core gpus,\" in Parallel Distributed Processing (IPDPS), 2010 IEEE International Symposium on, april 2010, pp. 1--10."},{"key":"e_1_3_2_1_12_1","first-page":"1","volume-title":"2010 IEEE International Symposium on, april","author":"Leischner N.","year":"2010","unstructured":"N. Leischner , V. Osipov , and P. Sanders , \" Gpu sample sort,\" in Parallel Distributed Processing (IPDPS) , 2010 IEEE International Symposium on, april 2010 , pp. 1 -- 10 . N. Leischner, V. Osipov, and P. Sanders, \"Gpu sample sort,\" in Parallel Distributed Processing (IPDPS), 2010 IEEE International Symposium on, april 2010, pp. 1--10."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/1498698.1564500"},{"key":"e_1_3_2_1_14_1","volume-title":"Gpumemsort: A high performance graphic co-processors sorting algorithm for large scale in-memory data","author":"Ye Y.","year":"2010","unstructured":"Y. Ye , Z. Du , and D. A. Bader , \" Gpumemsort: A high performance graphic co-processors sorting algorithm for large scale in-memory data ,\" 2010 . Y. Ye, Z. Du, and D. A. Bader, \"Gpumemsort: A high performance graphic co-processors sorting algorithm for large scale in-memory data,\" 2010."},{"key":"e_1_3_2_1_15_1","first-page":"1","volume-title":"2011 IEEE International Symposium on","volume":"0","author":"Peters H.","year":"2010","unstructured":"H. Peters , O. Schulz-Hildebrandt , and N. Luttenberger , \" Parallel external sorting for cuda-enabled gpus with load balancing and low transfer overhead,\" Parallel and Distributed Processing Workshops and PhD Forum , 2011 IEEE International Symposium on , vol. 0 , pp. 1 -- 8 , 2010 . H. Peters, O. Schulz-Hildebrandt, and N. Luttenberger, \"Parallel external sorting for cuda-enabled gpus with load balancing and low transfer overhead,\" Parallel and Distributed Processing Workshops and PhD Forum, 2011 IEEE International Symposium on, vol. 0, pp. 1--8, 2010."},{"key":"e_1_3_2_1_16_1","volume-title":"CUDA C Programming Guide","author":"NVIDIA","year":"2012","unstructured":"NVIDIA , CUDA C Programming Guide , 2012 . NVIDIA, CUDA C Programming Guide, 2012."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1093\/comjnl\/5.1.10"},{"key":"e_1_3_2_1_18_1","volume-title":"The Art of Computer Programming","author":"Knuth D.","year":"1998","unstructured":"D. Knuth , The Art of Computer Programming . Addison-Wesley , 1998 , vol. 3 Sorting and Searching, noted by the author in p. 158. D. Knuth, The Art of Computer Programming. Addison-Wesley, 1998, vol. 3 Sorting and Searching, noted by the author in p. 158."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/1735688.1735706"},{"key":"e_1_3_2_1_20_1","unstructured":"\"Amd fusion whitepaper.\" {Online}. Available: http:\/\/sites.amd.com\/us\/Documents\/AMD_fusion_Whitepaper.pdf  \"Amd fusion whitepaper.\" {Online}. Available: http:\/\/sites.amd.com\/us\/Documents\/AMD_fusion_Whitepaper.pdf"},{"key":"e_1_3_2_1_21_1","unstructured":"\"Intel quickpath interconnect \" 2012. {Online}. Available: http:\/\/www.intel.com\/technology\/quickpath  \"Intel quickpath interconnect \" 2012. {Online}. Available: http:\/\/www.intel.com\/technology\/quickpath"},{"key":"e_1_3_2_1_22_1","unstructured":"\"Hypertransport interconnect \" 2012. {Online}. Available: http:\/\/www.hypertransport.org  \"Hypertransport interconnect \" 2012. {Online}. Available: http:\/\/www.hypertransport.org"},{"key":"e_1_3_2_1_23_1","unstructured":"\"Heterogeneous system architecture \" 2012. {Online}. Available: www.hsafoundation.com  \"Heterogeneous system architecture \" 2012. {Online}. Available: www.hsafoundation.com"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/1736020.1736059"},{"key":"e_1_3_2_1_25_1","first-page":"1","volume-title":"IEEE CLUSTER '09","author":"Kindratenko V.","year":"2009","unstructured":"V. Kindratenko , J. Enos , G. Shi , M. Showerman , G. Arnold , J. Stone , J. Phillips , and W. mei Hwu , \"Gpu clusters for high-performance computing,\" in Workshop on Parallel Programming on Accelerator Clusters . IEEE CLUSTER '09 , 31 2009 -sept. 4 2009, pp. 1 -- 8 . V. Kindratenko, J. Enos, G. Shi, M. Showerman, G. Arnold, J. Stone, J. Phillips, and W. mei Hwu, \"Gpu clusters for high-performance computing,\" in Workshop on Parallel Programming on Accelerator Clusters. IEEE CLUSTER '09, 31 2009-sept. 4 2009, pp. 1--8."},{"key":"e_1_3_2_1_26_1","unstructured":"NVIDIA \"Thrust parallel algorithms library \" 2012. {Online}. Available: http:\/\/thrust.github.com  NVIDIA \"Thrust parallel algorithms library \" 2012. {Online}. Available: http:\/\/thrust.github.com"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1016\/0020-0190(89)90138-5"},{"key":"e_1_3_2_1_28_1","first-page":"682","volume-title":"A.-M","author":"Singler J.","year":"2007","unstructured":"J. Singler , P. Sanders , and F. Putze , \" Mcstl: The multi-core standard template library,\" in Euro-Par 2007 Parallel Processing, ser. Lecture Notes in Computer Science , A.-M . Kermarrec, L. Boug\u00e9, and T. Priol, Eds. Springer Berlin\/Heidelberg , 2007 , vol. 4641 , pp. 682 -- 694 . J. Singler, P. Sanders, and F. Putze, \"Mcstl: The multi-core standard template library,\" in Euro-Par 2007 Parallel Processing, ser. Lecture Notes in Computer Science, A.-M. Kermarrec, L. Boug\u00e9, and T. Priol, Eds. Springer Berlin\/Heidelberg, 2007, vol. 4641, pp. 682--694."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1002\/(SICI)1097-024X(199708)27:8%3C983::AID-SPE117%3E3.0.CO;2-#"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/1468075.1468121"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/SFCS.1986.41"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1007\/s002240000083"},{"key":"e_1_3_2_1_33_1","first-page":"1","volume-title":"2010 IEEE International Symposium on, april","author":"Solomonik E.","year":"2010","unstructured":"E. Solomonik and L. Kale , \" Highly scalable parallel sorting,\" in Parallel Distributed Processing (IPDPS) , 2010 IEEE International Symposium on, april 2010 , pp. 1 -- 12 . E. Solomonik and L. Kale, \"Highly scalable parallel sorting,\" in Parallel Distributed Processing (IPDPS), 2010 IEEE International Symposium on, april 2010, pp. 1--12."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/1807167.1807207"},{"key":"e_1_3_2_1_35_1","volume-title":"May","author":"Davidson A.","year":"2012","unstructured":"A. Davidson , D. Tarjan , M. Garland , and J. D. Owens , \" Efficient parallel merge sort for fixed and variable length keys,\" in Proceedings of Innovative Parallel Computing (InPar '12) , May 2012 . A. Davidson, D. Tarjan, M. Garland, and J. D. Owens, \"Efficient parallel merge sort for fixed and variable length keys,\" in Proceedings of Innovative Parallel Computing (InPar '12), May 2012."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/2304576.2304621"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/297096.297128"}],"event":{"name":"GPGPU-6: Sixth Workshop on General Purpose Processing Using GPUs","sponsor":["SIGARCH ACM Special Interest Group on Computer Architecture"],"location":"Houston Texas USA","acronym":"GPGPU-6"},"container-title":["Proceedings of the 6th Workshop on General Purpose Processor Using Graphics Processing Units"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/2458523.2458524","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/2458523.2458524","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T08:18:37Z","timestamp":1750234717000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/2458523.2458524"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2013,3,16]]},"references-count":37,"alternative-id":["10.1145\/2458523.2458524","10.1145\/2458523"],"URL":"https:\/\/doi.org\/10.1145\/2458523.2458524","relation":{},"subject":[],"published":{"date-parts":[[2013,3,16]]},"assertion":[{"value":"2013-03-16","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}