{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,18]],"date-time":"2026-08-18T01:40:41Z","timestamp":1787017241340,"version":"build-2736575974"},"publisher-location":"New York, NY, USA","reference-count":27,"publisher":"ACM","license":[{"start":{"date-parts":[[2013,2,23]],"date-time":"2013-02-23T00:00:00Z","timestamp":1361577600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2013,2,23]]},"DOI":"10.1145\/2442516.2442539","type":"proceedings-article","created":{"date-parts":[[2013,2,26]],"date-time":"2013-02-26T10:23:04Z","timestamp":1361874184000},"page":"229-238","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":58,"title":["StreamScan"],"prefix":"10.1145","author":[{"given":"Shengen","family":"Yan","sequence":"first","affiliation":[{"name":"Institute of Software, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Guoping","family":"Long","sequence":"additional","affiliation":[{"name":"Institute of Software, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yunquan","family":"Zhang","sequence":"additional","affiliation":[{"name":"Institute of Software, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2013,2,23]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/12.42122"},{"key":"e_1_3_2_1_2_1","volume-title":"Synthesis of Parallel Algorithms","author":"E.","year":"1990","unstructured":"Blelloch, G. E. Prefix Sums and Their Applications . Synthesis of Parallel Algorithms , 1990 . Blelloch, G.E. Prefix Sums and Their Applications. Synthesis of Parallel Algorithms, 1990."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/1460833.1460872"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/TC.1973.5009159"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/TC.1982.1675982"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/1556444.1556447"},{"key":"e_1_3_2_1_8_1","author":"Merrill D.","year":"2011","unstructured":"D. Merrill and A. Grimshaw . High Performance and Scalable Radix Sorting: A case study of implementing dynamic parallelism for GPU computing. Parallel Processing Letters , 2011 . D. Merrill and A. Grimshaw. High Performance and Scalable Radix Sorting: A case study of implementing dynamic parallelism for GPU computing. Parallel Processing Letters, 2011.","journal-title":"Parallel Processing Letters"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/1498698.1564500"},{"key":"e_1_3_2_1_10_1","volume-title":"PPAM 09: Proceedings of the International Conference on Parallel Processing and Applied Mathematics","author":"Peters H.","year":"2009","unstructured":"H. Peters , O. Schulz-Hildebrandt , and N. Luttenberger . Fast in-place sorting with cuda based on bitonic sort . PPAM 09: Proceedings of the International Conference on Parallel Processing and Applied Mathematics , 2009 . H. Peters, O. Schulz-Hildebrandt, and N. Luttenberger. Fast in-place sorting with cuda based on bitonic sort. PPAM 09: Proceedings of the International Conference on Parallel Processing and Applied Mathematics, 2009."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2009.5161005"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/2145816.2145832"},{"key":"e_1_3_2_1_13_1","volume-title":"University of Virginia","author":"Merrill D.","year":"2011","unstructured":"D. Merrill . Allocation-oriented Algorithm Design with Application to GPU Computing. PhD thesis , University of Virginia , 2011 . D. Merrill. Allocation-oriented Algorithm Design with Application to GPU Computing. PhD thesis, University of Virginia, 2011."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2011.174"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/1572769.1572795"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/1572769.1572796"},{"key":"e_1_3_2_1_17_1","volume-title":"Parallel & Distributed Processing (IPDPS)","author":"Wei Zheng","year":"2010","unstructured":"Zheng Wei , Joseph JaJa . Optimization of linked list prefix computations on multithreaded GPUs using CUDA . In Parallel & Distributed Processing (IPDPS) , 2010 . Zheng Wei, Joseph JaJa. Optimization of linked list prefix computations on multithreaded GPUs using CUDA. In Parallel & Distributed Processing (IPDPS), 2010."},{"key":"e_1_3_2_1_19_1","unstructured":"M. Harris J. Owens S. Sengupta Y. Zhang and A. Davidson. CUDPP: CUDA Data Parallel Primitives Library. http:\/\/gpgpu.org\/developer\/cudpp.  M. Harris J. Owens S. Sengupta Y. Zhang and A. Davidson. CUDPP: CUDA Data Parallel Primitives Library. http:\/\/gpgpu.org\/developer\/cudpp."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/1375527.1375559"},{"key":"e_1_3_2_1_22_1","unstructured":"NVIDIA Corporation. Nvidia Cuda C Programming Guide. http:\/\/developer.download.nvidia.com\/compute\/DevZone\/docs\/html\/C\/doc\/CUDA_C_Programming_Guide.pdf 2012.  NVIDIA Corporation. Nvidia Cuda C Programming Guide. http:\/\/developer.download.nvidia.com\/compute\/DevZone\/docs\/html\/C\/doc\/CUDA_C_Programming_Guide.pdf 2012."},{"key":"e_1_3_2_1_23_1","volume-title":"NVIDIA Tech. Rep","author":"Sengupta S.","year":"2008","unstructured":"S. Sengupta , M. Harris , and M. Garland . Efficient Parallel Scan Algorithms for GPUs . NVIDIA Tech. Rep , 2008 . S. Sengupta, M. Harris, and M. Garland. Efficient Parallel Scan Algorithms for GPUs. NVIDIA Tech. Rep, 2008."},{"key":"e_1_3_2_1_24_1","series-title":"Lecture Notes in Computer Science","volume-title":"Euro-Par 2010 Work-shop Proceedings","author":"Breitbart Jens","year":"2010","unstructured":"Jens Breitbart . Static GPU threads and an Improved Scan Algorithm . In Euro-Par 2010 Work-shop Proceedings , Lecture Notes in Computer Science , 2010 . Jens Breitbart. Static GPU threads and an Improved Scan Algorithm. In Euro-Par 2010 Work-shop Proceedings, Lecture Notes in Computer Science, 2010."},{"key":"e_1_3_2_1_25_1","volume-title":"Proceedings of the Workshop on Edge Computing Using New Commodity Architectures","author":"Owens Sengupta S., Lefohn","year":"2006","unstructured":"Sengupta S., Lefohn A. and Owens , J . A work-efficient step-efficient prefix-sum algorithm . Proceedings of the Workshop on Edge Computing Using New Commodity Architectures , 2006 . Sengupta S., Lefohn A. and Owens, J. A work-efficient step-efficient prefix-sum algorithm. Proceedings of the Workshop on Edge Computing Using New Commodity Architectures, 2006."},{"key":"e_1_3_2_1_26_1","volume-title":"Parallel & Distributed Processing (IPDPS)","author":"Xiao S.","year":"2010","unstructured":"S. Xiao and W. chun Feng . Inter-block GPU Communication via Fast Barrier Synchronization . In Parallel & Distributed Processing (IPDPS) , 2010 . S. Xiao and W. chun Feng. Inter-block GPU Communication via Fast Barrier Synchronization. In Parallel & Distributed Processing (IPDPS), 2010."},{"key":"e_1_3_2_1_27_1","first-page":"573","volume-title":"GPU Gems 2","author":"Horn","year":"2005","unstructured":"Horn D. Stream Reduction Operations for GPGPU Applications . In GPU Gems 2 , Pharr M., (Ed.). Addison Wesley , ch. 36, pp. 573 -- 589 , 2005 . Horn D. Stream Reduction Operations for GPGPU Applications. In GPU Gems 2, Pharr M., (Ed.).Addison Wesley, ch. 36, pp. 573--589, 2005."},{"key":"e_1_3_2_1_28_1","volume-title":"GPU Gems 3","author":"Harris M., Sengupta","year":"2007","unstructured":"Harris M., Sengupta S. and Owens J. D . Parallel Prefix sum (scan) with CUDA . In GPU Gems 3 , Nguyen H., (Ed.). Addison Wesley , ch. 39, 2007 . Harris M., Sengupta S. and Owens J. D. Parallel Prefix sum (scan) with CUDA. In GPU Gems 3, Nguyen H., (Ed.).Addison Wesley, ch. 39, 2007."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-32820-6_90"},{"key":"e_1_3_2_1_30_1","volume-title":"The OpenCL Specification Version: 1.2","author":"Khronos OpenCL Working Group","year":"2012","unstructured":"Khronos OpenCL Working Group , The OpenCL Specification Version: 1.2 , 2012 . Khronos OpenCL Working Group, The OpenCL Specification Version: 1.2, 2012."}],"event":{"name":"PPoPP '13: ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming","location":"Shenzhen China","acronym":"PPoPP '13","sponsor":["SIGPLAN ACM Special Interest Group on Programming Languages"]},"container-title":["Proceedings of the 18th ACM SIGPLAN symposium on Principles and practice of parallel programming"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/2442516.2442539","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/2442516.2442539","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T04:19:07Z","timestamp":1750220347000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/2442516.2442539"}},"subtitle":["fast scan algorithms for GPUs without global barrier synchronization"],"short-title":[],"issued":{"date-parts":[[2013,2,23]]},"references-count":27,"alternative-id":["10.1145\/2442516.2442539","10.1145\/2442516"],"URL":"https:\/\/doi.org\/10.1145\/2442516.2442539","relation":{"is-identical-to":[{"id-type":"doi","id":"10.1145\/2517327.2442539","asserted-by":"object"}]},"subject":[],"published":{"date-parts":[[2013,2,23]]},"assertion":[{"value":"2013-02-23","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}