{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T15:55:21Z","timestamp":1784303721692,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":32,"publisher":"ACM","license":[{"start":{"date-parts":[[2015,6,8]],"date-time":"2015-06-08T00:00:00Z","timestamp":1433721600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Singapore Ministry of Education Academic Research Fund (MoE-AcRF) Tier 1","award":["R263-000-B02-112"],"award-info":[{"award-number":["R263-000-B02-112"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2015,6,8]]},"DOI":"10.1145\/2751205.2751232","type":"proceedings-article","created":{"date-parts":[[2015,6,2]],"date-time":"2015-06-02T18:40:11Z","timestamp":1433270411000},"page":"109-118","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":39,"title":["Fine-Grained Synchronizations and Dataflow Programming on GPUs"],"prefix":"10.1145","author":[{"given":"Ang","family":"Li","sequence":"first","affiliation":[{"name":"Eindhoven University of Technology, Eindhoven, Netherlands"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Gert-Jan","family":"van den Braak","sequence":"additional","affiliation":[{"name":"Eindhoven University of Technology, Eindhoven, Netherlands"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Henk","family":"Corporaal","sequence":"additional","affiliation":[{"name":"Eindhoven University of Technology, Eindhoven, Netherlands"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Akash","family":"Kumar","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2015,6,8]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Pearson Education","author":"Taubenfeld Gadi","year":"2006"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/71.80120"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/12.40843"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/325096.325150"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/1273440.1250668"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.5555\/520549.822786"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICPP.1994.96"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/1594835.1504207"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/TC.1987.5009499"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/329466.329486"},{"key":"e_1_3_2_1_11_1","volume-title":"GPU Computing Gems Emerald Edition","author":"Wen-Mei Hwu","year":"2011"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2008.917757"},{"key":"e_1_3_2_1_13_1","unstructured":"NVIDIA. CUDA C Programming Guide. http:\/\/docs.nvidia.com\/cuda\/cuda-c-programming-guide\/.  NVIDIA. CUDA C Programming Guide. http:\/\/docs.nvidia.com\/cuda\/cuda-c-programming-guide\/."},{"key":"e_1_3_2_1_14_1","unstructured":"Nikolaj Leischner Vitaly Osipov and Peter Sanders. Fermi architecture white paper. http:\/\/www.nvidia.com\/content\/pdf\/fermi_white_papers\/nvidia_fermi_compute_architecture_whitepaper.pdf.  Nikolaj Leischner Vitaly Osipov and Peter Sanders. Fermi architecture white paper. http:\/\/www.nvidia.com\/content\/pdf\/fermi_white_papers\/nvidia_fermi_compute_architecture_whitepaper.pdf."},{"key":"e_1_3_2_1_15_1","first-page":"1","volume-title":"IEEE International Symposium on","author":"Xiao Shucai","year":"2010"},{"key":"e_1_3_2_1_16_1","volume-title":"Efficient synchronization primitives for GPUs. arXiv preprint arXiv:1110.4623","author":"Stuart Jeff A.","year":"2011"},{"key":"e_1_3_2_1_17_1","unstructured":"NVIDIA. PTX: Parallel Thread Execution ISA Version 4.0. http:\/\/docs.nvidia.com\/cuda\/parallel-thread-execution\/index.html.  NVIDIA. PTX: Parallel Thread Execution ISA Version 4.0. http:\/\/docs.nvidia.com\/cuda\/parallel-thread-execution\/index.html."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/2063384.2063400"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/2555243.2555258"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/2458523.2458533"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2011.80"},{"key":"e_1_3_2_1_22_1","volume-title":"Computer Science","author":"Ashwin","year":"2008"},{"key":"e_1_3_2_1_23_1","volume-title":"November 8","author":"Coon Brett W.","year":"2011"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2012.319"},{"key":"e_1_3_2_1_25_1","volume-title":"CUDA by example: an introduction to general-purpose GPU programming","author":"Sanders Jason","year":"2010"},{"key":"e_1_3_2_1_26_1","unstructured":"Yunqing Hou. Asfermi: An assembler for the NVIDIA Fermi instruction set. http:\/\/code.google.com\/p\/asfermi\/ 2011.  Yunqing Hou. Asfermi: An assembler for the NVIDIA Fermi instruction set. http:\/\/code.google.com\/p\/asfermi\/ 2011."},{"key":"e_1_3_2_1_27_1","first-page":"249","volume-title":"Frank Dehne, and Siang W. Song. A parallel wavefront algorithm for efficient biological sequence comparison. In Computational Science and Its Applications","author":"Alves Carlos E. R.","year":"2003"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/IEMBS.2004.1403804"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC.2009.5306797"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/360827.360844"},{"key":"e_1_3_2_1_31_1","volume-title":"Parallelising wavefront applications on general-purpose GPU devices","author":"Pennycook Simon J.","year":"2010"},{"key":"e_1_3_2_1_32_1","volume-title":"Efficient irregular wavefront propagation algorithms on hybrid CPU-GPU machines. Parallel computing, 39(4)","author":"Teodoro George","year":"2013"}],"event":{"name":"ICS'15: 2015 International Conference on Supercomputing","location":"Newport Beach California USA","acronym":"ICS'15","sponsor":["SIGARCH ACM Special Interest Group on Computer Architecture"]},"container-title":["Proceedings of the 29th ACM on International Conference on Supercomputing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/2751205.2751232","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/2751205.2751232","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T05:43:06Z","timestamp":1750225386000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/2751205.2751232"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2015,6,8]]},"references-count":32,"alternative-id":["10.1145\/2751205.2751232","10.1145\/2751205"],"URL":"https:\/\/doi.org\/10.1145\/2751205.2751232","relation":{},"subject":[],"published":{"date-parts":[[2015,6,8]]},"assertion":[{"value":"2015-06-08","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}