{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T14:59:05Z","timestamp":1784905145840,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":34,"publisher":"ACM","license":[{"start":{"date-parts":[[2018,6,12]],"date-time":"2018-06-12T00:00:00Z","timestamp":1528761600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100001659","name":"Deutsche Forschungsgemeinschaft","doi-asserted-by":"publisher","award":["STE 2565\/1-1"],"award-info":[{"award-number":["STE 2565\/1-1"]}],"id":[{"id":"10.13039\/501100001659","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100002428","name":"Austrian Science Fund","doi-asserted-by":"publisher","award":["I3007"],"award-info":[{"award-number":["I3007"]}],"id":[{"id":"10.13039\/501100002428","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2018,6,12]]},"DOI":"10.1145\/3205289.3205291","type":"proceedings-article","created":{"date-parts":[[2018,9,13]],"date-time":"2018-09-13T12:54:52Z","timestamp":1536843292000},"page":"76-85","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":19,"title":["The Broker Queue"],"prefix":"10.1145","author":[{"given":"Bernhard","family":"Kerbl","sequence":"first","affiliation":[{"name":"Graz University of Technology"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Michael","family":"Kenzel","sequence":"additional","affiliation":[{"name":"Graz University of Technology"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Joerg H.","family":"Mueller","sequence":"additional","affiliation":[{"name":"Graz University of Technology"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dieter","family":"Schmalstieg","sequence":"additional","affiliation":[{"name":"Graz University of Technology"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Markus","family":"Steinberger","sequence":"additional","affiliation":[{"name":"Graz University of Technology"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2018,6,12]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/1572769.1572792"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/277651.277678"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"crossref","unstructured":"Guy E. Blelloch Perry Cheng Phillip B. Gibbons and P. B. Gibbons. 2003. Theory of Computing Systems Scalable Room Synchronizations.  Guy E. Blelloch Perry Cheng Phillip B. Gibbons and P. B. Gibbons. 2003. Theory of Computing Systems Scalable Room Synchronizations.","DOI":"10.1007\/s00224-003-1081-y"},{"key":"e_1_3_2_1_4_1","volume-title":"Proc. ACM SIGGRAPH\/EUROGRAPHICS symposium on Graphics hardware (GH '08)","author":"Cederman Daniel","year":"2008","unstructured":"Daniel Cederman and Philippas Tsigas . 2008 . On dynamic load balancing on graphics processors . In Proc. ACM SIGGRAPH\/EUROGRAPHICS symposium on Graphics hardware (GH '08) . Aire-la-Ville, Switzerland, Switzerland, 57--64. Daniel Cederman and Philippas Tsigas. 2008. On dynamic load balancing on graphics processors. In Proc. ACM SIGGRAPH\/EUROGRAPHICS symposium on Graphics hardware (GH '08). Aire-la-Ville, Switzerland, Switzerland, 57--64."},{"key":"e_1_3_2_1_5_1","volume-title":"Proc. Workshop on Languages and Compilers for Parallel Computing (LCPC '11)","author":"Chatterjee Sanjay","year":"2011","unstructured":"Sanjay Chatterjee , Max Grossman , Alina Sbirlea , and Vivek Sarkar . 2011 . Dynamic Task Parallelism with a GPU Work-Stealing Runtime System . In Proc. Workshop on Languages and Compilers for Parallel Computing (LCPC '11) . Sanjay Chatterjee, Max Grossman, Alina Sbirlea, and Vivek Sarkar. 2011. Dynamic Task Parallelism with a GPU Work-Stealing Runtime System. In Proc. Workshop on Languages and Compilers for Parallel Computing (LCPC '11)."},{"key":"e_1_3_2_1_6_1","volume-title":"Parallel Distributed Processing (IPDPS), 2010 IEEE International Symposium on. 1--12","author":"Chen Long","unstructured":"Long Chen , O. Villa , S. Krishnamoorthy , and G.R. Gao . 2010. Dynamic load balancing on single- and multi-GPU systems . In Parallel Distributed Processing (IPDPS), 2010 IEEE International Symposium on. 1--12 . Long Chen, O. Villa, S. Krishnamoorthy, and G.R. Gao. 2010. Dynamic load balancing on single- and multi-GPU systems. In Parallel Distributed Processing (IPDPS), 2010 IEEE International Symposium on. 1--12."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICECCS.2005.49"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/1345206.1345215"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/69624.357206"},{"key":"e_1_3_2_1_10_1","volume-title":"Maxwell: The most advanced CUDA GPU ever made.","author":"Harris Mark","year":"2014","unstructured":"Mark Harris . 2014 . Maxwell: The most advanced CUDA GPU ever made. (2014). Mark Harris. 2014. Maxwell: The most advanced CUDA GPU ever made. (2014)."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/1810479.1810540"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1007\/s00446-005-0144-5"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.5555\/850929.851942"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/78969.78972"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.5555\/1782394.1782423"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/319566.319567"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/69624.357207"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/1882486.1882508"},{"key":"e_1_3_2_1_19_1","unstructured":"Jure Leskovec and Andrej Krevl. 2014. SNAP Datasets: Stanford Large Network Dataset Collection. http:\/\/snap.stanford.edu\/data. (June 2014).  Jure Leskovec and Andrej Krevl. 2014. SNAP Datasets: Stanford Large Network Dataset Collection. http:\/\/snap.stanford.edu\/data. (June 2014)."},{"key":"e_1_3_2_1_20_1","volume-title":"Scott","author":"Michael Maged M.","year":"1995","unstructured":"Maged M. Michael and Michael L . Scott . 1995 . Correction of a Memory Management Method for Lock-Free Data Structures. Technical Report. Rochester, NY, USA. Maged M. Michael and Michael L. Scott. 1995. Correction of a Memory Management Method for Lock-Free Data Structures. Technical Report. Rochester, NY, USA."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/248052.248106"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/2517327.2442527"},{"key":"e_1_3_2_1_23_1","unstructured":"Nvidia. 2017. CUDA Programming guide. (2017).  Nvidia. 2017. CUDA Programming guide. (2017)."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/2086696.2086728"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/2668930.2688048"},{"key":"e_1_3_2_1_26_1","volume-title":"Parallel and Distributed Systems, 2000. Proceedings. Seventh International Conference on. 470--475","author":"Shann Chien-Hua","year":"2000","unstructured":"Chien-Hua Shann , T.-L. Huang , and Cheng Chen . 2000 . A practical nonblocking queue algorithm using compare-and-swap . In Parallel and Distributed Systems, 2000. Proceedings. Seventh International Conference on. 470--475 . Chien-Hua Shann, T.-L. Huang, and Cheng Chen. 2000. A practical nonblocking queue algorithm using compare-and-swap. In Parallel and Distributed Systems, 2000. Proceedings. Seventh International Conference on. 470--475."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/2366145.2366180"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/2661229.2661250"},{"key":"e_1_3_2_1_29_1","first-page":"1","article-title":"ScatterAlloc: Massively parallel dynamic memory allocation for the GPU","volume":"2012","author":"Steinberger Markus","year":"2012","unstructured":"Markus Steinberger , Michael Kenzel , Bernhard Kainz , and Dieter Schmalstieg . 2012 . ScatterAlloc: Massively parallel dynamic memory allocation for the GPU . In Innovative Parallel Computing(InPar) , 2012. 1 -- 10 . Markus Steinberger, Michael Kenzel, Bernhard Kainz, and Dieter Schmalstieg. 2012. ScatterAlloc: Massively parallel dynamic memory allocation for the GPU. In Innovative Parallel Computing(InPar), 2012. 1--10.","journal-title":"Innovative Parallel Computing(InPar)"},{"key":"e_1_3_2_1_30_1","volume-title":"OpenCL: A parallel programming standard for heterogeneous computing systems. Computing in science & engineering 12, 3","author":"Stone John E","year":"2010","unstructured":"John E Stone , David Gohara , and Guochun Shi . 2010. OpenCL: A parallel programming standard for heterogeneous computing systems. Computing in science & engineering 12, 3 ( 2010 ), 66--73. John E Stone, David Gohara, and Guochun Shi. 2010. OpenCL: A parallel programming standard for heterogeneous computing systems. Computing in science & engineering 12, 3 (2010), 66--73."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/378580.378611"},{"key":"e_1_3_2_1_32_1","volume-title":"Proc. High Performance Graphics (HPG '10)","author":"Tzeng Stanley","unstructured":"Stanley Tzeng , Anjul Patney , and John D. Owens . 2010. Task management for irregular-parallel workloads on the GPU . In Proc. High Performance Graphics (HPG '10) . Aire-la-Ville, Switzerland, Switzerland, 29--37. Stanley Tzeng, Anjul Patney, and John D. Owens. 2010. Task management for irregular-parallel workloads on the GPU. In Proc. High Performance Graphics (HPG '10). Aire-la-Ville, Switzerland, Switzerland, 29--37."},{"key":"e_1_3_2_1_33_1","volume-title":"Proc. International Conference on Parallel and Distributed Computing Systems. 64--69","author":"Valois John D.","year":"1994","unstructured":"John D. Valois . 1994 . Implementing Lock-Free Queues . In Proc. International Conference on Parallel and Distributed Computing Systems. 64--69 . John D. Valois. 1994. Implementing Lock-Free Queues. In Proc. International Conference on Parallel and Distributed Computing Systems. 64--69."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/2851141.2851168"}],"event":{"name":"ICS '18: 2018 International Conference on Supercomputing","location":"Beijing China","acronym":"ICS '18","sponsor":["SIGARCH ACM Special Interest Group on Computer Architecture"]},"container-title":["Proceedings of the 2018 International Conference on Supercomputing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3205289.3205291","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3205289.3205291","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T01:08:33Z","timestamp":1750208913000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3205289.3205291"}},"subtitle":["A Fast, Linearizable FIFO Queue for Fine-Granular Work Distribution on the GPU"],"short-title":[],"issued":{"date-parts":[[2018,6,12]]},"references-count":34,"alternative-id":["10.1145\/3205289.3205291","10.1145\/3205289"],"URL":"https:\/\/doi.org\/10.1145\/3205289.3205291","relation":{},"subject":[],"published":{"date-parts":[[2018,6,12]]},"assertion":[{"value":"2018-06-12","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}