{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,19]],"date-time":"2025-03-19T11:51:00Z","timestamp":1742385060887},"publisher-location":"Berlin, Heidelberg","reference-count":23,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"type":"print","value":"9783642115141"},{"type":"electronic","value":"9783642115158"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2010]]},"DOI":"10.1007\/978-3-642-11515-8_4","type":"book-chapter","created":{"date-parts":[[2010,1,20]],"date-time":"2010-01-20T14:58:47Z","timestamp":1263999527000},"page":"18-34","source":"Crossref","is-referenced-by-count":23,"title":["Low-Overhead, High-Speed Multi-core Barrier Synchronization"],"prefix":"10.1007","author":[{"given":"John","family":"Sartori","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rakesh","family":"Kumar","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","reference":[{"issue":"6","key":"4_CR1","doi-asserted-by":"publisher","first-page":"591","DOI":"10.1109\/71.388040","volume":"6","author":"S. Shang","year":"1995","unstructured":"Shang, S., Hwang, K.: Distributed hardwired barrier synchronization for scalable multiprocessor clusters. IEEE Trans. Parallel Distrib. Syst.\u00a06(6), 591\u2013605 (1995)","journal-title":"IEEE Trans. Parallel Distrib. Syst."},{"key":"4_CR2","unstructured":"Hoefler, T.: A survey of barrier algorithms for coarse grained supercomputers. Chemnitzer Informatik-Berichte (2004)"},{"key":"4_CR3","doi-asserted-by":"crossref","unstructured":"Alm\u00e1si, G., et al.: Optimization of MPI collective communication on Bluegene\/L systems. In: ICS 2005, pp. 253\u2013262 (2005)","DOI":"10.1145\/1088149.1088183"},{"issue":"2","key":"4_CR4","doi-asserted-by":"publisher","first-page":"333","DOI":"10.1006\/jpdc.1999.1556","volume":"58","author":"V. Ramakrishnan","year":"1999","unstructured":"Ramakrishnan, V., Scherson, I.D.: Efficient techniques for nested and disjoint barrier synchronization. J. Parallel Distrib. Comput.\u00a058(2), 333\u2013356 (1999)","journal-title":"J. Parallel Distrib. Comput."},{"key":"4_CR5","doi-asserted-by":"crossref","unstructured":"Chen, J., Watson, W.: Software barrier performance on dual quad-core Opterons. In: NAS 2008, pp. 303\u2013309 (2008)","DOI":"10.1109\/NAS.2008.27"},{"key":"4_CR6","doi-asserted-by":"crossref","unstructured":"Nikolopoulos, D., Papatheodorou, T.: Fast synchronization on scalable cache-coherent multiprocessors using hybrid primitives. In: IPDPS 2000, p. 711 (2000)","DOI":"10.1109\/IPDPS.2000.846056"},{"key":"4_CR7","unstructured":"Lee, J.B., Jhon, C.S.: Reducing coherence overhead of barrier synchronization in software DSMs. In: ICS 1998, pp. 1\u201318 (1998)"},{"issue":"1","key":"4_CR8","doi-asserted-by":"publisher","first-page":"21","DOI":"10.1145\/103727.103729","volume":"9","author":"J.M. Mellor-Crummey","year":"1991","unstructured":"Mellor-Crummey, J.M., Scott, M.L.: Algorithms for scalable synchronization on shared-memory multiprocessors. ACM Trans. Comput. Syst.\u00a09(1), 21\u201365 (1991)","journal-title":"ACM Trans. Comput. Syst."},{"issue":"2-3","key":"4_CR9","doi-asserted-by":"publisher","first-page":"213","DOI":"10.1147\/rd.492.0213","volume":"49","author":"P. Coteus","year":"2005","unstructured":"Coteus, P., et al.: Packaging the BlueGene\/L supercomputer. IBM Journal of Research and Development\u00a049(2-3), 213\u2013248 (2005)","journal-title":"IBM Journal of Research and Development"},{"key":"4_CR10","unstructured":"Adams, D.: Cray T3D system architecture overview manual (1993), ftp:\/\/ftp.cray.com\/product-info\/mpp\/T3D_Architecture_Over\/T3D.overview.html"},{"key":"4_CR11","unstructured":"Freudenthal, E., Peze, O.: Efficient synchronization algorithms using fetch-and-add on multiple bitfield integers. Ultracomputer Note 148 (1988)"},{"key":"4_CR12","doi-asserted-by":"crossref","unstructured":"Beckmann, C., Polychronopoulos, C.: Fast barrier synchronization hardware. In: ICS 1990, pp. 180\u2013189 (1990)","DOI":"10.1109\/SUPERC.1990.130019"},{"key":"4_CR13","unstructured":"Biswas, R.: NAS parallel benchmarks (2009), http:\/\/www.nas.nasa.gov"},{"key":"4_CR14","doi-asserted-by":"crossref","unstructured":"Kumar, R., Zyuban, V., Tullsen, D.: Interconnections in multi-core architectures: Understanding mechanisms, overheads, and scaling. In: ISCA 2005 (2005)","DOI":"10.1145\/1080695.1070004"},{"key":"4_CR15","doi-asserted-by":"publisher","first-page":"120","DOI":"10.1016\/j.orl.2004.05.005","volume":"33","author":"E. Althaus","year":"2005","unstructured":"Althaus, E., Funke, S., Har-peled, S., Knemann, J.: Approximating k-hop minimum-spanning trees. Operations Research Letters\u00a033, 120 (2005)","journal-title":"Operations Research Letters"},{"issue":"2","key":"4_CR16","doi-asserted-by":"publisher","first-page":"150","DOI":"10.1145\/1273440.1250681","volume":"35","author":"A. Kumar","year":"2007","unstructured":"Kumar, A., et al.: Express virtual channels: Towards the ideal interconnection fabric. SIGARCH Comput. Archit. News\u00a035(2), 150\u2013161 (2007)","journal-title":"SIGARCH Comput. Archit. News"},{"issue":"4","key":"4_CR17","first-page":"52","volume":"26","author":"N.L. Binkert","year":"2006","unstructured":"Binkert, N.L., et al.: The M5 simulator: Modeling networked systems. MICRO\u00a026(4), 52\u201360 (2006)","journal-title":"MICRO"},{"key":"4_CR18","first-page":"235","volume":"39","author":"J. Sampson","year":"2006","unstructured":"Sampson, J., et al.: Exploiting fine-grained data parallelism with chip multiprocessors and fast barriers. MICRO\u00a039, 235\u2013246 (2006)","journal-title":"MICRO"},{"key":"4_CR19","unstructured":"McMahon, F.: Livermore loops coded in C (1992), http:\/\/www.netlib.org\/benchmark\/livermorec"},{"key":"4_CR20","unstructured":"E.M.B. Consortium: EEMBC (2009), http:\/\/www.eembc.org"},{"key":"4_CR21","doi-asserted-by":"crossref","unstructured":"Zhu, W., et al.: Synchronization state buffer: Supporting efficient fine-grain synchronization on many-core architectures. In: ISCA 2007, pp. 35\u201345 (2007)","DOI":"10.1145\/1273440.1250668"},{"key":"4_CR22","doi-asserted-by":"crossref","unstructured":"Villa, O., Palermo, G., Silvano, C.: Efficiency and scalability of barrier synchronization on NOC based many-core architectures. In: CASES 2008, pp. 81\u201390 (2008)","DOI":"10.1145\/1450095.1450110"},{"issue":"5","key":"4_CR23","doi-asserted-by":"publisher","first-page":"26","DOI":"10.1145\/248208.237144","volume":"30","author":"S.L. Scott","year":"1996","unstructured":"Scott, S.L.: Synchronization and communication in the T3E multiprocessor. SIGOPS Oper. Syst. Rev.\u00a030(5), 26\u201336 (1996)","journal-title":"SIGOPS Oper. Syst. Rev."}],"container-title":["Lecture Notes in Computer Science","High Performance Embedded Architectures and Compilers"],"original-title":[],"link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-642-11515-8_4.pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,5,29]],"date-time":"2023-05-29T17:53:27Z","timestamp":1685382807000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-642-11515-8_4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2010]]},"ISBN":["9783642115141","9783642115158"],"references-count":23,"URL":"https:\/\/doi.org\/10.1007\/978-3-642-11515-8_4","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2010]]}}}