{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,27]],"date-time":"2025-03-27T07:17:54Z","timestamp":1743059874749,"version":"3.40.3"},"publisher-location":"New York, NY","reference-count":42,"publisher":"Springer New York","isbn-type":[{"type":"print","value":"9781493920914"},{"type":"electronic","value":"9781493920921"}],"license":[{"start":{"date-parts":[[2015,1,1]],"date-time":"2015-01-01T00:00:00Z","timestamp":1420070400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2015,1,1]],"date-time":"2015-01-01T00:00:00Z","timestamp":1420070400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2015]]},"DOI":"10.1007\/978-1-4939-2092-1_26","type":"book-chapter","created":{"date-parts":[[2015,3,16]],"date-time":"2015-03-16T09:30:47Z","timestamp":1426498247000},"page":"753-803","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Efficient Hardware-Supported Synchronization Mechanisms for Manycores"],"prefix":"10.1007","author":[{"given":"Jos\u00e9 L.","family":"Abell\u00e1n","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Juan","family":"Fern\u00e1ndez","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Manuel E.","family":"Acacio","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2015,3,17]]},"reference":[{"key":"26_CR1","doi-asserted-by":"crossref","unstructured":"A. Flores, J. L. Arag\u00f3n and M. E. Acacio. Sim-PowerCMP: A Detailed Simulator for Energy Consumption Analysis in Future Embedded CMP Architectures. In Proceedings of the 21st International Conference on Advanced Information Networking and Applications Workshops, 2007.","DOI":"10.1109\/AINAW.2007.334"},{"key":"26_CR2","doi-asserted-by":"crossref","unstructured":"A. K\u00e4gi, D. Burger and J. R. Goodman. Efficient Synchronization: Let Them Eat QOLB. In Proceedings of the 24th International on Computer Architecture, 1997.","DOI":"10.1145\/264107.264166"},{"key":"26_CR3","doi-asserted-by":"crossref","unstructured":"A. P. Jose and K. L. Shepard. Distributed Loss-Compensation Techniques for Energy-Efficient Low-Latency On-Chip Communications. IEEE Journal of Solid State Circuits, 42(6):1415\u20131424, 2007.","DOI":"10.1109\/JSSC.2007.897165"},{"key":"26_CR4","doi-asserted-by":"crossref","unstructured":"B-H. Lim and A. Agarwal. Reactive Synchronization Algorithms for Multiprocessors. ACM SIGPLAN Notices, 29(11):25\u201335, 1994.","DOI":"10.1145\/195470.195490"},{"key":"26_CR5","unstructured":"B. M. Beckmann and D. A. Wood. TLC: Transmission Line Caches. In Proceedings of the 36th Annual IEEE\/ACM International Symposium on Microarchitecture, 2011."},{"key":"26_CR6","unstructured":"C-C. Kuo, J. B. Carter and R. Kuramkote. MP-LOCKs: Replacing H\/W Synchronization Primitives with Message Passing. In Proceedings of the 5th International Symposium on High-Performance Computer Architecture, 1999."},{"key":"26_CR7","unstructured":"C. Cascaval, J. G. Casta\u00c3os, L. Ceze, M. Denneau, M. Gupta, D. Lieber, J. E. Moreira, K. Strauss and H. S. Warren. Evaluation of a Multithreaded Architecture for Cellular Computing. In Proceedings of the 8th International Symposium on High-Performance Computer Architecture, 2002."},{"key":"26_CR8","doi-asserted-by":"crossref","unstructured":"C. E. Leiserson, Z. S. Abuhamdeh, D. C. Douglas, C. R. Feynman, M. N. Ganmukhi, J. V. Hill, W. D. Hillis, B. C. Kuszmaul, M. A. St. Pierre, D. S. Wells, M. C. Wong, S. W. Yang and R. Zak. The Network Architecture of the Connection Machine CM-5. In Proceedings of the ACM Symposium on Parallel Algorithms and Architectures, 1992.","DOI":"10.1145\/140901.141883"},{"key":"26_CR9","doi-asserted-by":"crossref","unstructured":"C. Wagner and F. Mueller. Token-based Read\/Write-Locks for Distributed Mutual Exclusion. In Proceedings of the 6th International Euro-Par Conference on Parallel Processing, 2000.","DOI":"10.1007\/3-540-44520-X_167"},{"key":"26_CR10","unstructured":"D. E. Culler, J. P. Singh and A. Gupta. Parallel Computer Architecture: A Hardware\/Software Approach. Morgan Kaufmann, 1998."},{"key":"26_CR11","doi-asserted-by":"crossref","unstructured":"E. Mensink, D. Schinkel, E. Klumperink, E. Tuijl and B. Nauta. A 0.28pf\/b 2gb\/s\/ch Transceiver in 90nm CMOS for 10mm On-Chip Interconnects. In Proceedings of the IEEE Solid-State Circuits Conference, 2007.","DOI":"10.1109\/ISSCC.2007.373470"},{"key":"26_CR12","doi-asserted-by":"crossref","unstructured":"E. W. Dijkstra. Solution of a Problem in Concurrent Programming Control. Communications of the ACM, 8(9):569, 1965.","DOI":"10.1145\/365559.365617"},{"key":"26_CR13","unstructured":"F. H. McMahon. Livermore Fortran Kernels: A Computer Test of Numerical Performance Range. Technical Report UCRL-53745, Lawrence Livermore National Laboratory, 1986. http:\/\/www.netlib.org\/benchmark\/livermorec."},{"key":"26_CR14","unstructured":"H. Franke, R. Russell and M. Kirkwood. Fuss, Futexes and Furwocks: Fast Userlevel Locking in Linux. In Proceedings of the Ottawa Linux Symposium, 2002."},{"key":"26_CR15","doi-asserted-by":"crossref","unstructured":"H. Ito, M. Kimura, K. Miyashita, T. Ishii, K. Okada and K. Masu. A Bidirectional-and Multi-Drop-Transmission-Line Interconnect for Multipoint-to-Multipoint On-Chip Communications. IEEE Journal of Solid State Circuits, 43(4):1020\u20131029, 2008.","DOI":"10.1109\/JSSC.2008.917547"},{"key":"26_CR16","unstructured":"HP Labs. CACTI, 2012. http:\/\/www.hpl.hp.com\/research\/cacti\/."},{"key":"26_CR17","unstructured":"Intel Labs. Single-chip Cloud Computer, 2009. http:\/\/techresearch.intel.com\/ articles\/Tera-Scale\/1826.htm."},{"key":"26_CR18","doi-asserted-by":"crossref","unstructured":"J. Eastep, D. Wingate, M. D. Santambrogio and A. Agarwal. Smartlocks: Self-Aware Synchronization through Lock Acquisition Scheduling. In Proceedings of the 7th IEEE\/ACM International Conference on Autonomic Computing and Communications, 2009.","DOI":"10.1145\/1809049.1809079"},{"key":"26_CR19","doi-asserted-by":"crossref","unstructured":"J. L. Abell\u00e1n, J. Fern\u00e1ndez, M. E. Acacio, D. Bertozzi, D. Bortolotti, A. Marongiu and L. Benini. Design of a Collective Communication Infrastructure for Barrier Synchronization in Cluster-Based Nanoscale MPSoCs. In Proceedings of the Design, Automation & Test in Europe Conference & Exhibition, 2012.","DOI":"10.1109\/DATE.2012.6176519"},{"key":"26_CR20","doi-asserted-by":"crossref","unstructured":"J. M. Mellor-Crummey and M. L. Scott. Algorithms for Scalable Synchronization on Shared-Memory Multiprocessors. ACM Transactions on Computer Systems, 9(1):21\u201365, 1991.","DOI":"10.1145\/103727.103729"},{"key":"26_CR21","unstructured":"J. Mauro, R. McDougall. Solaris Internals: Core Kernel Components. Sun Microsystem Press, 2001."},{"key":"26_CR22","doi-asserted-by":"crossref","unstructured":"J. Oh, M. Prvulovic and A. Zajic. TLSync: Support for Multiple Fast Barriers Using On-Chip Transmission Lines. In Proceedings of the 38th International Symposium on Computer Architecture, 2011.","DOI":"10.1145\/2000064.2000078"},{"key":"26_CR23","unstructured":"J. P. Lozi, G. Thomas, J. Lawall and G. Muller. Efficient Locking for Multicore Architectures. Technical Report RR-7779, INRIA, 2011."},{"key":"26_CR24","doi-asserted-by":"crossref","unstructured":"J. R. Goodman, M. K. Vernon and P. J. Woest. Efficient Synchronization Primitives for Large-Scale Cache-Coherent Multiprocessors. In Proceedings of the 3rd International Conference on Architectural Support for Programming Languages and Operating Systems, 1989.","DOI":"10.1145\/70082.68188"},{"key":"26_CR25","doi-asserted-by":"crossref","unstructured":"J. Sampson, R. Gonz\u00e1lez, J. F. Collard, N. P. Jouppi, M. Schlansker and B. Calder. Exploiting Fine-Grained Data Parallelism with Chip Multiprocessors and Fast Barriers. In Proceedings of the 39th Annual IEEE\/ACM International Symposium on Microarchitecture, 2006.","DOI":"10.1109\/MICRO.2006.23"},{"key":"26_CR26","doi-asserted-by":"crossref","unstructured":"J. Sartori and R. Kumar. Low-Overhead, High-Speed Multi-core Barrier Synchronization. In Proceedings of the 5th International Conference on High Performance Embedded Architectures and Compilers, 2010.","DOI":"10.1007\/978-3-642-11515-8_4"},{"key":"26_CR27","doi-asserted-by":"crossref","unstructured":"L. Barroso and Urs H\u00f6lzle. The Datacenter as a Computer. An Introduction to the Design of Warehouse-Scale Machines. Morgan and Claypool Publishers, 2009.","DOI":"10.1007\/978-3-031-01722-3"},{"key":"26_CR28","doi-asserted-by":"crossref","unstructured":"M. L. Scott and W. N. Scherer. Scalable Queue-Based Spin Locks with Timeout. In Proceedings of the 8th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, 2001.","DOI":"10.1145\/379539.379566"},{"key":"26_CR29","doi-asserted-by":"crossref","unstructured":"M. Monchiero, G. Palermo, C. Silvano and O. Villa. An Efficient Synchronization Technique for Multiprocessor Systems on-Chip. ACM SIGARCH Computer Architecture News, 34(1):33\u201340, 2006.","DOI":"10.1145\/1147349.1147357"},{"key":"26_CR30","doi-asserted-by":"crossref","unstructured":"N. R. Tallent, J. M. Mellor-Crummey and A. Porterfield. Analyzing Lock Contention in Multithreaded Applications. In Proceedings of the 15th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, 2010.","DOI":"10.1145\/1693453.1693489"},{"key":"26_CR31","doi-asserted-by":"crossref","unstructured":"P. Coteus, H. R. Bickford, T. M. Cipolla, P. G. Crumley, A. Gara, S. A. Hall, G. V. Kopcsay, A. P. Lanzetta, L. S. Mok, R. Rand, R. Swetz, T. Takken, P. La Rocca, C. Marroquin, P. R. Germann and M.J. Jeanson. Packaging the Blue Gene\/L Supercomputer. IBM Journal of Research and Development, 49(2):213\u2013248, 2005.","DOI":"10.1147\/rd.492.0213"},{"key":"26_CR32","unstructured":"P. Tang and P. C. Yew. Processor Self-Scheduling for Multiple-Nested Parallel Loops. In Proceedings of the the International Conference on Parallel Processing, 1986."},{"key":"26_CR33","doi-asserted-by":"crossref","unstructured":"R. Ho, T. Ono, R. D. Hopkins, A. Chow, J. Schauer, F. Y. Liu and R. Drost. High-Speed and Low-Energy Capacitively-Driven On-Chip Wires. IEEE Journal of Solid State Circuits, 43(1):52\u201360, 2008.","DOI":"10.1109\/JSSC.2007.910807"},{"key":"26_CR34","doi-asserted-by":"crossref","unstructured":"R. Rajwar and J. R. Goodman. Transactional Lock-free Execution of Lock-based Programs. In Proceedings of the 10th Annual Conference on Architectural Support for Programming Languages and Operating Systems, 2002.","DOI":"10.1145\/605397.605399"},{"key":"26_CR35","doi-asserted-by":"crossref","unstructured":"R. T. Chang, N. Talwalkar, P. Yue and S. S. Wong. Near Speed-of-Light Signaling over On-Chip Electrical Interconnects. IEEE Journal of Solid-State Circuits, 38(5):834\u2013838, 2003.","DOI":"10.1109\/JSSC.2003.810060"},{"key":"26_CR36","doi-asserted-by":"crossref","unstructured":"S. Bell et al. TILE64 - Processor: A 64-Core SoC with Mesh Interconnect. In Proceedings of the International Solid-State Circuits Conference Digest of Technical Papers, 2008.","DOI":"10.1109\/ISSCC.2008.4523070"},{"key":"26_CR37","doi-asserted-by":"crossref","unstructured":"S. C. Woo, M. Ohara, E. Torrie, J. P. Singh and A. Gupta. The SPLASH-2 programs: Characterization and Methodological Considerations. In Proceedings of the 22nd International Symposium on Computer Architecture, 1995.","DOI":"10.1145\/223982.223990"},{"key":"26_CR38","doi-asserted-by":"crossref","unstructured":"S. D. Sherlekar. Intel Many Integrated Core (MIC) Architecture. In Proceedings of the IEEE International Conference on Parallel and Distributed Systems, 2012.","DOI":"10.1109\/ICPADS.2012.162"},{"key":"26_CR39","doi-asserted-by":"crossref","unstructured":"T. E. Anderson. The Performance Implications of Spin-Waiting Alternatives for Shared Memory Multiprocessors. In Proceedings of the Intel Conference on Parallel Processing, 1989.","DOI":"10.1109\/12.40843"},{"key":"26_CR40","doi-asserted-by":"crossref","unstructured":"T. Krishna, A. Kumar, L-S. Peh, J. Postman, P. Chiang and M. Erez. Express Virtual Channels with Capacitively Driven Global Links. IEEE Micro, 29(4):48\u201361, 2009.","DOI":"10.1109\/MM.2009.64"},{"key":"26_CR41","doi-asserted-by":"crossref","unstructured":"W. T.-Y. Hsu and P.-C. Yew. An Effective Synchronization Network for Hot-Spot Accesses. ACM Transactions on Computer Systems, 10(3):167\u2013189, 1992.","DOI":"10.1145\/146937.146938"},{"key":"26_CR42","doi-asserted-by":"crossref","unstructured":"Z. Hu, J. del Cuvillo, W. Zhu and G. R. Gao. Optimization of Dense Matrix Multiplication on IBM Cyclops-64: Challenges and Experiences. In Proceedings of the 12th International European Conference on Parallel and Distributed Computing, 2006.","DOI":"10.1007\/11823285_14"}],"container-title":["Handbook on Data Centers"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-1-4939-2092-1_26","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,2,20]],"date-time":"2023-02-20T22:40:44Z","timestamp":1676932844000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-1-4939-2092-1_26"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2015]]},"ISBN":["9781493920914","9781493920921"],"references-count":42,"URL":"https:\/\/doi.org\/10.1007\/978-1-4939-2092-1_26","relation":{},"subject":[],"published":{"date-parts":[[2015]]},"assertion":[{"value":"17 March 2015","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}}]}}