{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,5,29]],"date-time":"2025-05-29T17:40:04Z","timestamp":1748540404258,"version":"3.41.0"},"publisher-location":"Berlin, Heidelberg","reference-count":18,"publisher":"Springer Berlin Heidelberg","isbn-type":[{"type":"print","value":"9783662480953"},{"type":"electronic","value":"9783662480960"}],"license":[{"start":{"date-parts":[[2015,1,1]],"date-time":"2015-01-01T00:00:00Z","timestamp":1420070400000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2015]]},"DOI":"10.1007\/978-3-662-48096-0_45","type":"book-chapter","created":{"date-parts":[[2015,7,24]],"date-time":"2015-07-24T06:16:03Z","timestamp":1437718563000},"page":"588-600","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":19,"title":["Effective Barrier Synchronization on Intel Xeon Phi Coprocessor"],"prefix":"10.1007","author":[{"given":"Andrey","family":"Rodchenko","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Andy","family":"Nisbet","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Antoniu","family":"Pop","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mikel","family":"Luj\u00e1n","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2015,7,25]]},"reference":[{"doi-asserted-by":"crossref","unstructured":"Agarwal, A., Cherian, M.: Adaptive backoff synchronization techniques. In: Proceedings of the of the International Symposium on Computer Architecture, pp. 396\u2013406 (1989)","key":"45_CR1","DOI":"10.1145\/74926.74970"},{"issue":"4","key":"45_CR2","doi-asserted-by":"publisher","first-page":"295","DOI":"10.1007\/BF01407877","volume":"15","author":"ED Brooks III","year":"1986","unstructured":"Brooks III, E.D.: The butterfly barrier. Int. J. Parallel Program. 15(4), 295\u2013307 (1986)","journal-title":"Int. J. Parallel Program."},{"unstructured":"Bull, J.M.: Measuring synchronisation and scheduling overheads in OpenMP. In: Proceedings of the First European Workshop on OpenMP, pp. 99\u2013105 (1999)","key":"45_CR3"},{"doi-asserted-by":"crossref","unstructured":"Caballero, D., Duran, A., Martorell, X.: An OpenMP barrier usingSIMD instructions for Intel $$^{\\textregistered }$$ \u00ae Xeon Phi $$^{\\rm TM}$$ TM coprocessor. In: Rendell, A.P., Chapman, B.M., M\u00fcller, M.S. (eds.) IWOMP 2013. LNCS, vol. 8122, pp. 99\u2013113. Springer, Heidelberg (2013)","key":"45_CR4","DOI":"10.1007\/978-3-642-40698-0_8"},{"unstructured":"Cownie, J.: Fastest possible barrier (Intel developer zone forum discussion) (2013). http:\/\/software.intel.com\/en-us\/forums\/topic\/392587 . Last accessed 1-Jun-2015","key":"45_CR5"},{"unstructured":"Dolbeau, R.: Address selection for efficient barriers on the Intel Xeon Phi (2013). http:\/\/www.dolbeau.name\/dolbeau\/publications\/barrierphi.pdf . Last accessed 1 Jun 2015","key":"45_CR6"},{"doi-asserted-by":"crossref","unstructured":"Grunwald, D., Vajracharya, S.: Efficient barriers for distributed shared memory computers. In: Proceedings of International Parallel Processing Symposium, pp. 604\u2013608 (1994)","key":"45_CR7","DOI":"10.1109\/IPPS.1994.288242"},{"issue":"1","key":"45_CR8","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/BF01379320","volume":"17","author":"D Hensgen","year":"1988","unstructured":"Hensgen, D., Finkel, R., Manber, U.: Two algorithms for barrier synchronization. Int. J. Parallel Program. 17(1), 1\u201317 (1988)","journal-title":"Int. J. Parallel Program."},{"doi-asserted-by":"crossref","unstructured":"Hoefler, T., Mehlan, T., Mietke, F., Rehm, W.: Fast barrier synchronization for InfiniBand. In: 20th International Parallel and Distributed Processing Symposium, p. 7 (2006)","key":"45_CR9","DOI":"10.1109\/IPDPS.2006.1639561"},{"unstructured":"Intel Xeon Phi coprocessor system software developers guide (2014). https:\/\/software.intel.com\/sites\/default\/files\/managed\/09\/07\/xeon-phi-coprocessor-system-software-developers-guide.pdf . Last accessed 1 Jun 2015","key":"45_CR10"},{"doi-asserted-by":"crossref","unstructured":"Krishnaiyer, R., Kultursay, E., Chawla, P., Preis, S., Zvezdin, A., Saito, H.: Compiler-based data prefetching and streaming non-temporal store generation for the Intel Xeon Phi coprocessor. In: Workshop on Multithreaded Architectures and Applications published as 27th IEEE IPDPSW, pp. 1575\u20131586 (2013)","key":"45_CR11","DOI":"10.1109\/IPDPSW.2013.231"},{"issue":"1","key":"45_CR12","doi-asserted-by":"publisher","first-page":"21","DOI":"10.1145\/103727.103729","volume":"9","author":"JM Mellor-Crummey","year":"1991","unstructured":"Mellor-Crummey, J.M., Scott, M.L.: Algorithms for scalable synchronization on shared-memory multiprocessors. ACM Trans. Comput. Syst. 9(1), 21\u201365 (1991)","journal-title":"ACM Trans. Comput. Syst."},{"unstructured":"NAS parallel benchmarks. http:\/\/www.nas.nasa.gov\/publications\/npb.html . Last accessed 1 Jun 2015","key":"45_CR13"},{"doi-asserted-by":"crossref","unstructured":"Ramos, S., Hoefler, T.: Modeling communication in cache-coherent smp systems: A case-study with Xeon Phi. In: High-Performance Parallel and Distributed Computing 2013, pp. 97\u2013108 (2013)","key":"45_CR14","DOI":"10.1145\/2462902.2462916"},{"doi-asserted-by":"crossref","unstructured":"Sartori, J., Kumar, R.: Low-overhead, high-speed multi-core barrier synchronization. In: Proceedings of the 5th International Conference on High Performance and Embedded Architecture and Compilation, pp. 18\u201334 (2010)","key":"45_CR15","DOI":"10.1007\/978-3-642-11515-8_4"},{"doi-asserted-by":"crossref","unstructured":"Seo, S., Jo, G., Lee, J.: Performance characterization of the NAS parallel benchmarks in OpenCL. In: 2011 IEEE International Symposium on Workload Characterization, pp. 137\u2013148 (2011)","key":"45_CR16","DOI":"10.1109\/IISWC.2011.6114174"},{"doi-asserted-by":"crossref","unstructured":"Shirako, J., Peixotto, D.M., Sarkar, V., Scherer, W.N.: Phasers: A unified deadlock-free construct for collective and point-to-point synchronization. In: Proceedings of the 22nd International Conference on Supercomputing, pp. 277\u2013288 (2008)","key":"45_CR17","DOI":"10.1145\/1375527.1375568"},{"issue":"4","key":"45_CR18","first-page":"388","volume":"C\u201336","author":"PC Yew","year":"1987","unstructured":"Yew, P.C., Tzeng, N.F., Lawrie, D.H.: Distributing hot-spot addressing in large-scale multiprocessors. IEEE Trans. Comput. C\u201336(4), 388\u2013395 (1987)","journal-title":"IEEE Trans. Comput."}],"container-title":["Lecture Notes in Computer Science","Euro-Par 2015: Parallel Processing"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-662-48096-0_45","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,29]],"date-time":"2025-05-29T16:59:06Z","timestamp":1748537946000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-662-48096-0_45"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2015]]},"ISBN":["9783662480953","9783662480960"],"references-count":18,"URL":"https:\/\/doi.org\/10.1007\/978-3-662-48096-0_45","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2015]]},"assertion":[{"value":"25 July 2015","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"This content has been made available to all.","name":"free","label":"Free to read"}]}}