{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T04:59:49Z","timestamp":1783745989294,"version":"3.55.0"},"reference-count":57,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2017,11,16]],"date-time":"2017-11-16T00:00:00Z","timestamp":1510790400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2017,11,16]],"date-time":"2017-11-16T00:00:00Z","timestamp":1510790400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"published-print":{"date-parts":[[2018,3]]},"DOI":"10.1007\/s11227-017-2177-5","type":"journal-article","created":{"date-parts":[[2017,11,16]],"date-time":"2017-11-16T07:56:18Z","timestamp":1510818978000},"page":"1341-1377","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":46,"title":["Theoretical peak FLOPS per instruction set: a tutorial"],"prefix":"10.1007","volume":"74","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4466-8948","authenticated-orcid":false,"given":"Romain","family":"Dolbeau","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2017,11,16]]},"reference":[{"issue":"3b","key":"2177_CR1","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/255129.255132","volume":"18","author":"R Alverson","year":"1990","unstructured":"Alverson R, Callahan D, Cummings D, Koblenz B, Porterfield A, Smith B (1990) The tera computer system. ACM SIGARCH Comput Archit News 18(3b):1\u20136","journal-title":"ACM SIGARCH Comput Archit News"},{"key":"2177_CR2","unstructured":"AMD\u00ae (2017) AMD optimizing C\/C++ compiler. http:\/\/developer.amd.com\/amd-aocc\/"},{"key":"2177_CR3","unstructured":"AMD\u00ae (2017) Introducing the Radeon\u2122 RX Vega$$^{64}$$. https:\/\/gaming.radeon.com\/en\/product\/vega\/radeon-rx-vega-64\/"},{"key":"2177_CR4","unstructured":"Arm\u00ae (2017) Cortex-A57 processor. https:\/\/www.arm.com\/products\/processors\/cortex-a\/cortex-a57-processor.php"},{"key":"2177_CR5","unstructured":"Arm\u00ae (2017) NEON. https:\/\/developer.arm.com\/technologies\/neon"},{"key":"2177_CR6","unstructured":"Arm\u00ae (2017) Arm compiler for HPC. https:\/\/developer.arm.com\/products\/software-development-tools\/hpc\/arm-compiler-for-hpc"},{"key":"2177_CR7","unstructured":"Zuras D, Cowlishaw M, Aiken A, Applegate M, Bailey D, Bass S, Bhandarkar D, Bhat M, Bindel D, Boldo S et al (2008) IEEE standard for floating-point arithmetic. IEEE Std 754\u20132008, pp 1\u201370"},{"issue":"1","key":"2177_CR8","doi-asserted-by":"publisher","first-page":"45","DOI":"10.1109\/2.19822","volume":"22","author":"MC August","year":"1989","unstructured":"August MC, Brost GM, Hsiung CC, Schiffleger AJ (1989) Cray X-MP: the birth of a supercomputer. Computer 22(1):45\u201352","journal-title":"Computer"},{"issue":"3","key":"2177_CR9","doi-asserted-by":"publisher","first-page":"331","DOI":"10.1109\/TC.2005.36","volume":"54","author":"N Brisebarre","year":"2005","unstructured":"Brisebarre N, Defour D, Kornerup P, Muller JM, Revol N (2005) A new range-reduction algorithm. IEEE Trans Comput 54(3):331\u2013339","journal-title":"IEEE Trans Comput"},{"key":"2177_CR10","volume-title":"Planning a computer system: project stretch","author":"W Buchholz","year":"1962","unstructured":"Buchholz W (1962) Planning a computer system: project stretch. McGraw-Hill Inc, Hightstown, NJ, USA"},{"key":"2177_CR11","doi-asserted-by":"crossref","unstructured":"Butler M (2010) Bulldozer: a new approach to multi-threaded compute performance. In: Hot Chips 22 Symposium (HCS), 2010 IEEE. IEEE, pp 1\u201317","DOI":"10.1109\/HOTCHIPS.2010.7480086"},{"key":"2177_CR12","doi-asserted-by":"publisher","unstructured":"Butler M, Barnes L, Sarma DD, Gelinas B (2011) Bulldozer: an approach to multithreaded compute performance. IEEE Micro 31(2):6\u201315. https:\/\/doi.org\/10.1109\/MM.2011.23","DOI":"10.1109\/MM.2011.23"},{"key":"2177_CR13","doi-asserted-by":"crossref","unstructured":"Clark M (2016) A new X86 core architecture for the next generation of computing. Hot Chips 28 Symposium (HCS). IEEE, pp 1\u201319","DOI":"10.1109\/HOTCHIPS.2016.7936224"},{"issue":"3","key":"2177_CR14","first-page":"162","volume":"1","author":"M Daumas","year":"1995","unstructured":"Daumas M, Mazenc C, Merrheim X, Muller JM (1995) Modular range reduction: a new algorithm for fast and accurate computation on the elementary functions. J Univers Comput Sci 1(3):162\u2013175","journal-title":"J Univers Comput Sci"},{"issue":"2","key":"2177_CR15","doi-asserted-by":"publisher","first-page":"85","DOI":"10.1109\/40.848475","volume":"20","author":"K Diefendorff","year":"2000","unstructured":"Diefendorff K, Dubey PK, Hochsprung R, Scale H (2000) Altivec extension to PowerPC accelerates media processing. IEEE Micro 20(2):85\u201395","journal-title":"IEEE Micro"},{"key":"2177_CR16","first-page":"1","volume":"6","author":"R Dolbeau","year":"2004","unstructured":"Dolbeau R, Seznec A (2004) CASH: revisiting hardware sharing in single-chip parallel processor. J Instr Level Parallelism 6:1\u201316","journal-title":"J Instr Level Parallelism"},{"key":"2177_CR17","doi-asserted-by":"crossref","unstructured":"Fayneh E, Yuffe M, Knoll E, Zelikson M, Abozaed M, Talker Y, Shmuely Z, Rahme SA (2016) 4.1 14nm 6th-Generation core processor soc with low power consumption and improved performance. In: Solid-State Circuits Conference (ISSCC), 2016 IEEE International. IEEE, pp 72\u201373","DOI":"10.1109\/ISSCC.2016.7417912"},{"key":"2177_CR18","unstructured":"Fog A (1996\u20132016) Instruction tables: lists of instruction latencies, throughputs and micro-operation breakdowns for Intel, AMD and VIA cpus. Copenhagen University College of Engineering. http:\/\/www.agner.org\/optimize\/instruction_tables.pdf"},{"key":"2177_CR19","doi-asserted-by":"crossref","unstructured":"Govindu G, Zhuo L, Choi S, Prasanna V (2004) Analysis of high-performance floating-point arithmetic on fpgas. In: Parallel and Distributed Processing Symposium, 2004. Proceedings. 18th International. IEEE, p 149","DOI":"10.1109\/IPDPS.2004.1303135"},{"key":"2177_CR20","unstructured":"Grisenthwaite R (2011) Armv8 technology preview. In: IEEE Conference"},{"issue":"13","key":"2177_CR21","first-page":"11","volume":"6","author":"L Gwennap","year":"2011","unstructured":"Gwennap L (2011) Adapteva: more flops, less watts. Microprocess Rep 6(13):11\u201302","journal-title":"Microprocess Rep"},{"issue":"1","key":"2177_CR22","first-page":"94","volume":"34","author":"D Henderson","year":"2000","unstructured":"Henderson D (2000) Elementary functions: algorithms and implementation. Math Comput Educ 34(1):94","journal-title":"Math Comput Educ"},{"key":"2177_CR23","volume-title":"Computer architecture: a quantitative approach","author":"JL Hennessy","year":"2011","unstructured":"Hennessy JL, Patterson DA (2011) Computer architecture: a quantitative approach, 5th edn. Elsevier, Amsterdam","edition":"5"},{"key":"2177_CR24","unstructured":"Intel\u00ae (2010) Intel\u00ae Xeon\u00ae Processor X5650 (12M Cache, 2.66 GHz, 6.40 GT\/s Intel\u00ae QPI). http:\/\/ark.intel.com\/products\/47922\/Intel-Xeon-Processor-X5650-12M-Cache-2_66-GHz-6_40-GTs-Intel-QPI"},{"key":"2177_CR25","unstructured":"Intel\u00ae (2014) Intel\u00ae Xeon\u00ae Processor E5-2695 v3 (35m Cache, 2.30 GHz). http:\/\/ark.intel.com\/products\/81057\/Intel-Xeon-Processor-E5-2695-v3-35M-Cache-2_30-GHz"},{"key":"2177_CR26","unstructured":"Intel\u00ae (2014) Intel\u00ae Xeon\u00ae Processor E5 v3 product families specification update. http:\/\/www.intel.com\/content\/dam\/www\/public\/us\/en\/documents\/specification-updates\/xeon-e5-v3-spec-update.pdf"},{"key":"2177_CR27","unstructured":"Intel\u00ae (2014) Optimizing performance with Intel\u00ae advanced vector extensions. http:\/\/www.intel.com\/content\/dam\/www\/public\/us\/en\/documents\/white-papers\/performance-xeon-e5-v3-advanced-vector-extensions-paper.pdf"},{"key":"2177_CR28","unstructured":"Intel\u00ae (2016) Intel\u00ae 64 and IA-32 architectures software developer\u2019s manual volume 2 (2A, 2B & 2C): instruction set reference, A\u2013Z. 325383-060. http:\/\/www.intel.com\/content\/www\/us\/en\/architecture-and-technology\/64-ia-32-architectures-software-developer-instruction-set-reference-manual-325383.html"},{"key":"2177_CR29","unstructured":"Intel\u00ae (2016) Intel\u00ae Xeon Phi\u2122 processor software optimization guide (334541-001). https:\/\/software.intel.com\/sites\/default\/files\/managed\/11\/56\/intel-xeon-phi-processor-software-optimization-guide.pdf"},{"key":"2177_CR30","unstructured":"Intel\u00ae (2017) Intel\u00ae Intrinsics guide. https:\/\/software.intel.com\/sites\/landingpage\/IntrinsicsGuide\/"},{"key":"2177_CR31","unstructured":"Kanter D (2016) AMD finds Zen in microarchitecture. Microprocess Rep. http:\/\/www.linleygroup.com\/newsletters\/newsletter_detail.php?num=5577"},{"issue":"2","key":"2177_CR32","doi-asserted-by":"publisher","first-page":"27","DOI":"10.1109\/40.592310","volume":"17","author":"A Kumar","year":"1997","unstructured":"Kumar A (1997) The HP PA-8000 RISC CPU. IEEE Micro 17(2):27\u201332","journal-title":"IEEE Micro"},{"key":"2177_CR33","doi-asserted-by":"crossref","unstructured":"Kumar R, Jouppi NP, Tullsen DM (2004) Conjoined-core chip multiprocessing. In: Proceedings of the 37th Annual IEEE\/ACM International Symposium on Microarchitecture. IEEE Computer Society, pp 195\u2013206","DOI":"10.1109\/MICRO.2004.12"},{"key":"2177_CR34","doi-asserted-by":"crossref","unstructured":"Lee B, Burgess N (2002) Parameterisable floating-point operations on FPGA. In: Conference Record of the Thirty-Sixth Asilomar Conference on Signals, Systems and Computers, 2002, vol 2. IEEE, pp 1064\u20131068","DOI":"10.1109\/ACSSC.2002.1196947"},{"key":"2177_CR35","unstructured":"LLVM (2017) LLVM.org. https:\/\/llvm.org"},{"key":"2177_CR36","unstructured":"LLVM Documentation (2017) Auto-vectorization in LLVM. https:\/\/llvm.org\/docs\/Vectorizers.html"},{"key":"2177_CR37","doi-asserted-by":"crossref","unstructured":"Lo YJ, Williams S, Van Straalen B, Ligocki TJ, Cordery MJ, Wright NJ, Hall MW, Oliker L (2014) Roofline model toolkit: a practical tool for architectural and program analysis. In: International Workshop on Performance Modeling, Benchmarking and Simulation of High Performance Computer Systems. Springer, pp 129\u2013148","DOI":"10.1007\/978-3-319-17248-4_7"},{"key":"2177_CR38","doi-asserted-by":"crossref","unstructured":"Mantor M (2012) AMD Radeon\u2122 HD 7970 with graphics core next (GCN) architecture. In: Hot Chips 24 Symposium (HCS), 2012 IEEE. IEEE, pp 1\u201335","DOI":"10.1109\/HOTCHIPS.2012.7476485"},{"issue":"1","key":"2177_CR39","doi-asserted-by":"publisher","first-page":"59","DOI":"10.1147\/rd.341.0059","volume":"34","author":"RK Montoye","year":"1990","unstructured":"Montoye RK, Hokenek E, Runyon SL (1990) Design of the IBM RISC System\/6000 floating-point execution unit. IBM J Res Dev 34(1):59\u201370","journal-title":"IBM J Res Dev"},{"issue":"1","key":"2177_CR40","doi-asserted-by":"publisher","first-page":"105","DOI":"10.1109\/JSSC.2015.2464688","volume":"51","author":"B Munger","year":"2016","unstructured":"Munger B, Akeson D, Arekapudi S, Burd T, Fair HR, Farrell J, Johnson D, Krishnan G, McIntyre H, McLellan E et al (2016) Carrizo: a high performance, energy efficient 28 nm APU. IEEE J Solid State Circuits 51(1):105\u2013116","journal-title":"IEEE J Solid State Circuits"},{"issue":"1","key":"2177_CR41","doi-asserted-by":"publisher","first-page":"42","DOI":"10.29292\/jics.v5i1.309","volume":"5","author":"DM Mu\u00f1oz","year":"2010","unstructured":"Mu\u00f1oz DM, Sanchez DF, Llanos CH, Ayala-Rinc\u00f3n M (2010) Tradeoff of FPGA design of a floating-point library for arithmetic operators. J Integr Circuits Syst 5(1):42\u201352","journal-title":"J Integr Circuits Syst"},{"key":"2177_CR42","unstructured":"NVidia (2008\u20132017) CUDA C programming guide. http:\/\/docs.nvidia.com\/cuda\/cuda-c-programming-guide\/"},{"key":"2177_CR43","unstructured":"NVidia (2008\u20132017) CUDA C programming guide: 5.4.1. Arithmetic instructions. http:\/\/docs.nvidia.com\/cuda\/cuda-c-programming-guide\/#arithmetic-instructions"},{"key":"2177_CR44","unstructured":"NVidia (2008\u20132017) CUDA GPUs. https:\/\/developer.nvidia.com\/cuda-gpus"},{"issue":"2","key":"2177_CR45","doi-asserted-by":"publisher","first-page":"37","DOI":"10.1109\/40.755466","volume":"19","author":"S Oberman","year":"1999","unstructured":"Oberman S, Favor G, Weber F (1999) AMD 3DNow! technology: architecture and implementations. IEEE Micro 19(2):37\u201348","journal-title":"IEEE Micro"},{"key":"2177_CR46","doi-asserted-by":"crossref","unstructured":"Olofsson A, Nordstr\u00f6m T, Ul-Abdin Z (2014) Kickstarting high-performance energy-efficient manycore architectures with epiphany. In: 2014 48th Asilomar Conference on Signals, Systems and Computers. IEEE, pp 1719\u20131726","DOI":"10.1109\/ACSSC.2014.7094761"},{"issue":"1","key":"2177_CR47","doi-asserted-by":"publisher","first-page":"63","DOI":"10.1145\/359327.359336","volume":"21","author":"RM Russell","year":"1978","unstructured":"Russell RM (1978) The CRAY-1 computer system. Commun ACM 21(1):63\u201372","journal-title":"Commun ACM"},{"key":"2177_CR48","unstructured":"Shayesteh A (2006) Factored multi-core architectures. PhD thesis, University of California Los Angeles"},{"key":"2177_CR49","doi-asserted-by":"crossref","unstructured":"Singh AYG, Favor G, Yeung A (2014) AppliedMicro X-Gene 2. In: HotChips","DOI":"10.1109\/HOTCHIPS.2014.7478817"},{"issue":"12","key":"2177_CR50","doi-asserted-by":"publisher","first-page":"1609","DOI":"10.1109\/5.476078","volume":"83","author":"JE Smith","year":"1995","unstructured":"Smith JE, Sohi GS (1995) The microarchitecture of superscalar processors. Proc IEEE 83(12):1609\u20131624","journal-title":"Proc IEEE"},{"key":"2177_CR51","doi-asserted-by":"crossref","unstructured":"Snavely A, Carter L, Boisseau J, Majumdar A, Gatlin KS, Mitchell N, Feo J, Koblenz B (1998) Multi-processor performance on the Tera MTA. In: Proceedings of the 1998 ACM\/IEEE Conference on Supercomputing. IEEE Computer Society, pp 1\u20138","DOI":"10.1109\/SC.1998.10049"},{"key":"2177_CR52","doi-asserted-by":"crossref","unstructured":"Sodani A (2015) Knights landing (KNL): 2nd Generation Intel\u00ae Xeon Phi Processor. In: Hot Chips 27 Symposium (HCS), 2015 IEEE. IEEE, pp 1\u201324","DOI":"10.1109\/HOTCHIPS.2015.7477467"},{"key":"2177_CR53","unstructured":"Stephens N (2016) Technology update: the scalable vector extension (sve) for the armv8-a architecture. https:\/\/community.arm.com\/groups\/processors\/blog\/2016\/08\/22\/technology-update-the-scalable-vector-extension-sve-for-the-armv8-a-architecture"},{"key":"2177_CR54","unstructured":"Strenski D (2007) FPGA floating point performance\u2014a pencil and paper evaluation. HPC Wire. https:\/\/www.hpcwire.com\/2007\/01\/12\/fpga_floating_point_performance\/"},{"key":"2177_CR55","doi-asserted-by":"publisher","unstructured":"Thornton JE (1965) Parallel operation in the control data 6600. In: Proceedings of the October 27\u201329, 1964, Fall Joint Computer Conference, Part II: Very High Speed Computer Systems. ACM, New York, NY, USA, AFIPS \u201964 (Fall, part II), pp 33\u201340. https:\/\/doi.org\/10.1145\/1464039.1464045","DOI":"10.1145\/1464039.1464045"},{"key":"2177_CR56","doi-asserted-by":"crossref","unstructured":"Tullsen DM, Eggers SJ, Levy HM (1995) Simultaneous multithreading: maximizing on-chip parallelism. ACM SIGARCH Comp Archit News 23(2):392\u2013403. http:\/\/doi.acm.org\/10.1145\/225830.224449","DOI":"10.1145\/225830.224449"},{"key":"2177_CR57","unstructured":"Wikipedia (2017) x87. https:\/\/en.wikipedia.org\/wiki\/X87"}],"updated-by":[{"DOI":"10.1007\/s11227-022-04443-1","type":"correction","label":"Correction","source":"publisher","updated":{"date-parts":[[2022,3,27]],"date-time":"2022-03-27T00:00:00Z","timestamp":1648339200000}}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-017-2177-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-017-2177-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-017-2177-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,4,6]],"date-time":"2022-04-06T14:22:29Z","timestamp":1649254949000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-017-2177-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017,11,16]]},"references-count":57,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2018,3]]}},"alternative-id":["2177"],"URL":"https:\/\/doi.org\/10.1007\/s11227-017-2177-5","relation":{"correction":[{"id-type":"doi","id":"10.1007\/s11227-022-04443-1","asserted-by":"object"}]},"ISSN":["0920-8542","1573-0484"],"issn-type":[{"value":"0920-8542","type":"print"},{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2017,11,16]]},"assertion":[{"value":"16 November 2017","order":1,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 March 2022","order":2,"name":"change_date","label":"Change Date","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"Correction","order":3,"name":"change_type","label":"Change Type","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"A Correction to this paper has been published:","order":4,"name":"change_details","label":"Change Details","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"https:\/\/doi.org\/10.1007\/s11227-022-04443-1","URL":"https:\/\/doi.org\/10.1007\/s11227-022-04443-1","order":5,"name":"change_details","label":"Change Details","group":{"name":"ArticleHistory","label":"Article History"}}]}}