{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2022,4,2]],"date-time":"2022-04-02T07:06:40Z","timestamp":1648883200342},"reference-count":143,"publisher":"Elsevier","license":[{"start":{"date-parts":[[2000,1,1]],"date-time":"2000-01-01T00:00:00Z","timestamp":946684800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2000]]},"DOI":"10.1016\/s0065-2458(00)80003-0","type":"book-chapter","created":{"date-parts":[[2011,1,19]],"date-time":"2011-01-19T05:56:21Z","timestamp":1295416581000},"page":"1-53","source":"Crossref","is-referenced-by-count":0,"title":["Shared-memory multiprocessing: Current state and future directions"],"prefix":"10.1016","author":[{"given":"Per","family":"Sterstr\u00f6m","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Erik","family":"Hagersten","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"David J.","family":"Lilja","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Margaret","family":"Martonosi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Madan","family":"Venugopal","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/S0065-2458(00)80003-0_bib1","author":"Culler","year":"1999"},{"issue":"11","key":"10.1016\/S0065-2458(00)80003-0_bib2","doi-asserted-by":"crossref","first-page":"26","DOI":"10.1109\/2.86784","article-title":"Reducing contention in shared-memory multiprocessors","volume":"21","author":"Stenstrom","year":"1988","journal-title":"IEEE Computer"},{"issue":"6","key":"10.1016\/S0065-2458(00)80003-0_bib3","doi-asserted-by":"crossref","first-page":"12","DOI":"10.1109\/2.55497","article-title":"A survey of cache coherence schemes for multiprocessors","volume":"23","author":"Stenstrom","year":"1990","journal-title":"IEEE Computer"},{"issue":"3","key":"10.1016\/S0065-2458(00)80003-0_bib4","doi-asserted-by":"crossref","first-page":"303","DOI":"10.1145\/158439.158907","article-title":"Cache coherence in large-scale shared-memory multiprocessors: issues and comparisons","volume":"25","author":"Lilja","year":"1993","journal-title":"ACM Computing Surveys"},{"key":"10.1016\/S0065-2458(00)80003-0_bib5","doi-asserted-by":"crossref","first-page":"39","DOI":"10.1109\/40.653032","article-title":"STARFIRE: Extending the SMP envelope","volume":"18","author":"Arpaci-Dusseau","year":"1998","journal-title":"IEEE Micro"},{"key":"10.1016\/S0065-2458(00)80003-0_bib6","series-title":"Proceedings of the 1985 International Conference on Parallel Processing","first-page":"764","article-title":"The IBM research parallel processor prototype (RP3)","author":"Pfister","year":"1985"},{"key":"10.1016\/S0065-2458(00)80003-0_bib7","article-title":"Butterfly Parallel Processor","year":"1986"},{"key":"10.1016\/S0065-2458(00)80003-0_bib8","series-title":"Proceedings of the 7th International Conference on Architectural Support for Programming Languages and Operating Systems","first-page":"279","article-title":"Operating system support for improving data locality on CC-NUMA compute servers","author":"Verghese","year":"1996"},{"key":"10.1016\/S0065-2458(00)80003-0_bib9","series-title":"Proceedings of the COMP-CON","first-page":"184","article-title":"Competitive management of distributed shared memory","author":"Black","year":"1989"},{"key":"10.1016\/S0065-2458(00)80003-0_bib10","doi-asserted-by":"crossref","first-page":"1112","DOI":"10.1109\/TC.1978.1675013","article-title":"A New solution to coherence problems in multicache systems","volume":"27","author":"Censier","year":"1978","journal-title":"IEEE Trans. Computers"},{"issue":"6","key":"10.1016\/S0065-2458(00)80003-0_bib11","doi-asserted-by":"crossref","first-page":"74","DOI":"10.1109\/2.55503","article-title":"Scalable coherent interface","volume":"23","author":"James","year":"1990","journal-title":"IEEE Computer"},{"key":"10.1016\/S0065-2458(00)80003-0_bib12","series-title":"Proceedings of the 4th IEEE Symposium on Parallel and Distributed Processing","first-page":"498","article-title":"The scalable tree protocol: a cache coherence approach for large-scale multiprocessors","author":"Nilsson","year":"1992"},{"key":"10.1016\/S0065-2458(00)80003-0_bib13","series-title":"Proceedings of the 4th International Conference on Architectural Support for Programming Languages","first-page":"224","article-title":"Limit LESS directories: A scalable cache coherence solution","author":"Chaiken","year":"1991"},{"issue":"4","key":"10.1016\/S0065-2458(00)80003-0_bib14","doi-asserted-by":"crossref","first-page":"300","DOI":"10.1145\/161541.161544","article-title":"Cooperative shared memory: software and hardware for scalable multiprocessors","volume":"11","author":"Larus","year":"1993","journal-title":"ACM Transactions on Computer Systems"},{"key":"10.1016\/S0065-2458(00)80003-0_bib15","series-title":"Proceedings of the 22nd Annual International Symposium on Computer Architecture","first-page":"38","article-title":"Efficient strategies for software-only directory protocols in shared-memory multiprocessors","author":"Grahn","year":"1995"},{"key":"10.1016\/S0065-2458(00)80003-0_bib16","doi-asserted-by":"crossref","first-page":"63","DOI":"10.1109\/2.121510","article-title":"The Stanford DASH multiprocessor","author":"Lenoski","year":"1992","journal-title":"IEEE Computer"},{"key":"10.1016\/S0065-2458(00)80003-0_bib17","series-title":"Proceedings of the 22nd International Symposium on Computer Architecture","first-page":"2","article-title":"The MIT Alewife machine: architecture and performance","author":"Agarwal","year":"1995"},{"key":"10.1016\/S0065-2458(00)80003-0_bib18","series-title":"Proceedings of the 24th International Symposium on Computer Architecture","first-page":"241","article-title":"The SGI Origin 2000: A CC-NUMA highly scalable server","author":"Laudon","year":"1997"},{"key":"10.1016\/S0065-2458(00)80003-0_bib19","series-title":"Proceedings of the 23rd International Symposium on Computer Architecture","first-page":"308","article-title":"A CC-NUMA computer system for the commercial marketplace","author":"Lovett","year":"1996"},{"key":"10.1016\/S0065-2458(00)80003-0_bib20","series-title":"Proceedings of the 19th Annual International, Symposium on Computer Architecture","first-page":"80","article-title":"Comparative performance evaluation of cache-coherent NUMA and COMA architectures","author":"Stenstrom","year":"1992"},{"issue":"9","key":"10.1016\/S0065-2458(00)80003-0_bib21","doi-asserted-by":"crossref","first-page":"44","DOI":"10.1109\/2.156381","article-title":"DDM-A cache-only memory architecture","volume":"25","author":"Hagersten","year":"1992","journal-title":"IEEE Computer"},{"key":"10.1016\/S0065-2458(00)80003-0_bib22","article-title":"Overview of the KSR-1 computer system","author":"Burkhardt","year":"1992"},{"key":"10.1016\/S0065-2458(00)80003-0_bib23","series-title":"Proceedings of the 3rd International Conference on High-Performance Computer Architecture","first-page":"272","article-title":"Reducing remote conflict misses: NUMA with remote cache versus COMA","author":"Zhang","year":"1997"},{"key":"10.1016\/S0065-2458(00)80003-0_bib24","series-title":"Proceedings of the 27th Hawaii International Conference on System Sciences","first-page":"522","article-title":"Simple COMA node implementations","author":"Hagersten","year":"1994"},{"key":"10.1016\/S0065-2458(00)80003-0_bib25","series-title":"Proceedings of the 5th International Conference on High-Performance Computer Architecture","first-page":"172","article-title":"Wildfire","author":"Hagersten","year":"1999"},{"issue":"1","key":"10.1016\/S0065-2458(00)80003-0_bib26","doi-asserted-by":"crossref","first-page":"5","DOI":"10.1145\/130823.130824","article-title":"SPLASH: Stanford Parallel applications for shared memory","volume":"20","author":"Singh","year":"1992","journal-title":"Computer Architecture News"},{"key":"10.1016\/S0065-2458(00)80003-0_bib27","article-title":"Simulation of multiprocessor: accuracy and performance","author":"Goldsmith","year":"1993"},{"key":"10.1016\/S0065-2458(00)80003-0_bib28","series-title":"Proceedings of the 26th IEEE Annual Simulation Symposium","first-page":"41","article-title":"The CacheMire test bench a flexible and effective approach for simulation of multiprocessors","author":"Brorsson","year":"1993"},{"key":"10.1016\/S0065-2458(00)80003-0_bib29","series-title":"22nd Annual International Symposium on Computer Architecture","first-page":"24","article-title":"The SPLASH-2 programs: characterization and methodological considerations","author":"Woo","year":"1995"},{"key":"10.1016\/S0065-2458(00)80003-0_bib30","series-title":"Proceedings of the 6th International Conference on Architectural Support for Programming Languages and Operating Systems","first-page":"145","article-title":"Contrasting characteristics and cache performance of technical and multi-user commercial workload","author":"Grizzaffi Maynard","year":"1994"},{"key":"10.1016\/S0065-2458(00)80003-0_bib31","series-title":"Proceedings of the 3rd International Conference on High-Performance Computer Architecture","first-page":"250","article-title":"The memory performance of DSS commercial workloads in shared-memory multiprocessors","author":"Trancoso","year":"1997"},{"key":"10.1016\/S0065-2458(00)80003-0_bib32","series-title":"Proceedings of the 25th Annual International Symposium on Computer Architecture","first-page":"3","article-title":"Memory system characterization of commercial workloads","author":"Barroso","year":"1998"},{"key":"10.1016\/S0065-2458(00)80003-0_bib33","series-title":"Proceedings of the USENIX'98","first-page":"119","article-title":"SimICS\/Sun4m: a virtual workstation","author":"Magnusson","year":"1998"},{"key":"10.1016\/S0065-2458(00)80003-0_bib34","doi-asserted-by":"crossref","DOI":"10.1109\/88.473612","article-title":"Complete system simulation: the SimOS approach","author":"Rosenblum","year":"1995","journal-title":"IEEE Parallel and Distributed Technology"},{"key":"10.1016\/S0065-2458(00)80003-0_bib35","doi-asserted-by":"crossref","first-page":"51","DOI":"10.1109\/2.612249","article-title":"One billion transistors, one uniprocessor, one chip","volume":"September","author":"Patt","year":"1997","journal-title":"IEEE Computer"},{"key":"10.1016\/S0065-2458(00)80003-0_bib36","series-title":"Proceedings of the 20th Annual International Symposium on Computer Architecture","first-page":"257","article-title":"A comparison of dynamic branch predictors that use two levels of branch history","author":"Yeh","year":"1993"},{"issue":"9","key":"10.1016\/S0065-2458(00)80003-0_bib37","doi-asserted-by":"crossref","first-page":"68","DOI":"10.1109\/2.612251","article-title":"Trace processors: moving to fourth-generation microarchitectures","volume":"30","author":"Smith","year":"1997","journal-title":"IEEE Computer"},{"key":"10.1016\/S0065-2458(00)80003-0_bib38","series-title":"Proceedings of the Second International Conference on Architectural Support for Programming Languages and Operating Systems","first-page":"180","article-title":"A VLIW architecture for a trace scheduling compiler","author":"Colwell","year":"1987"},{"key":"10.1016\/S0065-2458(00)80003-0_bib39","series-title":"Proceedings of the SIGPLAN'88 Conference on Programming Language Design and Implementation","first-page":"318","article-title":"Software pipelining: an effective scheduling technique for VLIW machines","author":"Lam","year":"1988"},{"key":"10.1016\/S0065-2458(00)80003-0_bib40","series-title":"25th Annual International Symposium on Microarchitecture","first-page":"55","article-title":"An efficient resource-constrained global scheduling technique for superscalar and VLIW processors","author":"Moon","year":"1992"},{"key":"10.1016\/S0065-2458(00)80003-0_bib41","series-title":"International Symposium on Computer Architecture","first-page":"203","article-title":"Evaluation of multithreaded uniprocessors for commercial application environments","author":"Eickemeyer","year":"1996"},{"key":"10.1016\/S0065-2458(00)80003-0_bib42","first-page":"39","article-title":"An analysis of database workload performance on simultaneous multithreading processors","author":"Lo","year":"1998"},{"key":"10.1016\/S0065-2458(00)80003-0_bib43","series-title":"Proceedings of the 22nd Annual International Symposium on Computer Architecture","first-page":"392","article-title":"Simultaneous multithreading: maximizing on-chip parallelism","author":"Tullsen","year":"1995"},{"key":"10.1016\/S0065-2458(00)80003-0_bib44","series-title":"International Symposium on Computer Architecture","first-page":"191","article-title":"Exploiting choice: instruction fetch and issue on an implementable simultaneous multithreading processor","author":"Tullsen","year":"1996"},{"key":"10.1016\/S0065-2458(00)80003-0_bib45","series-title":"Proceedings of the IFIP WG 10.3 Working Conference on Parallel Architectures and Complication Techniques, PACT '95","first-page":"109","article-title":"Single-program speculative multithreading (SPSM) Architecture: compiler-assisted fine-grained multithreading","author":"Dubey","year":"1995"},{"key":"10.1016\/S0065-2458(00)80003-0_bib46","series-title":"Proceedings of the 19th Annual International Symposium on Computer Architecture","first-page":"136","article-title":"An elementary processor architecture with simultaneous instruction issuing from multiple threads","author":"Hirata","year":"1992"},{"key":"10.1016\/S0065-2458(00)80003-0_bib47","series-title":"Proceedings of the 22nd Annual International Symposium on Computer Architecture","first-page":"414","article-title":"Multiscalar processors","author":"Sohi","year":"1995"},{"key":"10.1016\/S0065-2458(00)80003-0_bib48","first-page":"881","article-title":"The superthreaded processor architecture","volume":"September","author":"Tsai","year":"1999","journal-title":"IEEE Transactions on Computers, Special Issue on Multithreaded Architectures and Systems"},{"key":"10.1016\/S0065-2458(00)80003-0_bib49","first-page":"79","article-title":"A single-chip multiprocessor","volume":"September","author":"Hammond","year":"1997","journal-title":"IEEE Computer"},{"key":"10.1016\/S0065-2458(00)80003-0_bib50","doi-asserted-by":"crossref","first-page":"86","DOI":"10.1109\/2.612254","article-title":"Baring it all to software: raw machines","volume":"September","author":"Waingold","year":"1997","journal-title":"IEEE Computer"},{"key":"10.1016\/S0065-2458(00)80003-0_bib51","series-title":"Proceedings of the 23rd Annual International Symposium on Computer Architecture","first-page":"90","article-title":"Missing the memory wall: the case for processor\/memory integration","author":"Saulsbury","year":"1996"},{"issue":"2","key":"10.1016\/S0065-2458(00)80003-0_bib52","doi-asserted-by":"crossref","first-page":"34","DOI":"10.1109\/40.592312","article-title":"A case for intelligent RAM","volume":"17","author":"Patterson","year":"1997","journal-title":"IEEE Micro"},{"issue":"6","key":"10.1016\/S0065-2458(00)80003-0_bib53","doi-asserted-by":"crossref","first-page":"45","DOI":"10.1145\/129888.129890","article-title":"Metacomputing","volume":"35","author":"Smarr","year":"1992","journal-title":"Communications of the ACM"},{"key":"10.1016\/S0065-2458(00)80003-0_bib54","first-page":"110","article-title":"Resource-aware metacomputing","volume":"53","author":"Hollingsworth","year":"1999","journal-title":"Advances in Computing"},{"key":"10.1016\/S0065-2458(00)80003-0_bib55","series-title":"Proceedings of the 26th Annual International Symposium on Computer Architecture","first-page":"294","article-title":"Multicast snooping: a new coherence method using multicast address network","author":"Ender-Bilir","year":"1999"},{"key":"10.1016\/S0065-2458(00)80003-0_bib56","series-title":"Proceedings of COMPCON","article-title":"The evolution of the HP\/Convex exemplar","author":"Brewer","year":"1997"},{"key":"10.1016\/S0065-2458(00)80003-0_bib57","series-title":"Proceedings of the 20th Annual International Symposium on Computer Architecture","first-page":"14","article-title":"Working sets, cache sizes, and node granularity issues for large-scale multiprocessors","author":"Rothberg","year":"1993"},{"issue":"7","key":"10.1016\/S0065-2458(00)80003-0_bib58","doi-asserted-by":"crossref","first-page":"794","DOI":"10.1109\/12.256449","article-title":"Cache invalidation patterns in shared-memory multiprocessors","volume":"41","author":"Gupta","year":"1992","journal-title":"IEEE Transactions on Computers"},{"issue":"7","key":"10.1016\/S0065-2458(00)80003-0_bib59","doi-asserted-by":"crossref","first-page":"63","DOI":"10.1109\/2.596630","article-title":"Boosting the performance of shared memory multiprocessors","volume":"30","author":"Stenstr\u00f6m","year":"1997","journal-title":"IEEE Computer"},{"key":"10.1016\/S0065-2458(00)80003-0_bib60","series-title":"Proceedings of the 20th Annual International Symposium on Computer Architecture","first-page":"109","article-title":"An adaptive cache coherence protocol optimized for migratory sharing","author":"Stenstr\u00f6m","year":"1993"},{"key":"10.1016\/S0065-2458(00)80003-0_bib61","series-title":"Proceedings of the 20th Annual International Symposium on Computer Architecture","first-page":"98","article-title":"Adaptive cache coherency for detecting migratory shared data","author":"Cox","year":"1993"},{"issue":"6","key":"10.1016\/S0065-2458(00)80003-0_bib62","doi-asserted-by":"crossref","first-page":"659","DOI":"10.1145\/236114.236116","article-title":"Using Dataflow analysis to reduce overhead in cache coherence protocols","volume":"18","author":"Skeppstedt","year":"1996","journal-title":"ACM Transactions on Programming Languagues and Systems"},{"issue":"2","key":"10.1016\/S0065-2458(00)80003-0_bib63","doi-asserted-by":"crossref","first-page":"168","DOI":"10.1006\/jpdc.1996.0164","article-title":"Evaluation of an adaptive update-based cache protocol","volume":"39","author":"Grahn","year":"1996","journal-title":"Journal of Parallel and Distributed Computing"},{"key":"10.1016\/S0065-2458(00)80003-0_bib64","series-title":"Proceedings of the 8th International Conference on Architectural Support for Programming Languages and Operating Systems","first-page":"307","article-title":"Performance of database workloads on shared-memory multiprocessors with out-of-order processors","author":"Ranganathan","year":"1998"},{"key":"10.1016\/S0065-2458(00)80003-0_bib65","series-title":"Proceedings of the 1999 International Conference on Parallel Processing","first-page":"246","article-title":"Improving performance of load-store sequences for transaction processing work\u2014loads on multiprocessors","author":"Nilsson","year":"1999"},{"issue":"12","key":"10.1016\/S0065-2458(00)80003-0_bib66","doi-asserted-by":"crossref","first-page":"66","DOI":"10.1109\/2.546611","article-title":"Shared memory consistency models: a tutorial","volume":"29","author":"Adve","year":"1996","journal-title":"IEEE Computer"},{"key":"10.1016\/S0065-2458(00)80003-0_bib67","series-title":"Proceedings of the 18th Annual International Symposium on Computer Architecture","first-page":"254","article-title":"Comparative evaluation of latency-reducing and tolerating techniques","author":"Gupta","year":"1991"},{"issue":"8","key":"10.1016\/S0065-2458(00)80003-0_bib68","article-title":"Multiprocessors should support simple memory consistency models","volume":"31","author":"M.","year":"1998","journal-title":"IEEE Computer"},{"key":"10.1016\/S0065-2458(00)80003-0_bib69","article-title":"Tolerating latency through software-controlled data prefetching","author":"Mowry","year":"1994"},{"issue":"4","key":"10.1016\/S0065-2458(00)80003-0_bib70","first-page":"385","article-title":"Evaluation of stride and sequential hardware-based prefetching in shared-memory multiprocessors","volume":"7","author":"Dahlgren","year":"1996","journal-title":"IEEE Transactions on Computers"},{"key":"10.1016\/S0065-2458(00)80003-0_bib71","series-title":"Proceedings of the 7th International Conference on Architectural Support for Programming Languages","first-page":"222","article-title":"Compiler-based prefetching for recursive data structures","author":"Luk","year":"1996"},{"key":"10.1016\/S0065-2458(00)80003-0_bib72","series-title":"Proceedings of the 6th International Symposium on High-Performance Computer Architecture","first-page":"206","article-title":"A prefetching technique for irregular accesses to linked data structures","author":"Karlsson","year":"2000"},{"key":"10.1016\/S0065-2458(00)80003-0_bib73","series-title":"Proceedings of the 22nd International Symposium on Computer Architecture","first-page":"392","article-title":"Simultaneous multithreading: maximizing on-chip parallelism","author":"Tullsen","year":"1995"},{"key":"10.1016\/S0065-2458(00)80003-0_bib74","series-title":"Proceedings of International Conference on Parallel Architectures and Compilation Techniques, PACT '96","first-page":"35","article-title":"The superthreaded architecture: thread pipelining with run-time data dependence checking and control speculation","author":"Tsai","year":"1996"},{"key":"10.1016\/S0065-2458(00)80003-0_bib75","series-title":"Software Versus Hardware Shared-Memory Implementation: A Case Study. International Symposium on Computer Architecture","first-page":"106","author":"Cox","year":"1994"},{"key":"10.1016\/S0065-2458(00)80003-0_bib76","doi-asserted-by":"crossref","first-page":"42","DOI":"10.1109\/2.546608","article-title":"Parallelizing a GIS on a shared address space architecture","author":"Shekhar","year":"1996","journal-title":"IEEE Computer"},{"key":"10.1016\/S0065-2458(00)80003-0_bib77","series-title":"Proceedings of Parallel CFD '95","first-page":"231","article-title":"A commercial CFD application on a shared memory multiprocessor using MPI","author":"Fischer","year":"1995"},{"issue":"9","key":"10.1016\/S0065-2458(00)80003-0_bib78_1","first-page":"1","article-title":"Parallel programming with message passing and directives","volume":"32","author":"Gabb","year":"1999","journal-title":"SIAM News, November"},{"issue":"9","key":"10.1016\/S0065-2458(00)80003-0_bib78_2","first-page":"10","article-title":"Parallel programming with message passing and directives","volume":"32","author":"Gabb","year":"1999","journal-title":"SIAM News, November"},{"key":"10.1016\/S0065-2458(00)80003-0_bib79","series-title":"Parallel programming on Silicon Graphics multiprocessor systems","author":"Chen","year":"1994"},{"key":"10.1016\/S0065-2458(00)80003-0_bib80","year":"1993"},{"key":"10.1016\/S0065-2458(00)80003-0_bib81","author":"Dowd","year":"1993"},{"key":"10.1016\/S0065-2458(00)80003-0_bib82","author":"Brawer","year":"1990"},{"key":"10.1016\/S0065-2458(00)80003-0_bib83","author":"Bauer","year":"1992"},{"key":"10.1016\/S0065-2458(00)80003-0_bib84","author":"Lenoski","year":"1995"},{"key":"10.1016\/S0065-2458(00)80003-0_bib85","article-title":"Performance of various computers using standard linear equation software","author":"Dongarra","year":"1994"},{"key":"10.1016\/S0065-2458(00)80003-0_bib86","series-title":"Parallel processing using Linux","author":"Dietz","year":"1999"},{"key":"10.1016\/S0065-2458(00)80003-0_bib87","series-title":"Proceedings of HPC-ASIA '95","article-title":"A parallel implementation of ADAMS on a shared memory multiprocessor","author":"Anderson","year":"1995"},{"key":"10.1016\/S0065-2458(00)80003-0_bib88","series-title":"Proceedings of the International Conference on High Performance Computing in Automotive Design, Engineering, and Manufacturing","article-title":"Automotive modal analysis using CSA\/Nastran on high performance computers","author":"Schartz","year":"1996"},{"key":"10.1016\/S0065-2458(00)80003-0_bib89","series-title":"High Performance Computing","first-page":"305","article-title":"A commercial CFD application on a shared memory multiprocessor","author":"Venugopal","year":"1995"},{"key":"10.1016\/S0065-2458(00)80003-0_bib90","doi-asserted-by":"crossref","first-page":"108","DOI":"10.1109\/2.762808","article-title":"OpenMP: shared-memory parallelism from the ashes","volume":"May","author":"Throop","year":"1999","journal-title":"IEEE Computer"},{"key":"10.1016\/S0065-2458(00)80003-0_bib91","author":"Open"},{"key":"10.1016\/S0065-2458(00)80003-0_bib92","series-title":"Intel Developer Forum presentation","article-title":"Software lab: developing parallel applications on IA\/NT workstations","author":"Kuhn","year":"1999"},{"key":"10.1016\/S0065-2458(00)80003-0_bib93","first-page":"44","article-title":"Seeking the balance: large SMP warehouses","volume":"August","author":"Carlile","year":"1996","journal-title":"Database Programming Design"},{"key":"10.1016\/S0065-2458(00)80003-0_bib94","series-title":"Proceedings of the 3rd Conference on Hypercube Concurrent Computers and Applications","first-page":"847","article-title":"The nCUBE family of high performance parallel computer systems","author":"Palmer","year":"1988"},{"key":"10.1016\/S0065-2458(00)80003-0_bib95","series-title":"The Berkeley NOW project","author":"Arpaci-Dusseau","year":"1998"},{"key":"10.1016\/S0065-2458(00)80003-0_bib96","author":"Lusk","year":"1994"},{"key":"10.1016\/S0065-2458(00)80003-0_bib97","author":"Boyle","year":"1987"},{"issue":"3\/4","key":"10.1016\/S0065-2458(00)80003-0_bib98","article-title":"MPI: A message-passing interface standard","volume":"8","author":"MPI","year":"1994","journal-title":"International Journal of Supercomputing Applications"},{"key":"10.1016\/S0065-2458(00)80003-0_bib99","series-title":"PAMS\u2014parallel application management system for Unix","author":"Myrias Software Corporation","year":"1999"},{"key":"10.1016\/S0065-2458(00)80003-0_bib100","series-title":"Beowulf project at CESDIS","author":"Merkey","year":"1994"},{"key":"10.1016\/S0065-2458(00)80003-0_bib101","series-title":"International Workshop on High-Level Parallel Programming Models and Supportive Environments, International Parallel Processing Symposium","article-title":"Complexity and performance in parallel programming languages","author":"VanderWiel","year":"1997"},{"key":"10.1016\/S0065-2458(00)80003-0_bib102","series-title":"What does a company with 93% of the market do for an encore?. Table 6-2","year":"1998"},{"key":"10.1016\/S0065-2458(00)80003-0_bib103","first-page":"1","article-title":"Case studies in moving multi-threaded workstation applications from RISC\/UNIX to the Intel architecture","author":"Intel Corporation","year":"1998"},{"issue":"12","key":"10.1016\/S0065-2458(00)80003-0_bib104","doi-asserted-by":"crossref","first-page":"84","DOI":"10.1109\/2.546613","article-title":"Maximizing multiprocessor performance with the SUIF compiler","volume":"29","author":"Hall","year":"1996","journal-title":"IEEE Computer"},{"key":"10.1016\/S0065-2458(00)80003-0_bib105","volume":"May","author":"Silicon Graphics Inc.","year":"1999"},{"issue":"12","key":"10.1016\/S0065-2458(00)80003-0_bib106","article-title":"Porting scientific software to Intel SMPs under Windows\/NT","volume":"14","author":"Kuhn","year":"1997","journal-title":"Scientific Computing and Automation"},{"key":"10.1016\/S0065-2458(00)80003-0_bib107","series-title":"Proceeding of the 4th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming","article-title":"Experiences using the ParaScope Editor: An Interactive Parallel Programming Tool","author":"Hall","year":"1993"},{"key":"10.1016\/S0065-2458(00)80003-0_bib108","series-title":"Proceedings of the ACM SIGPLAN Conference on Programming Language Design and Implementation","article-title":"Efficient context-sensitive pointer analysis for C programs","author":"Wilson","year":"1995"},{"key":"10.1016\/S0065-2458(00)80003-0_bib109","series-title":"Proceedings of the 26th Annual International Symposium on Computer Architecture","article-title":"Maps: a compiler-managed memory system for raw machines","author":"Barua","year":"1999"},{"key":"10.1016\/S0065-2458(00)80003-0_bib110","series-title":"Proceedings of the ACM SIGMETRICS Symposium on Parallel and Distributed Tools","article-title":"Searching for the sorting record: experiences in tuning NOW-SOrt","author":"Arpaci-Dusseau","year":"1998"},{"issue":"2","key":"10.1016\/S0065-2458(00)80003-0_bib111","doi-asserted-by":"crossref","DOI":"10.1109\/71.80132","article-title":"IPS-2: the second generation of a parallel program measurement system","volume":"1","author":"Miller","year":"1990","journal-title":"IEEE Transactions on Parallel and Distributed Systems"},{"key":"10.1016\/S0065-2458(00)80003-0_bib112","doi-asserted-by":"crossref","first-page":"38","DOI":"10.1109\/2.42013","article-title":"Visualizing performance debugging","volume":"October","author":"Lehr","year":"1989","journal-title":"IEEE Computer"},{"issue":"2","key":"10.1016\/S0065-2458(00)80003-0_bib113","doi-asserted-by":"crossref","first-page":"243","DOI":"10.1109\/12.2157","article-title":"DPM: a measurement system for distributed programs","volume":"37","author":"Miller","year":"1988","journal-title":"IEEE Transactions on Computers"},{"key":"10.1016\/S0065-2458(00)80003-0_bib114","series-title":"Proceedings of the ACM SIGMETRICS Symposium on Parallel and Distributed Tools","article-title":"The SHRIMP hardware performance monitor: design and applications","author":"Martonosi","year":"1996"},{"key":"10.1016\/S0065-2458(00)80003-0_bib115","series-title":"Proceedings of the ACM SIGMETRICS Conference on Measurement and Modeling of Computer Systems","article-title":"Integrating performance monitoring and communication in parallel computers","author":"Martonosi","year":"1996"},{"key":"10.1016\/S0065-2458(00)80003-0_bib116","series-title":"Proceedings of the ACM SIGMETRICS Symposium on Parallel and Distributed Tools","article-title":"Performance monitoring in a Myrinet-connected SHRIMP cluster","author":"Liao","year":"1998"},{"key":"10.1016\/S0065-2458(00)80003-0_bib117","series-title":"Proceedings of the 7th SIAM conference on Parallel Processing for Scientific Computing","article-title":"Interprocedural parallelization analysis: A case study","author":"Hall","year":"1995"},{"issue":"12","key":"10.1016\/S0065-2458(00)80003-0_bib118","doi-asserted-by":"crossref","first-page":"31","DOI":"10.1145\/193209.193217","article-title":"SUIF: An infrastructure for research on parallelizing and optimizing compilers","volume":"29","author":"Wilson","year":"1994","journal-title":"ACM SIGPLAN Notices"},{"key":"10.1016\/S0065-2458(00)80003-0_bib119","article-title":"Memory referencing behavior in compiler-parallelized applications. Invited paper","volume":"August","author":"Torrie","year":"1996","journal-title":"International Journal of Parallel Programming"},{"key":"10.1016\/S0065-2458(00)80003-0_bib120","article-title":"Characterizing the memory behavior of compiler-parallelized applications","volume":"December","author":"Torrie","year":"1996"},{"key":"10.1016\/S0065-2458(00)80003-0_bib121","series-title":"Proceedings of the SIGPLAN '93 Conference on Programming Language Design and Implementation","article-title":"Global optimizations for parallelism and locality on scalable parallel machines","author":"Anderson","year":"1993"},{"key":"10.1016\/S0065-2458(00)80003-0_bib122","author":"Bach","year":"1990"},{"key":"10.1016\/S0065-2458(00)80003-0_bib123","article-title":"Efficient scheduling on multiprogrammed shared-memory multi-processors","author":"Tucker","year":"1993"},{"key":"10.1016\/S0065-2458(00)80003-0_bib124","series-title":"International Parallel processing Symposium","first-page":"448","article-title":"Efficient execution of parallel applications in multiprogrammed multiprocessor systems","author":"Yue","year":"1996"},{"key":"10.1016\/S0065-2458(00)80003-0_bib125","series-title":"Distributed Computer Systems Conference","first-page":"22","article-title":"Scheduling techniques for concurrent systems","author":"Ousterhout","year":"1982"},{"key":"10.1016\/S0065-2458(00)80003-0_bib126","series-title":"Conference on Measurement and Modeling of Computer Systems","first-page":"226","article-title":"The performance of multiprogrammed multiprocessor scheduling policies","author":"Leutenegger","year":"1990"},{"key":"10.1016\/S0065-2458(00)80003-0_bib127","first-page":"174","article-title":"Impact of loop granularity and self-preemption on the performance of loop parallel applications on a multiprogrammed shared-memory multiprocessor","volume":"volume II","author":"Natarajan","year":"1994"},{"key":"10.1016\/S0065-2458(00)80003-0_bib128","series-title":"Job Scheduling Stratetgies for Parallel Processing","first-page":"45","article-title":"A scalable multi-discipline, multiple-processor scheduling framework for IRIX","volume":"949","author":"Barton","year":"1995"},{"key":"10.1016\/S0065-2458(00)80003-0_bib129","series-title":"Proceedings of Supercomputing-93","first-page":"824","article-title":"Performance analysis of job scheduling policies in parallel supercomputing environments","author":"Naik","year":"1993"},{"issue":"1","key":"10.1016\/S0065-2458(00)80003-0_bib130","doi-asserted-by":"crossref","first-page":"53","DOI":"10.1145\/146941.146944","article-title":"Schedular activations: Effective kernel support for user-level management of parallelism","volume":"10","author":"Anderson","year":"1992","journal-title":"ACM Transactions on Computer Systems"},{"key":"10.1016\/S0065-2458(00)80003-0_bib131","first-page":"24","article-title":"Automatic self-allocating threads on the convex exemplar","volume":"volume I","author":"Severance","year":"1995"},{"issue":"10","key":"10.1016\/S0065-2458(00)80003-0_bib132","doi-asserted-by":"crossref","first-page":"1235","DOI":"10.1002\/(SICI)1096-9128(19981210)10:14<1235::AID-CPE373>3.0.CO;2-Z","article-title":"Adaptive parallelism in compiler-parallelized code","volume":"14","author":"Hall","year":"1998","journal-title":"Concurrency: Practice and experience"},{"key":"10.1016\/S0065-2458(00)80003-0_bib133","series-title":"Proceedings of Supercomputing '94","first-page":"518","article-title":"An efficient algorithm for the run-time parallelization of doacross loops","author":"Chen","year":"1994"},{"key":"10.1016\/S0065-2458(00)80003-0_bib134","first-page":"726","article-title":"A scheme to enforce data dependence on large multiprocessor systems","volume":"13","author":"Zhu","year":"1987","journal-title":"IEEE Transactions on Software Engineering, June"},{"key":"10.1016\/S0065-2458(00)80003-0_bib135","series-title":"SIGPLAN Conference on Supercomputing","first-page":"33","article-title":"The privatizing doall test: a run-time technique for DOALL loop identification and array privatization","author":"Rauchwerger","year":"1994"},{"key":"10.1016\/S0065-2458(00)80003-0_bib136","series-title":"International Symposium on High-Performance Computer Architecture","first-page":"162","article-title":"Hardware for speculative run-time parallelization in distributed shared-memory multiprocessors","author":"Zhang","year":"1998"},{"key":"10.1016\/S0065-2458(00)80003-0_bib137","series-title":"International Conference on Supercomputing","first-page":"93","article-title":"Coarse-grained speculative execution in shared-memory multiprocessors","author":"Kazi","year":"1998"},{"key":"10.1016\/S0065-2458(00)80003-0_bib138","author":"Heinrich","year":"1995"},{"key":"10.1016\/S0065-2458(00)80003-0_bib139","author":"Digital","year":"1992"},{"issue":"1","key":"10.1016\/S0065-2458(00)80003-0_bib140","first-page":"119","article-title":"Internal organization of the Alpha 21164, a 300 MHz 64-bit Quad-issue CMOS RISC microprocessor","volume":"7","author":"Edmonson","year":"1995","journal-title":"Digital Technology Journal"},{"key":"10.1016\/S0065-2458(00)80003-0_bib141","first-page":"191","article-title":"Pentium secrets","volume":"July","author":"Mathisen","year":"1994","journal-title":"Byte"},{"issue":"7","key":"10.1016\/S0065-2458(00)80003-0_bib142","first-page":"107","article-title":"Informing memory operations: memory performance feedback mechanisms and their applications","volume":"May, 16","author":"Horowitz","year":"1998","journal-title":"ACM Transactions on Computer Systems"}],"container-title":["Advances in Computers","Emphasizing Distributed Systems"],"original-title":[],"link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0065245800800030?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0065245800800030?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2018,12,5]],"date-time":"2018-12-05T13:21:52Z","timestamp":1544016112000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0065245800800030"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2000]]},"references-count":143,"URL":"https:\/\/doi.org\/10.1016\/s0065-2458(00)80003-0","relation":{},"ISSN":["0065-2458"],"issn-type":[{"value":"0065-2458","type":"print"}],"subject":[],"published":{"date-parts":[[2000]]}}}