{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2022,4,3]],"date-time":"2022-04-03T19:48:28Z","timestamp":1649015308307},"reference-count":40,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2014,11,5]],"date-time":"2014-11-05T00:00:00Z","timestamp":1415145600000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"published-print":{"date-parts":[[2015,3]]},"DOI":"10.1007\/s11227-014-1325-4","type":"journal-article","created":{"date-parts":[[2014,11,4]],"date-time":"2014-11-04T19:20:42Z","timestamp":1415128842000},"page":"781-807","update-policy":"http:\/\/dx.doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["DASC-DIR: a low-overhead coherence directory for many-core processors"],"prefix":"10.1007","volume":"71","author":[{"given":"Alberto","family":"Ros","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Manuel E.","family":"Acacio","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2014,11,5]]},"reference":[{"issue":"1","key":"1325_CR1","doi-asserted-by":"crossref","first-page":"5","DOI":"10.1147\/rd.461.0005","volume":"46","author":"JM Tendler","year":"2002","unstructured":"Tendler JM, Dodson JS, Fields JS, Le H, Sinharoy B (2002) POWER4 system microarchitecture. IBM J Res Develop 46(1):5\u201325","journal-title":"IBM J Res Develop"},{"key":"1325_CR2","unstructured":"Intel Xeon Phi Coprocessor, http:\/\/software.intel.com\/en-us\/mic-developer (2013)."},{"key":"1325_CR3","doi-asserted-by":"crossref","unstructured":"Kurian G, Miller JE, Psota J, Eastep J, Liu J, Michel J, Kimerling LC, Agarwal A (2010) Atac: a 1,000-core cache-coherent processor with on-chip optical network. In: 19th international conference on parallel architectures and compilation techniques (PACT), pp 477\u2013488","DOI":"10.1145\/1854273.1854332"},{"key":"1325_CR4","doi-asserted-by":"crossref","unstructured":"Zhang M, Asanovi\u0107 K (2005) Victim replication: maximizing capacity while hiding wire delay in tiled chip multiprocessors. In: 32nd international symposium on computer architecture (ISCA), pp 336\u2013345","DOI":"10.1145\/1080695.1069998"},{"issue":"2","key":"1325_CR5","doi-asserted-by":"crossref","first-page":"7","DOI":"10.1109\/MM.2010.38","volume":"30","author":"R Kalla","year":"2010","unstructured":"Kalla R, Sinharoy B, Starke WJ, Floyd M (2010) POWER7: IBMs next-generation server processor. IEEE Micro 30(2):7\u201315","journal-title":"IEEE Micro"},{"key":"1325_CR6","doi-asserted-by":"crossref","unstructured":"Shah M, Barreh J, Brooks J, Golla R, Grohoski G, Gura N, Hetherington R, Jordan P, Luttrell M, Olson C, Saha B, Sheahan D, Spracklen L, Wynn A (2007) UltraSPARC T2: a highly-threaded, power-efficient, SPARC SoC. In: IEEE Asian solid-state circuits conference, pp 22\u201325","DOI":"10.1109\/ASSCC.2007.4425786"},{"key":"1325_CR7","doi-asserted-by":"crossref","unstructured":"Loh GH (2008) 3d-stacked memory architectures for multi-core processors. In: 35th international symposium on computer architecture (ISCA), pp 453\u2013464","DOI":"10.1109\/ISCA.2008.15"},{"key":"1325_CR8","doi-asserted-by":"crossref","unstructured":"Cho S, Jin L (2006) Managing distributed, shared L2 caches through OS-level page allocation. In: 39th IEEE\/ACM international symposium on microarchitecture (MICRO), pp 455\u2013465","DOI":"10.1109\/MICRO.2006.31"},{"key":"1325_CR9","doi-asserted-by":"crossref","unstructured":"Hardavellas N, Ferdman M, Falsafi B, Ailamaki A (2009) Reactive NUCA: near-optimal block placement and replication in distributed caches. In: 36th international symposium on computer architecture (ISCA), pp 184\u2013195","DOI":"10.1145\/1555815.1555779"},{"issue":"6","key":"1325_CR10","doi-asserted-by":"crossref","first-page":"25","DOI":"10.1109\/MM.2010.83","volume":"30","author":"CJ Hughes","year":"2010","unstructured":"Hughes CJ, Kim C, Chen Y-K (2010) Performance and energy implications of many-core caches for throughput computing. IEEE Micro 30(6):25\u201335","journal-title":"IEEE Micro"},{"key":"1325_CR11","doi-asserted-by":"crossref","unstructured":"Ros A, Cintra M, Acacio ME, Garc\u00eda JM (2009) Distance-aware round-robin mapping for large NUCA caches. In: 16th international conference on high performance computing (HiPC), pp 79\u201388","DOI":"10.1109\/HIPC.2009.5433220"},{"key":"1325_CR12","doi-asserted-by":"crossref","unstructured":"Kim C, Burger D, Keckler SW (2002) An adaptive, non-uniform cache structure for wire-delay dominated on-chip caches. In: 10th international conference on architectural support for programming language and operating systems (ASPLOS), pp 211\u2013222","DOI":"10.1145\/605397.605420"},{"key":"1325_CR13","doi-asserted-by":"crossref","unstructured":"Das S, Fan A, Chen K-N, Tan CS, Checka N, Reif R (2004) Technology, performance, and computer-aided design of three-dimensional integrated circuits. In: International symposium on physical design, pp 108\u2013115","DOI":"10.1145\/981066.981091"},{"issue":"1","key":"1325_CR14","doi-asserted-by":"crossref","first-page":"67","DOI":"10.1109\/TPDS.2005.4","volume":"16","author":"ME Acacio","year":"2005","unstructured":"Acacio ME, Gonz\u00e1lez J, Garc\u00eda JM, Duato J (2005) A two-level directory architecture for highly scalable cc-NUMA multiprocessors. IEEE Trans Parall Distrib Syst (TPDS) 16(1):67\u201379","journal-title":"IEEE Trans Parall Distrib Syst (TPDS)"},{"issue":"2","key":"1325_CR15","doi-asserted-by":"crossref","first-page":"50","DOI":"10.1109\/2.982916","volume":"35","author":"PS Magnusson","year":"2002","unstructured":"Magnusson PS, Christensson M, Eskilson J, Forsgren D, Hallberg G, Hogberg J, Larsson F, Moestedt A, Werner B (2002) Simics: a full system simulation platform. IEEE Comput 35(2):50\u201358","journal-title":"IEEE Comput"},{"issue":"4","key":"1325_CR16","doi-asserted-by":"crossref","first-page":"92","DOI":"10.1145\/1105734.1105747","volume":"33","author":"MM Martin","year":"2005","unstructured":"Martin MM, Sorin DJ, Beckmann BM, Marty MR, Xu M, Alameldeen AR, Moore KE, Hill MD, Wood DA (2005) Multifacet\u2019s general execution-driven multiprocessor simulator (GEMS) toolset. Comput Architect News 33(4):92\u201399","journal-title":"Comput Architect News"},{"key":"1325_CR17","doi-asserted-by":"crossref","unstructured":"Puente V, Gregorio JA, Beivide R (2002) SICOSYS: an integrated framework for studying interconnection network in multiprocessor systems. In: 10th Euromicro workshop on parallel, distributed and network-based processing, pp 15\u201322","DOI":"10.1109\/EMPDP.2002.994207"},{"key":"1325_CR18","unstructured":"Alameldeen AR, Wood DA (2003) Variability in architectural simulations of multi-threaded workloads. In: 9th international symposium on high-performance computer architecture (HPCA), pp 7\u201318"},{"key":"1325_CR19","doi-asserted-by":"crossref","unstructured":"Woo SC, Ohara M, Torrie E, Singh JP, Gupta A (1995) The SPLASH-2 programs: characterization and methodological considerations. In: 22nd international symposium on computer architecture (ISCA), pp 24\u201336","DOI":"10.1145\/223982.223990"},{"key":"1325_CR20","unstructured":"Gupta A, Weber W-D, Mowry TC (1990) Reducing memory traffic requirements for scalable directory-based cache coherence schemes. In: International conference on parallel processing (ICPP), pp 312\u2013321"},{"key":"1325_CR21","doi-asserted-by":"crossref","unstructured":"Chaiken D, Kubiatowicz J, Agarwal A (1991) LimitLESS directories: a scalable cache coherence scheme. In: 4th international conference on architectural support for programming language and operating systems (ASPLOS), pp 224\u2013234","DOI":"10.1145\/106972.106995"},{"key":"1325_CR22","unstructured":"Simoni R, Horowitz MA (2001) Dynamic pointer allocation for scalable cache coherence directories. In: International symposium on shared memory multiprocessing, pp 72\u201381"},{"key":"1325_CR23","doi-asserted-by":"crossref","unstructured":"Chishti Z, Powell MD, Vijaykumar TN (2003) Distance associativity for high-performance energy-efficient non-uniform cache architectures. In: 36th IEEE\/ACM international symposium on microarchitecture (MICRO), pp 55\u201366","DOI":"10.1109\/MICRO.2003.1253183"},{"key":"1325_CR24","unstructured":"Beckmann BM, Wood DA (2004) Managing wire delay in large chip-multiprocessor caches. In: 37th IEEE\/ACM international symposium on microarchitecture (MICRO), pp 319\u2013330"},{"key":"1325_CR25","doi-asserted-by":"crossref","unstructured":"Zhang M, Asanovi\u0107 K (Oct. 2005) Victim migration: dynamically adapting between private and shared CMP caches. Tech. rep, Massachusetts Institute of Technology Computer Science and Artificial Intelligence Laboratory","DOI":"10.21236\/ADA466772"},{"key":"1325_CR26","unstructured":"Lin J, Lu Q, Ding X, Zhang Z, Zhang X, Sadayappan P (2008) Gaining insights into multicore cache partitioning: Bridging the gap between simulation and real systems. In: 14th international symposium on high-performance computer architecture (HPCA), pp 367\u2013378"},{"key":"1325_CR27","doi-asserted-by":"crossref","unstructured":"Awasthi M, Sudan K, Balasubramonian R, Carter J (2009) Dynamic hardware-assisted software-controlled page placement to manage capacity allocation and sharing within large caches. In: 15th international symposium on high-performance computer architecture (HPCA), pp 250\u2013261","DOI":"10.1109\/HPCA.2009.4798260"},{"key":"1325_CR28","doi-asserted-by":"crossref","unstructured":"Chaudhuri M (2009) PageNUCA: selected policies for page-grain locality management in large shared chip-multiprocessor caches. In: 15th international symposium on high-performance computer architecture (HPCA), pp 227\u2013238","DOI":"10.1109\/HPCA.2009.4798258"},{"key":"1325_CR29","doi-asserted-by":"crossref","unstructured":"Garc\u00eda-Guirado A, Fern\u00e1ndez-Pascual R, Ros A, Garc\u00eda JM (2012) Dapsco: distance-aware partially shared cache organization. ACM Trans Architech Code Opt (TACO) 8(4), 25:1\u201325:19","DOI":"10.1145\/2086696.2086704"},{"key":"1325_CR30","doi-asserted-by":"crossref","unstructured":"Li Y, Abousamra A, Melhem R, Jones AK (2010) Compiler-assisted data distribution for chip multiprocessors. In: 19th international conference on parallel architectures and compilation techniques (PACT), pp 501\u2013512","DOI":"10.1145\/1854273.1854335"},{"key":"1325_CR31","doi-asserted-by":"crossref","unstructured":"Li Y, Melhem RG, Jones AK (2012) Practically private: enabling high performance cmps through compiler-assisted data classification. In: 21st international conference on parallel architectures and compilation techniques (PACT), pp 231\u2013240","DOI":"10.1145\/2370816.2370852"},{"key":"1325_CR32","unstructured":"Ros A, Acacio ME, Garc\u00eda JM (2008) Scalable directory organization for tiled CMP architectures. In: International conference on computer design (CDES), pp 112\u2013118"},{"key":"1325_CR33","doi-asserted-by":"crossref","unstructured":"Zebchuk J, Srinivasan V, Qureshi MK, Moshovos A (2009) A tagless coherence directory. In: 42nd IEEE\/ACM international symposium on microarchitecture (MICRO), pp 423\u2013434","DOI":"10.1145\/1669112.1669166"},{"key":"1325_CR34","doi-asserted-by":"crossref","unstructured":"Cuesta B, Ros A, G\u00f3mez ME, Robles A, Duato J (2011) Increasing the effectiveness of directory caches by deactivating coherence for private memory blocks. In: 38th international symposium on computer architecture (ISCA), pp 93\u2013103","DOI":"10.1145\/2024723.2000076"},{"key":"1325_CR35","doi-asserted-by":"crossref","unstructured":"Ferdman M, Lotfi-Kamran P, Balet K, Falsafi B (2011) Cuckoo directory: a scalable directory for many-core systems. In: 17th international symposium on high-performance computer architecture (HPCA), pp 169\u2013180","DOI":"10.1109\/HPCA.2011.5749726"},{"key":"1325_CR36","doi-asserted-by":"crossref","unstructured":"Sanchez D, Kozyrakis C (2012) SCD: a scalable coherence directory with flexible sharer set encoding. In: 18th international symposium on high-performance computer architecture (HPCA), pp 129\u2013140","DOI":"10.1109\/HPCA.2012.6168950"},{"key":"1325_CR37","doi-asserted-by":"crossref","unstructured":"Kuskin J, Ofelt D, Heinrich M, Heinlein J, Simoni R, Gharachorloo K, Chapin J, Nakahira D, Baxter J, Horowitz MA, Gupta A, Rosenblum M, Hennessy JL (1994) The stanford FLASH multiprocessor. In: 21st international symposium on computer architecture (ISCA), pp 302\u2013313","DOI":"10.1109\/ISCA.1994.288140"},{"key":"1325_CR38","doi-asserted-by":"crossref","unstructured":"Agarwal A, Bianchini R, Chaiken D, Kranz D, Kubiatowicz J, Hong Lim B, Mackenzie K, Yeung D (1995) The MIT Alewife machine: architecture and performance. In: 22nd international symposium on computer architecture (ISCA), pp 2\u201313","DOI":"10.1145\/223982.223985"},{"key":"1325_CR39","doi-asserted-by":"crossref","unstructured":"Agarwal A, Simoni R, Hennessy JL, Horowitz MA (1988) An evaluation of directory schemes for cache coherence. In: 15th international symposium on computer architecture (ISCA), pp 280\u2013289","DOI":"10.1145\/633625.52432"},{"key":"1325_CR40","unstructured":"Mukherjee SS, Hill MD (1994) An evaluation of directory protocols for medium-scale shared-memory multiprocessors. In: 8th international conference on supercomputing (ICS), pp 64\u201374"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-014-1325-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11227-014-1325-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-014-1325-4","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,8,17]],"date-time":"2019-08-17T01:13:38Z","timestamp":1566004418000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11227-014-1325-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2014,11,5]]},"references-count":40,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2015,3]]}},"alternative-id":["1325"],"URL":"https:\/\/doi.org\/10.1007\/s11227-014-1325-4","relation":{},"ISSN":["0920-8542","1573-0484"],"issn-type":[{"value":"0920-8542","type":"print"},{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2014,11,5]]}}}