{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T04:20:05Z","timestamp":1750220405123,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":42,"publisher":"ACM","license":[{"start":{"date-parts":[[2021,10,17]],"date-time":"2021-10-17T00:00:00Z","timestamp":1634428800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Vetenskapsradet","award":["2018-05254"],"award-info":[{"award-number":["2018-05254"]}]},{"DOI":"10.13039\/501100000781","name":"European Research Council","doi-asserted-by":"publisher","award":["819134"],"award-info":[{"award-number":["819134"]}],"id":[{"id":"10.13039\/501100000781","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Spanish MCIU, AEI, and European Commission FEDER funds","award":["RTI2018-098156-B-C53"],"award-info":[{"award-number":["RTI2018-098156-B-C53"]}]},{"name":"Spanish MCIU - Juan de la Cierva","award":["FJC2018-036021-I"],"award-info":[{"award-number":["FJC2018-036021-I"]}]},{"name":"European joint Effort toward a Highly Productive Programming Environment for Heterogeneous Exascale Computing (EPEEC)","award":["801051"],"award-info":[{"award-number":["801051"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2021,10,18]]},"DOI":"10.1145\/3466752.3480086","type":"proceedings-article","created":{"date-parts":[[2021,10,17]],"date-time":"2021-10-17T19:12:05Z","timestamp":1634497925000},"page":"1296-1308","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["ITSLF: Inter-Thread Store-to-Load Forwardingin Simultaneous Multithreading"],"prefix":"10.1145","author":[{"given":"Josu\u00e9","family":"Feliu","sequence":"first","affiliation":[{"name":"Universidad de Murcia, Spain"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Alberto","family":"Ros","sequence":"additional","affiliation":[{"name":"Universidad de Murcia, Spain"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Manuel E.","family":"Acacio","sequence":"additional","affiliation":[{"name":"Universidad de Murcia, Spain"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Stefanos","family":"Kaxiras","sequence":"additional","affiliation":[{"name":"Uppsala University, Sweden"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2021,10,17]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/2.546611"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISPASS.2009.4919636"},{"key":"e_1_3_2_1_3_1","unstructured":"Khary\u00a0J. Alexander Christian\u00a0Jacobi Jonathan T.\u00a0Hsieh and Martin Recktenwald. 2014. Load and Store Ordering for a Strongly Ordered Simultaneous Multithreading Core. U.S. Patent US14511408.  Khary\u00a0J. Alexander Christian\u00a0Jacobi Jonathan T.\u00a0Hsieh and Martin Recktenwald. 2014. Load and Store Ordering for a Strongly Ordered Simultaneous Multithreading Core. U.S. Patent US14511408."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511811258"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC.2009.5306792"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/1454115.1454128"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/1555754.1555785"},{"volume-title":"2008 Conf. on Programming Language Design and Implementation (PLDI). 68\u201378","author":"J.","key":"e_1_3_2_1_9_1","unstructured":"Hans- J. Boehm and Sarita\u00a0V. Adve. 2008. Foundations of the C++ Concurrency Memory Model . In 2008 Conf. on Programming Language Design and Implementation (PLDI). 68\u201378 . Hans-J. Boehm and Sarita\u00a0V. Adve. 2008. Foundations of the C++ Concurrency Memory Model. In 2008 Conf. on Programming Language Design and Implementation (PLDI). 68\u201378."},{"key":"e_1_3_2_1_10_1","volume-title":"An Oldest-First Selection Logic Implementation for Non-Compacting Issue Queues. In 15th Annual Int\u2019l ASIC\/SOC Conference. 31\u201335","author":"Buyuktosunoglu Alper","year":"2002","unstructured":"Alper Buyuktosunoglu , Ali El-Moursy , and David\u00a0 H. Albonesi . 2002 . An Oldest-First Selection Logic Implementation for Non-Compacting Issue Queues. In 15th Annual Int\u2019l ASIC\/SOC Conference. 31\u201335 . Alper Buyuktosunoglu, Ali El-Moursy, and David\u00a0H. Albonesi. 2002. An Oldest-First Selection Logic Implementation for Non-Compacting Issue Queues. In 15th Annual Int\u2019l ASIC\/SOC Conference. 31\u201335."},{"key":"e_1_3_2_1_11_1","volume-title":"Sniper: Exploring the Level of Abstraction for Scalable and Accurate Parallel Multi-Core Simulations. In Conf. on Supercomputing (SC). 52:1\u201352:12","author":"Carlson E.","year":"2011","unstructured":"Trevor\u00a0 E. Carlson , Wim Heirman , and Lieven Eeckhout . 2011 . Sniper: Exploring the Level of Abstraction for Scalable and Accurate Parallel Multi-Core Simulations. In Conf. on Supercomputing (SC). 52:1\u201352:12 . Trevor\u00a0E. Carlson, Wim Heirman, and Lieven Eeckhout. 2011. Sniper: Exploring the Level of Abstraction for Scalable and Accurate Parallel Multi-Core Simulations. In Conf. on Supercomputing (SC). 52:1\u201352:12."},{"volume-title":"25th Int\u2019l Symp. on Computer Architecture (ISCA). 142\u2013153","author":"Z.","key":"e_1_3_2_1_12_1","unstructured":"George\u00a0 Z. Chrysos and Joel\u00a0S. Emer. 1998. Memory Dependence Prediction using Store Sets . In 25th Int\u2019l Symp. on Computer Architecture (ISCA). 142\u2013153 . George\u00a0Z. Chrysos and Joel\u00a0S. Emer. 1998. Memory Dependence Prediction using Store Sets. In 25th Int\u2019l Symp. on Computer Architecture (ISCA). 142\u2013153."},{"key":"e_1_3_2_1_13_1","volume-title":"Non-Volatile Memories. In 16th Int\u2019l Conf. on Architectural Support for Programming Language and Operating Systems (ASPLOS). 105\u2013118","author":"Coburn Joel","year":"2011","unstructured":"Joel Coburn , Adrian\u00a0 M. Caulfield , Ameen Akel , Laura\u00a0 M. Grupp , Rajesh\u00a0 K. Gupta , Ranjit Jhala , and Steven Swanson . 2011 . NV-Heaps: Making Persistent Objects Fast and Safe with next-Generation , Non-Volatile Memories. In 16th Int\u2019l Conf. on Architectural Support for Programming Language and Operating Systems (ASPLOS). 105\u2013118 . Joel Coburn, Adrian\u00a0M. Caulfield, Ameen Akel, Laura\u00a0M. Grupp, Rajesh\u00a0K. Gupta, Ranjit Jhala, and Steven Swanson. 2011. NV-Heaps: Making Persistent Objects Fast and Safe with next-Generation, Non-Volatile Memories. In 16th Int\u2019l Conf. on Architectural Support for Programming Language and Operating Systems (ASPLOS). 105\u2013118."},{"key":"e_1_3_2_1_14_1","first-page":"3","article-title":"The next-generation Intel core microarchitecture","volume":"14","author":"Dixon Martin","year":"2010","unstructured":"Martin Dixon , Per Hammarlund , Stephan Jourdan , and Ronak Singhal . 2010 . The next-generation Intel core microarchitecture . Intel Technology Journal 14 , 3 (March 2010), 8\u201328. Martin Dixon, Per Hammarlund, Stephan Jourdan, and Ronak Singhal. 2010. The next-generation Intel core microarchitecture. Intel Technology Journal 14, 3 (March 2010), 8\u201328.","journal-title":"Intel Technology Journal"},{"volume-title":"Parallel Computer Organization and Design","author":"Dubois Michel","key":"e_1_3_2_1_15_1","unstructured":"Michel Dubois , Murali Annavaram , and Per Stenstr\u00f6m . 2012. Parallel Computer Organization and Design . Cambridge University Press . Michel Dubois, Murali Annavaram, and Per Stenstr\u00f6m. 2012. Parallel Computer Organization and Design. Cambridge University Press."},{"key":"e_1_3_2_1_16_1","unstructured":"Agner Fog. 2021. The microarchitecture of Intel AMD and VIA CPUs: An optimization guide for assembly programmers and compiler makers. https:\/\/www.agner.org\/optimize\/microarchitecture.pdf.  Agner Fog. 2021. The microarchitecture of Intel AMD and VIA CPUs: An optimization guide for assembly programmers and compiler makers. https:\/\/www.agner.org\/optimize\/microarchitecture.pdf."},{"key":"e_1_3_2_1_17_1","unstructured":"Andrei Frumusanu. 2020. Apple Announces The Apple Silicon M1: Ditching x86 - What to Expect Based on A14. https:\/\/www.anandtech.com\/show\/16226\/apple-silicon-m1-a14-deep-dive\/2.  Andrei Frumusanu. 2020. Apple Announces The Apple Silicon M1: Ditching x86 - What to Expect Based on A14. https:\/\/www.anandtech.com\/show\/16226\/apple-silicon-m1-a14-deep-dive\/2."},{"key":"e_1_3_2_1_18_1","unstructured":"Kourosh Gharachorloo. 1995. Memory Consistency Models for Shared-Memory Multiprocessors. Research report 95\/9. Western Research Laboratory.  Kourosh Gharachorloo. 1995. Memory Consistency Models for Shared-Memory Multiprocessors. Research report 95\/9. Western Research Laboratory."},{"key":"e_1_3_2_1_19_1","volume-title":"20th Int\u2019l Conf. on Parallel Processing (ICPP). 355\u2013364","author":"Gharachorloo Kourosh","year":"1991","unstructured":"Kourosh Gharachorloo , Anoop Gupta , and John Hennessy . 1991 . Two Techniques to Enhance the Performance of Memory Consistency Models . In 20th Int\u2019l Conf. on Parallel Processing (ICPP). 355\u2013364 . Kourosh Gharachorloo, Anoop Gupta, and John Hennessy. 1991. Two Techniques to Enhance the Performance of Memory Consistency Models. In 20th Int\u2019l Conf. on Parallel Processing (ICPP). 355\u2013364."},{"key":"e_1_3_2_1_20_1","volume-title":"SynCron: Efficient Synchronization Support for Near-Data-Processing Architectures. 27th Int\u2019l Symp. on High-Performance Computer Architecture (HPCA) (Feb.","author":"Giannoula Christina","year":"2021","unstructured":"Christina Giannoula , Nandita Vijaykumar , Nikela Papadopoulou , Vasileios Karakostas , Ivan Fernandez , Juan G\u00f3mez-Luna , Lois Orosa , Nectarios Koziris , Georgios Goumas , and Onur Mutlu . 2021 . SynCron: Efficient Synchronization Support for Near-Data-Processing Architectures. 27th Int\u2019l Symp. on High-Performance Computer Architecture (HPCA) (Feb. 2021). Christina Giannoula, Nandita Vijaykumar, Nikela Papadopoulou, Vasileios Karakostas, Ivan Fernandez, Juan G\u00f3mez-Luna, Lois Orosa, Nectarios Koziris, Georgios Goumas, and Onur Mutlu. 2021. SynCron: Efficient Synchronization Support for Near-Data-Processing Architectures. 27th Int\u2019l Symp. on High-Performance Computer Architecture (HPCA) (Feb. 2021)."},{"volume-title":"26th Int\u2019l Symp. on Computer Architecture (ISCA). 162\u2013171","author":"Gniady K.","key":"e_1_3_2_1_21_1","unstructured":"K. Gniady , B. Falsafi , and T. Vijaykumar . 1999. Is SC + ILP = RC? . In 26th Int\u2019l Symp. on Computer Architecture (ISCA). 162\u2013171 . K. Gniady, B. Falsafi, and T. Vijaykumar. 1999. Is SC + ILP = RC?. In 26th Int\u2019l Symp. on Computer Architecture (ISCA). 162\u2013171."},{"key":"e_1_3_2_1_22_1","volume-title":"Persistency for Synchronization-Free Regions. In 39th Conf. on Programming Language Design and Implementation (PLDI). 46\u201361","author":"Gogte Vaibhav","year":"2018","unstructured":"Vaibhav Gogte , Stephan Diestelhorst , William Wang , Satish Narayanasamy , Peter\u00a0 M. Chen , and Thomas\u00a0 F. Wenisch . 2018 . Persistency for Synchronization-Free Regions. In 39th Conf. on Programming Language Design and Implementation (PLDI). 46\u201361 . Vaibhav Gogte, Stephan Diestelhorst, William Wang, Satish Narayanasamy, Peter\u00a0M. Chen, and Thomas\u00a0F. Wenisch. 2018. Persistency for Synchronization-Free Regions. In 39th Conf. on Programming Language Design and Implementation (PLDI). 46\u201361."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/L-CA.2010.8"},{"key":"e_1_3_2_1_24_1","unstructured":"Intel. 2016. Intel\u00ae 64 and IA-32 Architectures Optimization Reference Manual. www.intel.com.  Intel. 2016. Intel\u00ae 64 and IA-32 Architectures Optimization Reference Manual. www.intel.com."},{"key":"e_1_3_2_1_25_1","volume-title":"Language-Level Persistency. In 44th Int\u2019l Symp. on Computer Architecture (ISCA). 481\u2013493","author":"Kolli Aasheesh","year":"2017","unstructured":"Aasheesh Kolli , Vaibhav Gogte , Ali Saidi , Stephan Diestelhorst , Peter\u00a0 M. Chen , Satish Narayanasamy , and Thomas\u00a0 F. Wenisch . 2017 . Language-Level Persistency. In 44th Int\u2019l Symp. on Computer Architecture (ISCA). 481\u2013493 . Aasheesh Kolli, Vaibhav Gogte, Ali Saidi, Stephan Diestelhorst, Peter\u00a0M. Chen, Satish Narayanasamy, and Thomas\u00a0F. Wenisch. 2017. Language-Level Persistency. In 44th Int\u2019l Symp. on Computer Architecture (ISCA). 481\u2013493."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCAD.2011.6105405"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/2749469.2750396"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/1105734.1105747"},{"volume-title":"A Primer on Memory Consistency and Cache Coherence","author":"Nagarajan Vijay","key":"e_1_3_2_1_29_1","unstructured":"Vijay Nagarajan , Daniel\u00a0 J. Sorin , Mark\u00a0 D. Hill , and David\u00a0 A. Wood . 2020. A Primer on Memory Consistency and Cache Coherence , Second Edition. Morgan & Claypool Publishers . Vijay Nagarajan, Daniel\u00a0J. Sorin, Mark\u00a0D. Hill, and David\u00a0A. Wood. 2020. A Primer on Memory Consistency and Cache Coherence, Second Edition. Morgan & Claypool Publishers."},{"key":"e_1_3_2_1_30_1","unstructured":"Simo Neuvonen Antoni Wolski Markku Manner and Vilho Raatikka. 2011. Telecom Application Transaction Processing Benchmark. http:\/\/tatpbenchmark.sourceforge.net\/. http:\/\/tatpbenchmark.sourceforge.net\/  Simo Neuvonen Antoni Wolski Markku Manner and Vilho Raatikka. 2011. Telecom Application Transaction Processing Benchmark. http:\/\/tatpbenchmark.sourceforge.net\/. http:\/\/tatpbenchmark.sourceforge.net\/"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/HCS49909.2020.9220434"},{"key":"e_1_3_2_1_32_1","volume-title":"Memory Persistency. In 41st Int\u2019l Symp. on Computer Architecture (ISCA). 265\u2013276","author":"Pelley Steven","year":"2014","unstructured":"Steven Pelley , Peter\u00a0 M. Chen , and Thomas\u00a0 F. Wenisch . 2014 . Memory Persistency. In 41st Int\u2019l Symp. on Computer Architecture (ISCA). 265\u2013276 . Steven Pelley, Peter\u00a0M. Chen, and Thomas\u00a0F. Wenisch. 2014. Memory Persistency. In 41st Int\u2019l Symp. on Computer Architecture (ISCA). 265\u2013276."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3158107"},{"key":"e_1_3_2_1_34_1","volume-title":"Speculative Enforcement of Store Atomicity. In 53rd Int\u2019l Symp. on Microarchitecture (MICRO). 555\u2013567","author":"Ros Alberto","year":"2020","unstructured":"Alberto Ros and Stefanos Kaxiras . 2020 . Speculative Enforcement of Store Atomicity. In 53rd Int\u2019l Symp. on Microarchitecture (MICRO). 555\u2013567 . Alberto Ros and Stefanos Kaxiras. 2020. Speculative Enforcement of Store Atomicity. In 53rd Int\u2019l Symp. on Microarchitecture (MICRO). 555\u2013567."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/3307650.3322216"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISPASS.2016.7482078"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.5555\/2534458"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/1785414.1785443"},{"key":"e_1_3_2_1_39_1","article-title":"The L-TAGE Branch Predictor","author":"Seznec Andr\u00e9","year":"2007","unstructured":"Andr\u00e9 Seznec . 2007 . The L-TAGE Branch Predictor . The Journal of Instruction-Level Parallelism 9 ( May 2007), 1\u201313. Andr\u00e9 Seznec. 2007. The L-TAGE Branch Predictor. The Journal of Instruction-Level Parallelism 9 (May 2007), 1\u201313.","journal-title":"The Journal of Instruction-Level Parallelism 9"},{"key":"e_1_3_2_1_40_1","unstructured":"Transaction Processing Performance Council (TPC). 2010. TPC Benchmark B. http:\/\/www.tpc.org\/tpc_documents_current_versions\/pdf\/tpc-c_v5-11.pdf. http:\/\/www.tpc.org\/tpc_documents_current_versions\/pdf\/tpc-c_v5-11.pdf  Transaction Processing Performance Council (TPC). 2010. TPC Benchmark B. http:\/\/www.tpc.org\/tpc_documents_current_versions\/pdf\/tpc-c_v5-11.pdf. http:\/\/www.tpc.org\/tpc_documents_current_versions\/pdf\/tpc-c_v5-11.pdf"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3037697.3037719"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/223982.224449"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/223982.223990"}],"event":{"name":"MICRO '21: 54th Annual IEEE\/ACM International Symposium on Microarchitecture","sponsor":["SIGMICRO ACM Special Interest Group on Microarchitectural Research and Processing"],"location":"Virtual Event Greece","acronym":"MICRO '21"},"container-title":["MICRO-54: 54th Annual IEEE\/ACM International Symposium on Microarchitecture"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3466752.3480086","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3466752.3480086","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T20:18:56Z","timestamp":1750191536000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3466752.3480086"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,10,17]]},"references-count":42,"alternative-id":["10.1145\/3466752.3480086","10.1145\/3466752"],"URL":"https:\/\/doi.org\/10.1145\/3466752.3480086","relation":{},"subject":[],"published":{"date-parts":[[2021,10,17]]},"assertion":[{"value":"2021-10-17","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}