{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T05:28:29Z","timestamp":1782970109685,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":53,"publisher":"ACM","license":[{"start":{"date-parts":[[2019,11,17]],"date-time":"2019-11-17T00:00:00Z","timestamp":1573948800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"European Science Foundation"},{"name":"European Research Council (ERC)","award":["678880, 732631"],"award-info":[{"award-number":["678880, 732631"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2019,11,17]]},"DOI":"10.1145\/3295500.3356189","type":"proceedings-article","created":{"date-parts":[[2019,11,7]],"date-time":"2019-11-07T19:43:22Z","timestamp":1573155802000},"page":"1-14","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":19,"title":["Network-accelerated non-contiguous memory transfers"],"prefix":"10.1145","author":[{"given":"Salvatore","family":"Di Girolamo","sequence":"first","affiliation":[{"name":"ETH Z\u00fcrich, Z\u00fcrich, Switzerland and Cray UK Ltd."}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Konstantin","family":"Taranov","sequence":"additional","affiliation":[{"name":"ETH Z\u00fcrich, Z\u00fcrich, Switzerland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Andreas","family":"Kurth","sequence":"additional","affiliation":[{"name":"ETH Z\u00fcrich, Z\u00fcrich, Switzerland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Michael","family":"Schaffner","sequence":"additional","affiliation":[{"name":"ETH Z\u00fcrich, Z\u00fcrich, Switzerland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Timo","family":"Schneider","sequence":"additional","affiliation":[{"name":"ETH Z\u00fcrich, Z\u00fcrich, Switzerland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jakub","family":"Ber\u00e1nek","sequence":"additional","affiliation":[{"name":"Technical University of Ostrava"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Maciej","family":"Besta","sequence":"additional","affiliation":[{"name":"ETH Z\u00fcrich, Z\u00fcrich, Switzerland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Luca","family":"Benini","sequence":"additional","affiliation":[{"name":"ETH Z\u00fcrich, Z\u00fcrich, Switzerland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Duncan","family":"Roweth","sequence":"additional","affiliation":[{"name":"Cray UK Ltd."}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Torsten","family":"Hoefler","sequence":"additional","affiliation":[{"name":"ETH Z\u00fcrich, Z\u00fcrich, Switzerland"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2019,11,17]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Infiniband Network Architecture","author":"Shanley T.","unstructured":"T. Shanley . 2003. Infiniband Network Architecture . Addison-Wesley Professional . T. Shanley. 2003. Infiniband Network Architecture. Addison-Wesley Professional."},{"key":"e_1_3_2_1_2_1","unstructured":"2019. Mellanox Technologies. http:\/\/http:\/\/www.mellanox.com\/. (2019).  2019. Mellanox Technologies. http:\/\/http:\/\/www.mellanox.com\/. (2019)."},{"key":"e_1_3_2_1_3_1","unstructured":"B. Alverson etal 2012. Cray XC series network. Cray Inc. White Paper WP-Aries01-1112 (2012).  B. Alverson et al. 2012. Cray XC series network. Cray Inc. White Paper WP-Aries01-1112 (2012)."},{"key":"e_1_3_2_1_4_1","volume-title":"Sandia National Laboratories","author":"Barrett B. W","year":"2018","unstructured":"B. W Barrett , 2018 . The Portals 4.2 Network Programming Interface . Sandia National Laboratories , November 2012, Technical Report SAND 2012-10087 (2018). B. W Barrett, et al. 2018. The Portals 4.2 Network Programming Interface. Sandia National Laboratories, November 2012, Technical Report SAND2012-10087 (2018)."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICPP.2013.73"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2016.56"},{"key":"e_1_3_2_1_7_1","first-page":"2012","article-title":"Micro-Applications for Communication Data Access Patterns and MPI Datatypes. In Recent Advances in the Message Passing Interface - Proceedings of the 19th European MPI Users' Group Meeting","volume":"2012","author":"Schneider T.","year":"2012","unstructured":"T. Schneider , R. Gerstenberger , and T. Hoefler . 2012 . Micro-Applications for Communication Data Access Patterns and MPI Datatypes. In Recent Advances in the Message Passing Interface - Proceedings of the 19th European MPI Users' Group Meeting , EuroMPI 2012 , 2012 ., Vol. 7490. Springer, 121--131. T. Schneider, R. Gerstenberger, and T. Hoefler. 2012. Micro-Applications for Communication Data Access Patterns and MPI Datatypes. In Recent Advances in the Message Passing Interface - Proceedings of the 19th European MPI Users' Group Meeting, EuroMPI 2012, 2012., Vol. 7490. Springer, 121--131.","journal-title":"EuroMPI"},{"key":"e_1_3_2_1_8_1","first-page":"4","article-title":"Application-oriented ping-pong benchmarking: how to assess the real communication overheads","volume":"96","author":"Schneider T.","year":"2014","unstructured":"T. Schneider , R. Gerstenberger , and T. Hoefler . 2014 . Application-oriented ping-pong benchmarking: how to assess the real communication overheads . Journal of Computing 96 , 4 (Apr. 2014), 279--292. T. Schneider, R. Gerstenberger, and T. Hoefler. 2014. Application-oriented ping-pong benchmarking: how to assess the real communication overheads. Journal of Computing 96, 4 (Apr. 2014), 279--292.","journal-title":"Journal of Computing"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"crossref","unstructured":"T. Hoefler and S. Gottlieb. 2010. Parallel Zero-Copy Algorithms for Fast Fourier Transform and Conjugate Gradient using MPI Datatypes. In Recent Advances in the Message Passing Interface (EuroMPI'10) Vol. LNCS 6305. Springer 132--141.  T. Hoefler and S. Gottlieb. 2010. Parallel Zero-Copy Algorithms for Fast Fourier Transform and Conjugate Gradient using MPI Datatypes. In Recent Advances in the Message Passing Interface (EuroMPI'10) Vol. LNCS 6305. Springer 132--141.","DOI":"10.1007\/978-3-642-15646-5_14"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"crossref","unstructured":"W. Gropp etal 2011. Performance Expectations and Guidelines for MPI Derived Datatypes. In Recent Advances in the Message Passing Interface (EuroMPI'11) Vol. 6960. Springer 150--159.  W. Gropp et al. 2011. Performance Expectations and Guidelines for MPI Derived Datatypes. In Recent Advances in the Message Passing Interface (EuroMPI'11) Vol. 6960. Springer 150--159.","DOI":"10.1007\/978-3-642-24449-0_18"},{"key":"e_1_3_2_1_11_1","volume-title":"Proceedings of the 20th European MPI Users' Group Meeting. ACM, 19--24","author":"Schneider T.","unstructured":"T. Schneider , F. Kjolstad , and T. Hoefler . 2013. MPI Datatype Processing using Runtime Compilation . In Proceedings of the 20th European MPI Users' Group Meeting. ACM, 19--24 . T. Schneider, F. Kjolstad, and T. Hoefler. 2013. MPI Datatype Processing using Runtime Compilation. In Proceedings of the 20th European MPI Users' Group Meeting. ACM, 19--24."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-30218-6_14"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CLUSTER.2011.42"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3126908.3126970"},{"key":"e_1_3_2_1_15_1","volume-title":"F Van der Wijngaart and P. Wong","author":"R.","year":"2002","unstructured":"R. F Van der Wijngaart and P. Wong . 2002 . NAS parallel benchmarks version 2.4. (2002). R. F Van der Wijngaart and P. Wong. 2002. NAS parallel benchmarks version 2.4. (2002)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1177\/1094342006064504"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/2020373.2020375"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/1809961.1809969"},{"key":"e_1_3_2_1_19_1","volume-title":"Proceedings of the 2006 ACM\/IEEE conference on Supercomputing. ACM, 27","author":"El-Ghazawi T.","unstructured":"T. El-Ghazawi and L. Smith . 2006. UPC: unified parallel C . In Proceedings of the 2006 ACM\/IEEE conference on Supercomputing. ACM, 27 . T. El-Ghazawi and L. Smith. 2006. UPC: unified parallel C. In Proceedings of the 2006 ACM\/IEEE conference on Supercomputing. ACM, 27."},{"key":"e_1_3_2_1_20_1","volume-title":"MPI: A Message-Passing Interface Standard Version 3.0. (09","author":"Interface Forum Message Passing","year":"2012","unstructured":"Message Passing Interface Forum . 2012 . MPI: A Message-Passing Interface Standard Version 3.0. (09 2012). Chapter author for Collective Communication, Process Topologies, and One Sided Communications . Message Passing Interface Forum. 2012. MPI: A Message-Passing Interface Standard Version 3.0. (09 2012). Chapter author for Collective Communication, Process Topologies, and One Sided Communications."},{"key":"e_1_3_2_1_21_1","volume-title":"Proceedings of the Third MPI Developer's and User's Conference. MPI Software Technology Press, 25--30","author":"Gropp W.","unstructured":"W. Gropp , E. Lusk , and D. Swider . 1999. Improving the performance of MPI derived datatypes . In Proceedings of the Third MPI Developer's and User's Conference. MPI Software Technology Press, 25--30 . W. Gropp, E. Lusk, and D. Swider. 1999. Improving the performance of MPI derived datatypes. In Proceedings of the Third MPI Developer's and User's Conference. MPI Software Technology Press, 25--30."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1007\/11846802_36"},{"key":"e_1_3_2_1_23_1","volume-title":"European Parallel Virtual Machine\/Message Passing Interface Users' Group Meeting. Springer, 324--325","author":"Tanabe N.","unstructured":"N. Tanabe and H. Nakajo . 2008. Introduction to acceleration for MPI derived datatypes using an enhancer of memory and network . In European Parallel Virtual Machine\/Message Passing Interface Users' Group Meeting. Springer, 324--325 . N. Tanabe and H. Nakajo. 2008. Introduction to acceleration for MPI derived datatypes using an enhancer of memory and network. In European Parallel Virtual Machine\/Message Passing Interface Users' Group Meeting. Springer, 324--325."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/2642769.2642771"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-03770-2_11"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-39924-7_55"},{"key":"e_1_3_2_1_27_1","volume-title":"HERO: Heterogeneous embedded research platform for exploring RISC-V manycore accelerators on FPGA. arXiv preprint arXiv:1712.06497","author":"Kurth A.","year":"2017","unstructured":"A. Kurth , 2017 . HERO: Heterogeneous embedded research platform for exploring RISC-V manycore accelerators on FPGA. arXiv preprint arXiv:1712.06497 (2017). A. Kurth, et al. 2017. HERO: Heterogeneous embedded research platform for exploring RISC-V manycore accelerators on FPGA. arXiv preprint arXiv:1712.06497 (2017)."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2017.3711645"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/TVLSI.2017.2654506"},{"key":"e_1_3_2_1_30_1","unstructured":"Mellanox Technologies. 2019. Mellanox BlueField SmartNIC. http:\/\/www.mellanox.com\/related-docs\/prod_adapter_cards\/PB_BlueField_Smart_NIC.pdf. (2019). Online; accessed 05. April 2019.  Mellanox Technologies. 2019. Mellanox BlueField SmartNIC. http:\/\/www.mellanox.com\/related-docs\/prod_adapter_cards\/PB_BlueField_Smart_NIC.pdf. (2019). Online; accessed 05. April 2019."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISSCC.2016.7417914"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISSCC.2015.7063105"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/LCN.2010.5735719"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.4018\/jdst.2010040104"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/2024716.2024718"},{"key":"e_1_3_2_1_36_1","volume-title":"2014 International Conference on Embedded Computer Systems: Architectures, Modeling, and Simulation (SAMOS XIV). IEEE, 266--273","author":"Endo F. A","unstructured":"F. A Endo , D. Courouss\u00e9 , and H. Charles . 2014. Micro-architectural simulation of in-order and out-of-order arm microprocessors with gem5 . In 2014 International Conference on Embedded Computer Systems: Architectures, Modeling, and Simulation (SAMOS XIV). IEEE, 266--273 . F. A Endo, D. Courouss\u00e9, and H. Charles. 2014. Micro-architectural simulation of in-order and out-of-order arm microprocessors with gem5. In 2014 International Conference on Embedded Computer Systems: Architectures, Modeling, and Simulation (SAMOS XIV). IEEE, 266--273."},{"key":"e_1_3_2_1_37_1","unstructured":"A. Tousi and C. Zhu. 2017. Arm Research Starter Kit: System Modeling using gem5. (2017).  A. Tousi and C. Zhu. 2017. Arm Research Starter Kit: System Modeling using gem5. (2017)."},{"key":"e_1_3_2_1_38_1","volume-title":"Proceedings of the 19th ACM International Symposium on High Performance Distributed Computing. ACM, 597--604","author":"Hoefler T.","unstructured":"T. Hoefler , T. Schneider , and A. Lumsdaine . 2010. LogGOPSim - Simulating Large-Scale Applications in the LogGOPS Model . In Proceedings of the 19th ACM International Symposium on High Performance Distributed Computing. ACM, 597--604 . T. Hoefler, T. Schneider, and A. Lumsdaine. 2010. LogGOPSim - Simulating Large-Scale Applications in the LogGOPS Model. In Proceedings of the 19th ACM International Symposium on High Performance Distributed Computing. ACM, 597--604."},{"key":"e_1_3_2_1_39_1","unstructured":"Lawrence Livermore National Laboratory. 2018. Comb is a communication performance benchmarking tool. (2018). https:\/\/github.com\/LLNL\/Comb  Lawrence Livermore National Laboratory. 2018. Comb is a communication performance benchmarking tool. (2018). https:\/\/github.com\/LLNL\/Comb"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1006\/jcph.1995.1039"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1177\/109434209100500406"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.5555\/1413370.1413432"},{"key":"e_1_3_2_1_43_1","volume-title":"SW4 final report for iCOE","author":"Sjogreen B","unstructured":"B Sjogreen . 2018. SW4 final report for iCOE . Technical Report. Lawrence Livermore National Lab.(LLNL), Livermore, CA (United States) . B Sjogreen. 2018. SW4 final report for iCOE. Technical Report. Lawrence Livermore National Lab.(LLNL), Livermore, CA (United States)."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jcp.2007.01.037"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3230543.3230560"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.5555\/3014904.3014989"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICPP.2009.70"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"crossref","unstructured":"J. L. Tr\u00e4ff etal 1999. Flattening on the Fly: efficient handling of MPI derived datatypes. In Recent Advances in Parallel Virtual Machine and Message Passing Interface Jack Dongarra Emilio Luque and Tom\u00e0s Margalef (Eds.). Springer Berlin Heidelberg Berlin Heidelberg 109--116.  J. L. Tr\u00e4ff et al. 1999. Flattening on the Fly: efficient handling of MPI derived datatypes. In Recent Advances in Parallel Virtual Machine and Message Passing Interface Jack Dongarra Emilio Luque and Tom\u00e0s Margalef (Eds.). Springer Berlin Heidelberg Berlin Heidelberg 109--116.","DOI":"10.1007\/3-540-48158-3_14"},{"key":"e_1_3_2_1_49_1","volume-title":"Proceedings of the 22nd European MPI Users' Group Meeting. ACM, 4.","author":"Prabhu T.","unstructured":"T. Prabhu and W. Gropp . 2015. DAME: A runtime-compiled engine for derived datatypes . In Proceedings of the 22nd European MPI Users' Group Meeting. ACM, 4. T. Prabhu and W. Gropp. 2015. DAME: A runtime-compiled engine for derived datatypes. In Proceedings of the 22nd European MPI Users' Group Meeting. ACM, 4."},{"key":"e_1_3_2_1_50_1","volume-title":"International Workshop on Languages and Compilers for Parallel Computing. Springer, 307--321","author":"Schneider T.","unstructured":"T. Schneider , R. Gerstenberger , and T. Hoefler . 2013. Compiler optimizations for non-contiguous remote data movement . In International Workshop on Languages and Compilers for Parallel Computing. Springer, 307--321 . T. Schneider, R. Gerstenberger, and T. Hoefler. 2013. Compiler optimizations for non-contiguous remote data movement. In International Workshop on Languages and Compilers for Parallel Computing. Springer, 307--321."},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA.1992.753322"},{"key":"e_1_3_2_1_52_1","volume-title":"Proceedings of the 24th Symposium on High-Performance Parallel and Distributed Computing (HPDC'15)","author":"Besta M.","unstructured":"M. Besta and T. Hoefler . 2015. Accelerating Irregular Computations with Hardware Transactional Memory and Active Messages . In Proceedings of the 24th Symposium on High-Performance Parallel and Distributed Computing (HPDC'15) . ACM, 161--172. M. Besta and T. Hoefler. 2015. Accelerating Irregular Computations with Hardware Transactional Memory and Active Messages. In Proceedings of the 24th Symposium on High-Performance Parallel and Distributed Computing (HPDC'15). ACM, 161--172."},{"key":"e_1_3_2_1_53_1","volume-title":"Proceedings of the 29th International Conference on Supercomputing (ICS'15)","author":"Besta M.","unstructured":"M. Besta and T. Hoefler . 2015. Active Access: A Mechanism for High-Performance Distributed Data-Centric Computations . In Proceedings of the 29th International Conference on Supercomputing (ICS'15) . ACM, 155--164. M. Besta and T. Hoefler. 2015. Active Access: A Mechanism for High-Performance Distributed Data-Centric Computations. In Proceedings of the 29th International Conference on Supercomputing (ICS'15). ACM, 155--164."}],"event":{"name":"SC '19: The International Conference for High Performance Computing, Networking, Storage, and Analysis","location":"Denver Colorado","acronym":"SC '19","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing","IEEE CS"]},"container-title":["Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3295500.3356189","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3295500.3356189","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T01:02:13Z","timestamp":1750208533000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3295500.3356189"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,11,17]]},"references-count":53,"alternative-id":["10.1145\/3295500.3356189","10.1145\/3295500"],"URL":"https:\/\/doi.org\/10.1145\/3295500.3356189","relation":{},"subject":[],"published":{"date-parts":[[2019,11,17]]},"assertion":[{"value":"2019-11-17","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}