{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T15:44:40Z","timestamp":1782834280649,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":83,"publisher":"ACM","license":[{"start":{"date-parts":[[2021,11,13]],"date-time":"2021-11-13T00:00:00Z","timestamp":1636761600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100000780","name":"European Commission","doi-asserted-by":"publisher","award":["955776, 801039"],"award-info":[{"award-number":["955776, 801039"]}],"id":[{"id":"10.13039\/501100000780","id-type":"DOI","asserted-by":"publisher"}]},{"name":"ETH Zurich","award":["19-2 FEL-50"],"award-info":[{"award-number":["19-2 FEL-50"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2021,11,14]]},"DOI":"10.1145\/3458817.3476178","type":"proceedings-article","created":{"date-parts":[[2021,10,21]],"date-time":"2021-10-21T04:49:21Z","timestamp":1634791761000},"page":"1-16","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":44,"title":["Flare"],"prefix":"10.1145","author":[{"given":"Daniele","family":"De Sensi","sequence":"first","affiliation":[{"name":"ETH Zurich, Zurich, Switzerland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Salvatore","family":"Di Girolamo","sequence":"additional","affiliation":[{"name":"ETH Zurich, Zurich, Switzerland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Saleh","family":"Ashkboos","sequence":"additional","affiliation":[{"name":"ETH Zurich, Zurich, Switzerland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shigang","family":"Li","sequence":"additional","affiliation":[{"name":"ETH Zurich, Zurich, Switzerland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Torsten","family":"Hoefler","sequence":"additional","affiliation":[{"name":"ETH Zurich, Zurich, Switzerland"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2021,11,13]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"USA","author":"Forum Message Passing","year":"1994","unstructured":"Message Passing Forum . Mpi : A message-passing interface standard. Technical report , USA , 1994 . Message Passing Forum. Mpi: A message-passing interface standard. Technical report, USA, 1994."},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2018.00033"},{"key":"e_1_3_2_2_3_1","volume-title":"Hybrid-molecular-dynamics algorithms for the numerical simulation of quantum chromodynamics. Physical review D: Particles and fields, 35(8):2531--2542","author":"Gottlieb Steven","year":"1987","unstructured":"Steven Gottlieb , W. Liu , William D Toussaint , R. L. Renken , and R. L. Sugar . Hybrid-molecular-dynamics algorithms for the numerical simulation of quantum chromodynamics. Physical review D: Particles and fields, 35(8):2531--2542 , 1987 . Steven Gottlieb, W. Liu, William D Toussaint, R. L. Renken, and R. L. Sugar. Hybrid-molecular-dynamics algorithms for the numerical simulation of quantum chromodynamics. Physical review D: Particles and fields, 35(8):2531--2542, 1987."},{"key":"e_1_3_2_2_4_1","volume-title":"August","author":"Ben-Nun Tal","year":"2019","unstructured":"Tal Ben-Nun and Torsten Hoefler . Demystifying parallel and distributed deep learning: An in-depth concurrency analysis. ACM Comput. Surv., 52(4) , August 2019 . Tal Ben-Nun and Torsten Hoefler. Demystifying parallel and distributed deep learning: An in-depth concurrency analysis. ACM Comput. Surv., 52(4), August 2019."},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICPP.2014.36"},{"key":"e_1_3_2_2_6_1","volume-title":"Sparse allreduce: Efficient scalable communication for power-law data. CoRR, abs\/1312.3020","author":"Zhao Huasha","year":"2013","unstructured":"Huasha Zhao and John F. Canny . Sparse allreduce: Efficient scalable communication for power-law data. CoRR, abs\/1312.3020 , 2013 . Huasha Zhao and John F. Canny. Sparse allreduce: Efficient scalable communication for power-law data. CoRR, abs\/1312.3020, 2013."},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-03770-2_30"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jpdc.2008.09.002"},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"crossref","unstructured":"Richard L. Graham Devendar Bureddy Pak Lui Hal Rosenstock Gilad Shainer Gil Bloch Dror Goldenerg Mike Dubman Sasha Kotchubievsky Vladimir Koushnir Lion Levi Alex Margolin Tamir Ronen Alexander Shpiner Oded Wertheim and Eitan Zahavi. Scalable Hierarchical Aggregation Protocol (SHArP): A Hardware Architecture for Efficient Data Reduction. In Proceedings of COM-HPC 2016:  1st Workshop on Optimization of Communication in HPC Runtime Systems - Held in conjunction with SC 2016: The International Conference for High Performance Computing Networking Storage and Analysis pages 1--10. Institute of Electrical and Electronics Engineers Inc. jan 2017.  Richard L. Graham Devendar Bureddy Pak Lui Hal Rosenstock Gilad Shainer Gil Bloch Dror Goldenerg Mike Dubman Sasha Kotchubievsky Vladimir Koushnir Lion Levi Alex Margolin Tamir Ronen Alexander Shpiner Oded Wertheim and Eitan Zahavi. Scalable Hierarchical Aggregation Protocol (SHArP): A Hardware Architecture for Efficient Data Reduction. In Proceedings of COM-HPC 2016: 1st Workshop on Optimization of Communication in HPC Runtime Systems - Held in conjunction with SC 2016: The International Conference for High Performance Computing Networking Storage and Analysis pages 1--10. Institute of Electrical and Electronics Engineers Inc. jan 2017.","DOI":"10.1109\/COMHPC.2016.006"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA45697.2020.00085"},{"key":"e_1_3_2_2_11_1","volume-title":"18th USENIX Symposium on Networked Systems Design and Implementation (NSDI 21)","author":"Sapio Amedeo","year":"2021","unstructured":"Amedeo Sapio , Marco Canini , Chen-Yu Ho , Jacob Nelson , Panos Kalnis , Changhoon Kim , Arvind Krishnamurthy , Masoud Moshref , Dan R. K. Ports , and Peter Richt\u00e1rik . Scaling distributed machine learning with in-network aggregation . In 18th USENIX Symposium on Networked Systems Design and Implementation (NSDI 21) . USENIX Association , April 2021 . Amedeo Sapio, Marco Canini, Chen-Yu Ho, Jacob Nelson, Panos Kalnis, Changhoon Kim, Arvind Krishnamurthy, Masoud Moshref, Dan R. K. Ports, and Peter Richt\u00e1rik. Scaling distributed machine learning with in-network aggregation. In 18th USENIX Symposium on Networked Systems Design and Implementation (NSDI 21). USENIX Association, April 2021."},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/1816038.1816004"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA52012.2021.00079"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3126908.3126970"},{"key":"e_1_3_2_2_16_1","series-title":"Lecture Notes in Computer Science (including subseries Lecture Notes in Artificial Intelligence and Lecture Notes in Bioinformatics)","first-page":"41","volume-title":"Scalable Hierarchical Aggregation and Reduction Protocol (SHARP)TM Streaming-Aggregation Hardware Design and Evaluation","author":"Graham Richard L.","year":"2020","unstructured":"Richard L. Graham , Lion Levi , Devendar Burredy , Gil Bloch , Gilad Shainer , David Cho , George Elias , Daniel Klein , Joshua Ladd , Ophir Maor , Ami Marelli , Valentin Petrov , Evyatar Romlet , Yong Qin , and Ido Zemah . Scalable Hierarchical Aggregation and Reduction Protocol (SHARP)TM Streaming-Aggregation Hardware Design and Evaluation . In Lecture Notes in Computer Science (including subseries Lecture Notes in Artificial Intelligence and Lecture Notes in Bioinformatics) , volume 12151 LNCS, pages 41 -- 59 . Springer , jun 2020 . Richard L. Graham, Lion Levi, Devendar Burredy, Gil Bloch, Gilad Shainer, David Cho, George Elias, Daniel Klein, Joshua Ladd, Ophir Maor, Ami Marelli, Valentin Petrov, Evyatar Romlet, Yong Qin, and Ido Zemah. Scalable Hierarchical Aggregation and Reduction Protocol (SHARP)TM Streaming-Aggregation Hardware Design and Evaluation. In Lecture Notes in Computer Science (including subseries Lecture Notes in Artificial Intelligence and Lecture Notes in Bioinformatics), volume 12151 LNCS, pages 41--59. Springer, jun 2020."},{"key":"e_1_3_2_2_17_1","volume-title":"Cray xc series network","author":"Alverson Bob","year":"2012","unstructured":"Bob Alverson , Edwin Froese , Larry Kaplan , and Duncan Roweth . Cray xc series network . Cray Inc., White Paper WP-Aries 01-1112, 2012 . Bob Alverson, Edwin Froese, Larry Kaplan, and Duncan Roweth. Cray xc series network. Cray Inc., White Paper WP-Aries01-1112, 2012."},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CLUSTER.2018.00090"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/HOTI.2010.16"},{"key":"e_1_3_2_2_20_1","unstructured":"B. Arimilli Bernard C. Drerup Paul F. Lecocq and Hanhong Xue. Collective acceleration unit tree structure U.S. Patent US8756270B2 17\/06\/2014.  B. Arimilli Bernard C. Drerup Paul F. Lecocq and Hanhong Xue. Collective acceleration unit tree structure U.S. Patent US8756270B2 17\/06\/2014."},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2015.42"},{"key":"e_1_3_2_2_22_1","volume-title":"Conference on Machine Learning and Systems (MLSys)","author":"Gebara Nadeen","year":"2021","unstructured":"Nadeen Gebara , Paolo Costa , and Manya Ghobadi . Panama : In-network aggregation for shared machine learning clusters . In Conference on Machine Learning and Systems (MLSys) , April 2021 . Nadeen Gebara, Paolo Costa, and Manya Ghobadi. Panama: In-network aggregation for shared machine learning clusters. In Conference on Machine Learning and Systems (MLSys), April 2021."},{"key":"e_1_3_2_2_23_1","volume-title":"Netreduce: Rdma-compatible in-network reduction for distributed dnn training acceleration","author":"Liu Shuo","year":"2020","unstructured":"Shuo Liu , Qiaoling Wang , Junyi Zhang , Qinliang Lin , Yao Liu , Meng Xu , Ray C. C. Chueng , and Jianfei He . Netreduce: Rdma-compatible in-network reduction for distributed dnn training acceleration , 2020 . Shuo Liu, Qiaoling Wang, Junyi Zhang, Qinliang Lin, Yao Liu, Meng Xu, Ray C. C. Chueng, and Jianfei He. Netreduce: Rdma-compatible in-network reduction for distributed dnn training acceleration, 2020."},{"key":"e_1_3_2_2_24_1","volume-title":"18th USENIX Symposium on Networked Systems Design and Implementation (NSDI 21)","author":"Lao ChonLam","year":"2021","unstructured":"ChonLam Lao , Yanfang Le , Kshiteej Mahajan , Yixi Chen , Wenfei Wu , Aditya Akella , and Michael Swift . ATP : In-network aggregation for multi-tenant learning . In 18th USENIX Symposium on Networked Systems Design and Implementation (NSDI 21) . USENIX Association , April 2021 . ChonLam Lao, Yanfang Le, Kshiteej Mahajan, Yixi Chen, Wenfei Wu, Aditya Akella, and Michael Swift. ATP: In-network aggregation for multi-tenant learning. In 18th USENIX Symposium on Networked Systems Design and Implementation (NSDI 21). USENIX Association, April 2021."},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/3452296.3472904"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/2534169.2486011"},{"key":"e_1_3_2_2_27_1","volume-title":"Intel Tofino Series. https:\/\/www.intel.com\/content\/www\/us\/en\/products\/network-io\/programmable-ethernet-switch.html, mar","year":"2021","unstructured":"Intel. Intel Tofino Series. https:\/\/www.intel.com\/content\/www\/us\/en\/products\/network-io\/programmable-ethernet-switch.html, mar 2021 . Intel. Intel Tofino Series. https:\/\/www.intel.com\/content\/www\/us\/en\/products\/network-io\/programmable-ethernet-switch.html, mar 2021."},{"key":"e_1_3_2_2_28_1","volume-title":"https:\/\/people.ucsc.edu\/warner\/Bufs\/ethernet-switch-fm6000-sdn-paper.pdf, mar","author":"Intel Recep Ozdag","year":"2021","unstructured":"Recep Ozdag Intel . Intel (R) Ethernet Switch FM6000 Series - Software Defined Networking . https:\/\/people.ucsc.edu\/warner\/Bufs\/ethernet-switch-fm6000-sdn-paper.pdf, mar 2021 . Recep Ozdag Intel. Intel (R) Ethernet Switch FM6000 Series - Software Defined Networking. https:\/\/people.ucsc.edu\/warner\/Bufs\/ethernet-switch-fm6000-sdn-paper.pdf, mar 2021."},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPSR.2018.8850761"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/2656877.2656890"},{"key":"e_1_3_2_2_31_1","volume-title":"A survey on data plane programming with","author":"Hauser Frederik","year":"2021","unstructured":"Frederik Hauser , Marco H\u00e4berle , Daniel Merling , Steffen Lindner , Vladimir Gurevich , Florian Zeiger , Reinhard Frank , and Michael Menth . A survey on data plane programming with p4: Fundamentals, advances, and applied research, 2021 . Frederik Hauser, Marco H\u00e4berle, Daniel Merling, Steffen Lindner, Vladimir Gurevich, Florian Zeiger, Reinhard Frank, and Michael Menth. A survey on data plane programming with p4: Fundamentals, advances, and applied research, 2021."},{"key":"e_1_3_2_2_32_1","first-page":"4","volume-title":"An exhaustive survey on","author":"Kfoury Elie F.","year":"2021","unstructured":"Elie F. Kfoury , Jorge Crichigno , and Elias Bou-Harb . An exhaustive survey on p 4 programmable data plane switches: Taxonomy , applications, challenges, and future trends, 2021 . Elie F. Kfoury, Jorge Crichigno, and Elias Bou-Harb. An exhaustive survey on p4 programmable data plane switches: Taxonomy, applications, challenges, and future trends, 2021."},{"key":"e_1_3_2_2_33_1","first-page":"209","volume-title":"Proceedings of the Workshop on Hot Topics in Operating Systems, HotOS '19","author":"Dan R.","year":"2019","unstructured":"Dan R. K. Ports and Jacob Nelson. When should the network be the computer? In Proceedings of the Workshop on Hot Topics in Operating Systems, HotOS '19 , page 209 -- 215 , New York, NY, USA , 2019 . Association for Computing Machinery. Dan R. K. Ports and Jacob Nelson. When should the network be the computer? In Proceedings of the Workshop on Hot Topics in Operating Systems, HotOS '19, page 209--215, New York, NY, USA, 2019. Association for Computing Machinery."},{"key":"e_1_3_2_2_34_1","first-page":"67","volume-title":"14th USENIX Symposium on Networked Systems Design and Implementation (NSD1 17)","author":"Sharma Naveen Kr.","year":"2017","unstructured":"Naveen Kr. Sharma , Antoine Kaufmann , Thomas Anderson , Arvind Krishnamurthy , Jacob Nelson , and Simon Peter . Evaluating the power of flexible packet processing for network resource allocation . In 14th USENIX Symposium on Networked Systems Design and Implementation (NSD1 17) , pages 67 -- 82 , Boston, MA , March 2017 . USENIX Association. Naveen Kr. Sharma, Antoine Kaufmann, Thomas Anderson, Arvind Krishnamurthy, Jacob Nelson, and Simon Peter. Evaluating the power of flexible packet processing for network resource allocation. In 14th USENIX Symposium on Networked Systems Design and Implementation (NSD1 17), pages 67--82, Boston, MA, March 2017. USENIX Association."},{"key":"e_1_3_2_2_35_1","volume-title":"Sparsity in deep learning: Pruning and growth for efficient inference and training in neural networks","author":"Hoefler Torsten","year":"2021","unstructured":"Torsten Hoefler , Dan Alistarh , Tal Ben-Nun , Nikoli Dryden , and Alexandra Peste . Sparsity in deep learning: Pruning and growth for efficient inference and training in neural networks , 2021 . Torsten Hoefler, Dan Alistarh, Tal Ben-Nun, Nikoli Dryden, and Alexandra Peste. Sparsity in deep learning: Pruning and growth for efficient inference and training in neural networks, 2021."},{"key":"e_1_3_2_2_36_1","volume-title":"Deep compression: Compressing deep neural networks with pruning, trained quantization and huffman coding","author":"Han Song","year":"2016","unstructured":"Song Han , Huizi Mao , and William J. Dally . Deep compression: Compressing deep neural networks with pruning, trained quantization and huffman coding , 2016 . Song Han, Huizi Mao, and William J. Dally. Deep compression: Compressing deep neural networks with pruning, trained quantization and huffman coding, 2016."},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.5555\/3018874.3018875"},{"key":"e_1_3_2_2_38_1","volume-title":"Deep Learning and Unsupervised Feature Learning Workshop, NIPS 2011","author":"Vanhoucke Vincent","year":"2011","unstructured":"Vincent Vanhoucke , Andrew Senior , and Mark Z. Mao . Improving the speed of neural networks on cpus . In Deep Learning and Unsupervised Feature Learning Workshop, NIPS 2011 , 2011 . Vincent Vanhoucke, Andrew Senior, and Mark Z. Mao. Improving the speed of neural networks on cpus. In Deep Learning and Unsupervised Feature Learning Workshop, NIPS 2011, 2011."},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3295500.3356222"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"crossref","unstructured":"Christoph Sch\u00e4r Oliver Fuhrer Andrea Arteaga Nikolina Ban Christophe Charpilloz Salvatore Di Girolamo Laureline Hentgen Torsten Hoefler Xavier Lapillonne David Leutwyler Katherine Osterried Davide Panosetti Stefan R\u00fcdis\u00fchli Linda Schlemmer Thomas Schulthess Michael Sprenger Stefano Ubbiali and Heini Wernli. Kilometer-scale climate models: Prospects and challenges. Bulletin of the American Meteorological Society 100(12) Dec. 2019. Early Online Release.  Christoph Sch\u00e4r Oliver Fuhrer Andrea Arteaga Nikolina Ban Christophe Charpilloz Salvatore Di Girolamo Laureline Hentgen Torsten Hoefler Xavier Lapillonne David Leutwyler Katherine Osterried Davide Panosetti Stefan R\u00fcdis\u00fchli Linda Schlemmer Thomas Schulthess Michael Sprenger Stefano Ubbiali and Heini Wernli. Kilometer-scale climate models: Prospects and challenges. Bulletin of the American Meteorological Society 100(12) Dec. 2019. Early Online Release.","DOI":"10.1175\/BAMS-D-18-0167.1"},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.3402\/tellusa.v21i3.10086"},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.1038\/s43247-020-00085-4"},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.parco.2015.09.001"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCC.and.EUC.2013.65"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/TC.2014.2322593"},{"key":"e_1_3_2_2_46_1","first-page":"1","volume-title":"In CUG 2009 Proceedings","author":"Villa Oreste","year":"2009","unstructured":"Oreste Villa , Vidhya Gurumoorthi , and Sriram Krishnamoorthy . Effects of floating-point nonassociativity on numerical computations on massively multi-threaded systems . In In CUG 2009 Proceedings , pages 1 -- 11 , 2009 . Oreste Villa, Vidhya Gurumoorthi, and Sriram Krishnamoorthy. Effects of floating-point nonassociativity on numerical computations on massively multi-threaded systems. In In CUG 2009 Proceedings, pages 1--11, 2009."},{"key":"e_1_3_2_2_47_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.micpro.2019.102910"},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/ANCS.2013.6665172"},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/2602204.2602219"},{"key":"e_1_3_2_2_50_1","unstructured":"Internet Control Message Protocol. RFC 894 April 1984.  Internet Control Message Protocol. RFC 894 April 1984."},{"key":"e_1_3_2_2_51_1","doi-asserted-by":"publisher","DOI":"10.1145\/2934872.2934900"},{"key":"e_1_3_2_2_52_1","first-page":"531","volume-title":"16th USENIX Symposium on Networked Systems Design and Implementation (NSD1 19)","author":"Pontarelli Salvatore","year":"2019","unstructured":"Salvatore Pontarelli , Roberto Bifulco , Marco Bonola , Carmelo Cascone , Marco Spaziani , Valerio Bruschi , Davide Sanvito , Giuseppe Siracusano , Antonio Capone , Michio Honda , Felipe Huici , and Giuseppe Siracusano . Flowblaze : Stateful packet processing in hardware . In 16th USENIX Symposium on Networked Systems Design and Implementation (NSD1 19) , pages 531 -- 548 , Boston, MA , February 2019 . USENIX Association. Salvatore Pontarelli, Roberto Bifulco, Marco Bonola, Carmelo Cascone, Marco Spaziani, Valerio Bruschi, Davide Sanvito, Giuseppe Siracusano, Antonio Capone, Michio Honda, Felipe Huici, and Giuseppe Siracusano. Flowblaze: Stateful packet processing in hardware. In 16th USENIX Symposium on Networked Systems Design and Implementation (NSD1 19), pages 531--548, Boston, MA, February 2019. USENIX Association."},{"key":"e_1_3_2_2_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/HOTCHIPS.2015.7477325"},{"key":"e_1_3_2_2_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/TVLSI.2017.2654506"},{"key":"e_1_3_2_2_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/TVLSI.2020.3044752"},{"key":"e_1_3_2_2_56_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2020.2996660"},{"key":"e_1_3_2_2_57_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC41405.2020.00039"},{"key":"e_1_3_2_2_58_1","volume-title":"Tomahawk4 BCM56990 Series. https:\/\/www.broadcom.com\/products\/ethernet-connectivity\/switching\/strataxgs\/bcm56990-series, mar","year":"2019","unstructured":"Broadcom. Tomahawk4 BCM56990 Series. https:\/\/www.broadcom.com\/products\/ethernet-connectivity\/switching\/strataxgs\/bcm56990-series, mar 2019 . Broadcom. Tomahawk4 BCM56990 Series. https:\/\/www.broadcom.com\/products\/ethernet-connectivity\/switching\/strataxgs\/bcm56990-series, mar 2019."},{"key":"e_1_3_2_2_59_1","doi-asserted-by":"publisher","DOI":"10.1287\/opre.9.3.383"},{"key":"e_1_3_2_2_60_1","doi-asserted-by":"publisher","DOI":"10.1145\/3332466.3374528"},{"key":"e_1_3_2_2_61_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10766-008-0070-9"},{"key":"e_1_3_2_2_62_1","doi-asserted-by":"publisher","DOI":"10.1145\/3295500.3356196"},{"key":"e_1_3_2_2_63_1","doi-asserted-by":"publisher","DOI":"10.1109\/CLUSTER.2017.76"},{"key":"e_1_3_2_2_64_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2009.5161095"},{"key":"e_1_3_2_2_65_1","doi-asserted-by":"publisher","DOI":"10.1145\/3126908.3126926"},{"key":"e_1_3_2_2_66_1","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC.2005.1526010"},{"key":"e_1_3_2_2_67_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2010.12"},{"key":"e_1_3_2_2_68_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2011.84"},{"key":"e_1_3_2_2_69_1","doi-asserted-by":"publisher","DOI":"10.1145\/2600212.2600225"},{"key":"e_1_3_2_2_70_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2018.00029"},{"key":"e_1_3_2_2_71_1","doi-asserted-by":"publisher","DOI":"10.1142\/S0129626409000419"},{"key":"e_1_3_2_2_72_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2016.63"},{"key":"e_1_3_2_2_73_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2018.00120"},{"key":"e_1_3_2_2_74_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2018.00030"},{"key":"e_1_3_2_2_75_1","doi-asserted-by":"publisher","DOI":"10.1109\/DAC18072.2020.9218661"},{"key":"e_1_3_2_2_76_1","doi-asserted-by":"publisher","DOI":"10.1109\/PACT.2015.24"},{"key":"e_1_3_2_2_77_1","doi-asserted-by":"publisher","DOI":"10.1145\/2749246.2749263"},{"key":"e_1_3_2_2_78_1","volume-title":"Mellanox Quantum Switches. https:\/\/www.mellanox.com\/products\/infiniband-switches-ic\/quantum, mar","year":"2019","unstructured":"Mellanox. Mellanox Quantum Switches. https:\/\/www.mellanox.com\/products\/infiniband-switches-ic\/quantum, mar 2019 . Mellanox. Mellanox Quantum Switches. https:\/\/www.mellanox.com\/products\/infiniband-switches-ic\/quantum, mar 2019."},{"key":"e_1_3_2_2_79_1","doi-asserted-by":"publisher","DOI":"10.1145\/1964218.1964225"},{"key":"e_1_3_2_2_80_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_2_81_1","doi-asserted-by":"publisher","DOI":"10.5555\/4492.4495"},{"key":"e_1_3_2_2_82_1","volume-title":"Horovod: fast and easy distributed deep learning in tensorflow","author":"Sergeev Alexander","year":"2018","unstructured":"Alexander Sergeev and Mike Del Balso . Horovod: fast and easy distributed deep learning in tensorflow , 2018 . Alexander Sergeev and Mike Del Balso. Horovod: fast and easy distributed deep learning in tensorflow, 2018."},{"key":"e_1_3_2_2_83_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2018.00054"},{"key":"e_1_3_2_2_84_1","doi-asserted-by":"publisher","DOI":"10.23919\/DATE51398.2021.9474230"}],"event":{"name":"SC '21: The International Conference for High Performance Computing, Networking, Storage and Analysis","location":"St. Louis Missouri","acronym":"SC '21","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing","IEEE CS"]},"container-title":["Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3458817.3476178","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3458817.3476178","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T20:12:21Z","timestamp":1750191141000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3458817.3476178"}},"subtitle":["flexible in-network allreduce"],"short-title":[],"issued":{"date-parts":[[2021,11,13]]},"references-count":83,"alternative-id":["10.1145\/3458817.3476178","10.1145\/3458817"],"URL":"https:\/\/doi.org\/10.1145\/3458817.3476178","relation":{},"subject":[],"published":{"date-parts":[[2021,11,13]]},"assertion":[{"value":"2021-11-13","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}