{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,14]],"date-time":"2026-02-14T10:24:21Z","timestamp":1771064661327,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":33,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,2,19]],"date-time":"2025-02-19T00:00:00Z","timestamp":1739923200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,2,19]]},"DOI":"10.1145\/3712031.3712328","type":"proceedings-article","created":{"date-parts":[[2025,3,27]],"date-time":"2025-03-27T12:28:34Z","timestamp":1743078514000},"page":"162-172","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Lossy Compressed Collective Inter-FPGA Communications"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5790-6992","authenticated-orcid":false,"given":"Michihiro","family":"Koibuchi","sequence":"first","affiliation":[{"name":"National Institute of Informatics, Chiyoda-ku, Tokyo, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-8706-4167","authenticated-orcid":false,"given":"Yoshinobu","family":"Ishida","sequence":"additional","affiliation":[{"name":"NSW Inc., Shibuya-ku, Tokyo, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3313-4158","authenticated-orcid":false,"given":"Shoichi","family":"Hirasawa","sequence":"additional","affiliation":[{"name":"National Institute of Informatics, Chiyoda-ku, Tokyo, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6358-0570","authenticated-orcid":false,"given":"Yao","family":"Hu","sequence":"additional","affiliation":[{"name":"The University of Tokyo, Kashiwa, Ibaraki, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-6056-0738","authenticated-orcid":false,"given":"Takumi","family":"Honda","sequence":"additional","affiliation":[{"name":"Fujitsu, Kawasaki, Kanagawa, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-2204-2424","authenticated-orcid":false,"given":"Yusuke","family":"Nagasaka","sequence":"additional","affiliation":[{"name":"Fujitsu, Kawasaki, Kanagawa, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2103-881X","authenticated-orcid":false,"given":"Naoto","family":"Fukumoto","sequence":"additional","affiliation":[{"name":"Fujitsu, Kawasaki, Kanagawa, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,3,27]]},"reference":[{"key":"e_1_3_3_1_2_2","first-page":"1","volume-title":"High Performance Extreme Computing Conference (HPEC)","author":"Yang A. D. George and M. C. Herbordt and H. Lam and A. G. Lawande and J. Sheng and C.","year":"2016","unstructured":"A. D. George and M. C. Herbordt and H. Lam and A. G. Lawande and J. Sheng and C. Yang. 2016. Novo-G#: Large-scale reconfigurable computing with direct and programmable interconnects. In High Performance Extreme Computing Conference (HPEC). 1\u20137."},{"key":"e_1_3_3_1_3_2","unstructured":"AIO CORE. (accessed 7 Jul 2022). AIO CORE. http:\/\/www.aiocore.com\/."},{"key":"e_1_3_3_1_4_2","first-page":"328","volume-title":"IEEE International Symposium on Embedded Multicore\/Many-core Systems-on-Chip (MCSoC)","author":"Azegami K.","year":"2019","unstructured":"K. Azegami, K. Musha, K. Hironaka, A.\u00a0B. Ahmed, M. Koibuchi, Y. Hu, and H. Amano. 2019. A STDM (Static Time Division Multiplexing) Switch on a Multi-FPGA System. In IEEE International Symposium on Embedded Multicore\/Many-core Systems-on-Chip (MCSoC). 328\u2013333."},{"key":"e_1_3_3_1_5_2","doi-asserted-by":"crossref","unstructured":"F. Betzel K. Khatamifard H. Suresh D.\u00a0J. Lilja J. Sartori and U. Karpuzcu. 2018. Approximate Communication: Techniques for Reducing Communication Bottlenecks in Large-Scale Parallel Systems. ACM Comput. Surv. 51 1 Article 1 (jan 2018) 32\u00a0pages.","DOI":"10.1145\/3145812"},{"key":"e_1_3_3_1_6_2","first-page":"293","volume-title":"Data Compression Conference (DCC)","author":"Burtscher M.","year":"2007","unstructured":"M. Burtscher and P. Ratanaworabhan. 2007. High Throughput Compression of Double-Precision Floating-Point Data. In Data Compression Conference (DCC). 293\u2013302."},{"key":"e_1_3_3_1_7_2","unstructured":"Center for Computational Sciences University of Tsukuba. (accessed 7 Jul 2022). Overall specification of Cygnus. https:\/\/www.ccs.tsukuba.ac.jp\/wp-content\/uploads\/sites\/14\/2018\/12\/About-Cygnus.pdf."},{"key":"e_1_3_3_1_8_2","first-page":"113:1\u2013113:9","volume-title":"Design Automation Conference, DAC","author":"Chippa V.\u00a0K.","year":"2013","unstructured":"V.\u00a0K. Chippa, S.\u00a0T. Chakradhar, K. Roy, and A. Raghunathan. 2013. Analysis and characterization of inherent application resilience for approximate computing. In Design Automation Conference, DAC. 113:1\u2013113:9."},{"key":"e_1_3_3_1_9_2","unstructured":"C.\u00a0C. Cutler. 1952. Differential Quantization of Communication Signals U.S. patent 2605361."},{"key":"e_1_3_3_1_10_2","first-page":"730","volume-title":"IEEE International Parallel and Distributed Processing Symposium (IPDPS)","author":"Di S.","year":"2016","unstructured":"S. Di and F. Cappello. 2016. Fast Error-Bounded Lossy HPC Data Compression with SZ. In IEEE International Parallel and Distributed Processing Symposium (IPDPS). 730\u2013739."},{"key":"e_1_3_3_1_11_2","first-page":"47:1\u201347:11","volume-title":"Proceedings of the International Conference for High Performance Computing, Networking, Storage, and Analysis, SC","author":"Haidar A.","year":"2018","unstructured":"A. Haidar, S. Tomov, J.\u00a0J. Dongarra, and N.\u00a0J. Higham. 2018. Harnessing GPU tensor cores for fast FP16 arithmetic to speed up mixed-precision iterative refinement solvers. In Proceedings of the International Conference for High Performance Computing, Networking, Storage, and Analysis, SC. 47:1\u201347:11."},{"key":"e_1_3_3_1_12_2","unstructured":"HPL - A Portable Implementation of the High-Performance Linpack Benchmark for Distributed-Memory Computers. (accessed 7 Jan 2024). HPL. http:\/\/www.netlib.org\/benchmark\/hpl\/."},{"key":"e_1_3_3_1_13_2","first-page":"55","volume-title":"10th International Conference on Parallel, Distributed Computing Techn ologies and Applications (PDCTA)","author":"Hu Y.","year":"2021","unstructured":"Y. Hu and M. Koibuchi. 2021. The Case for Error-bounded Lossy Floating-Point Data Compression on Interconnection Networks. In 10th International Conference on Parallel, Distributed Computing Techn ologies and Applications (PDCTA). 55\u201376."},{"key":"e_1_3_3_1_14_2","unstructured":"Intel. (accessed 7 Jan 2024). Stratix 10 GX 10M FPGA. https:\/\/www.intel.com\/content\/www\/us\/en\/products\/sku\/210290\/intel-stratix-10-gx-10m-fpga\/specifications.html."},{"key":"e_1_3_3_1_15_2","doi-asserted-by":"crossref","unstructured":"J. Lant and J. Navaridas and M. Jujan and J. Goodacre. 2020. Toward FPGA-Based HPC: Advancing Interconnect Technologies. IEEE Micro 40 1 (2020) 25\u201334.","DOI":"10.1109\/MM.2019.2950655"},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"crossref","first-page":"120","DOI":"10.1109\/FPL.2012.6339275","volume-title":"International Conference on Field Programmable Logic and Applications (FPL)","author":"Kono Y.","year":"2012","unstructured":"Y. Kono, K. Sano, and S. Yamamoto. 2012. Scalability analysis of tightly-coupled FPGA-cluster for lattice Boltzmann computation. In International Conference on Field Programmable Logic and Applications (FPL). 120\u2013127."},{"key":"e_1_3_3_1_17_2","first-page":"57","volume-title":"The ACM\/SIGDA International Symposium on Field-Programmable Gate Arrays","author":"Langhammer M.","year":"2021","unstructured":"M. Langhammer, E. Nurvitadhi, B. Pasca, and S. Gribok. 2021. Stratix 10 NX Architecture and Applications. In The ACM\/SIGDA International Symposium on Field-Programmable Gate Arrays. 57\u201367."},{"key":"e_1_3_3_1_18_2","doi-asserted-by":"crossref","unstructured":"J. Lee C. Killian S. Le\u00a0Beux and D. Chillet. 2021. Distance-Aware Approximate Nanophotonic Interconnect. ACM Trans. Des. Autom. Electron. Syst. 27 2 Article 17 (nov 2021) 30\u00a0pages.","DOI":"10.1145\/3484309"},{"key":"e_1_3_3_1_19_2","first-page":"33:1\u201333:26","volume-title":"Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis, SC","author":"Liang X.","year":"2019","unstructured":"X. Liang, S. Di, S. Li, D. Tao, B. Nicolae, Z. Chen, and F. Cappello. 2019. Significantly improving lossy compression quality based on an optimized hybrid prediction model. In Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis, SC. 33:1\u201333:26."},{"key":"e_1_3_3_1_20_2","doi-asserted-by":"crossref","unstructured":"L. Liu J. Zhu Z. Li Y. Lu Y. Deng J. Han S. Yin and S. Wei. 2019. A Survey of Coarse-Grained Reconfigurable Architecture and Design: Taxonomy Challenges and Applications. ACM Comput. Surv. 52 6 (Oct 2019) 1\u201339.","DOI":"10.1145\/3357375"},{"key":"e_1_3_3_1_21_2","doi-asserted-by":"crossref","unstructured":"K. Mizutani H. Yamaguchi Y. Urino and M. Koibuchi. 2021. OPTWEB: A Lightweight Fully Connected Inter-FPGA Network for Efficient Collectives. IEEE Trans. Computers 70 6 (2021) 849\u2013862.","DOI":"10.1109\/TC.2021.3068715"},{"key":"e_1_3_3_1_22_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-44534-8_24"},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"crossref","unstructured":"M. Narasimha and A. Peterson. 1978. On the Computation of the Discrete Cosine Transform. IEEE Transactions on Communications 26 6 (June 1978) 934\u2013936.","DOI":"10.1109\/TCOM.1978.1094144"},{"key":"e_1_3_3_1_24_2","first-page":"56","volume-title":"Ninth International Symposium on Computing and Networking, CANDAR","author":"Niwa N.","year":"2021","unstructured":"N. Niwa, H. Amano, and M. Koibuchi. 2021. Low-Latency High-Bandwidth Interconnection Networks by Selective Packet Compression. In Ninth International Symposium on Computing and Networking, CANDAR. 56\u201364."},{"key":"e_1_3_3_1_25_2","doi-asserted-by":"crossref","unstructured":"P. Lindstrom and M. Isenburg. 2006. Fast and Efficient Compression of Floating-Point Data. IEEE Transactions on VLSI Systems 12 5 (Sep and Oct 2006) 1245\u20131250.","DOI":"10.1109\/TVCG.2006.143"},{"key":"e_1_3_3_1_26_2","doi-asserted-by":"crossref","unstructured":"A. Putnam A.\u00a0M. Caulfield E.\u00a0S. Chung D. Chiou K. Constantinides J. Demme H. Esmaeilzadeh J. Fowers G.\u00a0P. Gopal J. Gray M. Haselman S. Hauck S. Heil A. Hormati J.\u00a0Y. Kim S. Lanka J. Larus E. Peterson S. Pope A. Smith J. Thong P.\u00a0Y. Xiao and D. Burger. 2015. A Reconfigurable Fabric for Accelerating Large-Scale Datacenter Services. IEEE Micro 45 3 (2015) 10\u201322.","DOI":"10.1109\/MM.2015.42"},{"key":"e_1_3_3_1_27_2","first-page":"258","volume-title":"International Symposium on Field-Programmable Custom Computing Machines (FCCM)","author":"Herbordt Q. Xiong and R. Patel and C. Yang and T. Geng and A. Skjellum and M. C.","year":"2019","unstructured":"Q. Xiong and R. Patel and C. Yang and T. Geng and A. Skjellum and M. C. Herbordt. 2019. GhostSZ: A Transparent FPGA-Accelerated Lossy Compression Framework. In International Symposium on Field-Programmable Custom Computing Machines (FCCM). 258\u2013266."},{"key":"e_1_3_3_1_28_2","first-page":"914","volume-title":"IEEE International Parallel and Distributed Processing Symposium","author":"Sasaki N.","year":"2015","unstructured":"N. Sasaki, K. Sato, T. Endo, and S. Matsuoka. 2015. Exploration of Lossy Compression for Application-Level Checkpoint\/Restart. In IEEE International Parallel and Distributed Processing Symposium. 914\u2013922."},{"key":"e_1_3_3_1_29_2","first-page":"1","volume-title":"European Conference on Optical Communication (ECOC)","author":"Shimizu T.","year":"2021","unstructured":"T. Shimizu, S. Nakamura, H. Yamaguchi, K. Takemura, K. Mizutani, T. Usuki, and Y. Urino. 2021. Error-Free Operation for Fully Connected Wavelength-Routing Interconnect among 8 FPGAs with 2.8-Tbit\/s Total Bandwidth. In European Conference on Optical Communication (ECOC). 1\u20134."},{"key":"e_1_3_3_1_30_2","first-page":"1129","volume-title":"IEEE International Parallel and Distributed Processing Symposium (IPDPS)","author":"Tao D.","year":"2017","unstructured":"D. Tao, S. Di, Z. Chen, and F. Cappello. 2017. Significantly Improving Lossy Compression for Scientific Data Sets Based on Multidimensional Prediction and Error-Controlled Quantization. In IEEE International Parallel and Distributed Processing Symposium (IPDPS). 1129\u20131139."},{"key":"e_1_3_3_1_31_2","doi-asserted-by":"crossref","unstructured":"T. Ueno K. Sano and S. Yamamoto. 2017. Bandwidth Compression of Floating-Point Numerical Data Streams for FPGA-Based High-Performance Computing. ACM Transactions on Reconfigurable Technology and Systems 10 (05 2017) 1\u201322.","DOI":"10.1145\/3053688"},{"key":"e_1_3_3_1_32_2","doi-asserted-by":"crossref","unstructured":"Y.Hu. 2023. Exploring Approximate Communication Using Lossy Bitwise Compression on Interconnection Networks. IEEE Access 11 (2023) 59238\u201359249.","DOI":"10.1109\/ACCESS.2023.3281834"},{"key":"e_1_3_3_1_33_2","first-page":"683","volume-title":"USENIX Symposium on Networked Systems Design and Implementation, NSDI","author":"Yuan Y.","year":"2022","unstructured":"Y. Yuan, O. Alama, J. Fei, J. Nelson, D.\u00a0R.\u00a0K. Ports, A. Sapio, M. Canini, and N.\u00a0S. Kim. 2022. Unlocking the Power of Inline Floating-Point Operations on Programmable Switches. In USENIX Symposium on Networked Systems Design and Implementation, NSDI, Amar Phanishayee and Vyas Sekar (Eds.). USENIX Association, 683\u2013700."},{"key":"e_1_3_3_1_34_2","doi-asserted-by":"crossref","unstructured":"C. Yuechen and L. Ahmed. 2020. An Approximate Communication Framework for Network-on-Chips. IEEE Transactions on Parallel and Distributed Systems 31 6 (2020) 1434\u20131446.","DOI":"10.1109\/TPDS.2020.2968068"}],"event":{"name":"HPCASIA '25: Proceedings of the International Conference on High Performance Computing in Asia-Pacific Region","location":"Hsinchu Taiwan","acronym":"HPCASIA '25"},"container-title":["Proceedings of the International Conference on High Performance Computing in Asia-Pacific Region"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3712031.3712328","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3712031.3712328","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:18:10Z","timestamp":1750295890000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3712031.3712328"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,2,19]]},"references-count":33,"alternative-id":["10.1145\/3712031.3712328","10.1145\/3712031"],"URL":"https:\/\/doi.org\/10.1145\/3712031.3712328","relation":{},"subject":[],"published":{"date-parts":[[2025,2,19]]},"assertion":[{"value":"2025-03-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}