{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,5]],"date-time":"2026-08-05T18:06:30Z","timestamp":1785953190364,"version":"3.56.0"},"publisher-location":"New York, NY, USA","reference-count":86,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,4,26]],"date-time":"2026-04-26T00:00:00Z","timestamp":1777161600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"CSC \u2013 Centro Nazionale di Ricerca in High Performance Computing, Big Data and Quantum Computing","award":["--"],"award-info":[{"award-number":["--"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,4,27]]},"DOI":"10.1145\/3767295.3803620","type":"proceedings-article","created":{"date-parts":[[2026,4,24]],"date-time":"2026-04-24T20:20:04Z","timestamp":1777062004000},"page":"1829-1846","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["RoPeerTo: A Datacenter-Scale Architecture for Peer-To-Peer DMA between GPUs and FPGAs"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-8991-1443","authenticated-orcid":false,"given":"Marco","family":"Venere","sequence":"first","affiliation":[{"name":"Politecnico di Milano, Milan, Italy"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-3025-8620","authenticated-orcid":false,"given":"Giuseppe","family":"Sorrentino","sequence":"additional","affiliation":[{"name":"Politecnico di Milano, Milan, Italy"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0026-1281","authenticated-orcid":false,"given":"Benjamin","family":"Ramhorst","sequence":"additional","affiliation":[{"name":"ETH Zurich, Zurich, Switzerland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-9231-0089","authenticated-orcid":false,"given":"Maximilian Jakob","family":"Heer","sequence":"additional","affiliation":[{"name":"ETH Zurich, Zurich, Switzerland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3480-0570","authenticated-orcid":false,"given":"Lucian","family":"Petrica","sequence":"additional","affiliation":[{"name":"AMD Research and Advanced Development, Dublin, Ireland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-3984-3948","authenticated-orcid":false,"given":"Dario","family":"Korolija","sequence":"additional","affiliation":[{"name":"AMD Research and Advanced Development, Dublin, Ireland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9883-9693","authenticated-orcid":false,"given":"Marco D.","family":"Santambrogio","sequence":"additional","affiliation":[{"name":"Politecnico di Milano, Milan, Italy"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5834-0812","authenticated-orcid":false,"given":"Davide","family":"Conficconi","sequence":"additional","affiliation":[{"name":"Politecnico di Milano, Milan, Italy"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4396-6695","authenticated-orcid":false,"given":"Gustavo","family":"Alonso","sequence":"additional","affiliation":[{"name":"ETH Zurich, Zurich, Switzerland"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5644-2110","authenticated-orcid":false,"given":"Ken","family":"O'Brien","sequence":"additional","affiliation":[{"name":"AMD Research and Advanced Development, Dublin, Ireland"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,4,26]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2023.3323932"},{"key":"e_1_3_2_1_2_1","volume-title":"HIP documentation. https:\/\/rocm.docs.amd.com\/projects\/HIP\/en\/latest\/. Accessed","author":"AMD.","year":"2025","unstructured":"AMD. 2025. HIP documentation. https:\/\/rocm.docs.amd.com\/projects\/HIP\/en\/latest\/. Accessed 22 September 2025."},{"key":"e_1_3_2_1_3_1","volume-title":"https:\/\/www.amd.com\/en\/products\/software\/rocm.html. Accessed","author":"Cm Software AMD.","year":"2025","unstructured":"AMD. 2025. ROCm Software. https:\/\/www.amd.com\/en\/products\/software\/rocm.html. Accessed 22 September 2025."},{"key":"e_1_3_2_1_4_1","volume-title":"https:\/\/github.com\/ROCm\/rocm-systems. Accessed","author":"Runtime AMD.","year":"2025","unstructured":"AMD. 2025. ROCR-Runtime. https:\/\/github.com\/ROCm\/rocm-systems. Accessed 22 September 2025."},{"key":"e_1_3_2_1_5_1","volume-title":"Davide Rossetti, Francesco Simula, Laura Tosoratto, and Piero Vicini.","author":"Ammendola Roberto","year":"2012","unstructured":"Roberto Ammendola, Andrea Biagioni, Ottorino Frezza, F Lo Cicero, Alessandro Lonardo, Pier Stanislao Paolucci, Davide Rossetti, Francesco Simula, Laura Tosoratto, and Piero Vicini. 2012. APEnet+: a 3D Torus network optimized for GPU-based HPC Systems. In Journal of Physics: Conference Series, Vol. 396. IOP Publishing, 042059."},{"key":"e_1_3_2_1_6_1","unstructured":"Brian B Avants Nicholas J Tustison Gang Song Philip A Cook Arno Klein and James C Gee. 2011. Advanced Normalization Tools (ANTs). http:\/\/stnava.github.io\/ANTs\/"},{"key":"e_1_3_2_1_7_1","volume-title":"Empowering Azure Storage with RDMA. In 20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23)","author":"Bai Wei","year":"2023","unstructured":"Wei Bai, Shanim Sainul Abdeen, Ankit Agrawal, Krishan Kumar Attre, Paramvir Bahl, Ameya Bhagat, Gowri Bhaskara, Tanya Brokhman, Lei Cao, Ahmad Cheema, Rebecca Chow, Jeff Cohen, Mahmoud Elhaddad, Vivek Ette, Igal Figlin, Daniel Firestone, Mathew George, Ilya German, Lakhmeet Ghai, Eric Green, Albert Greenberg, Manish Gupta, Randy Haagens, Matthew Hendel, Ridwan Howlader, Neetha John, Julia Johnstone, Tom Jolly, Greg Kramer, David Kruse, Ankit Kumar, Erica Lan, Ivan Lee, Avi Levy, Marina Lipshteyn, Xin Liu, Chen Liu, Guohan Lu, Yuemin Lu, Xiakun Lu, Vadim Makhervaks, Ulad Malashanka, David A. Maltz, Ilias Marinos, Rohan Mehta, Sharda Murthi, Anup Namdhari, Aaron Ogus, Jitendra Padhye, Madhav Pandya, Douglas Phillips, Adrian Power, Suraj Puri, Shachar Raindel, Jordan Rhee, Anthony Russo, Maneesh Sah, Ali Sheriff, Chris Sparacino, Ashutosh Srivastava, Weixiang Sun, Nick Swanson, Fuhou Tian, Lukasz Tomczyk, Vamsi Vadlamuri, Alec Wolman, Ying Xie, Joyce Yom, Lihua Yuan, Yanzhao Zhang, and Brian Zill. 2023. Empowering Azure Storage with RDMA. In 20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23). USENIX Association, Boston, MA, 49\u201367. https:\/\/www.usenix.org\/conference\/nsdi23\/presentation\/bai"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00964"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/SURV.2012.090512.00043"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICPPW.2012.20"},{"key":"e_1_3_2_1_11_1","volume-title":"A survey of image registration techniques. ACM computing surveys (CSUR) 24, 4","author":"Brown Lisa Gottesfeld","year":"1992","unstructured":"Lisa Gottesfeld Brown. 1992. A survey of image registration techniques. ACM computing surveys (CSUR) 24, 4 (1992), 325\u2013376."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3543668"},{"key":"e_1_3_2_1_13_1","volume-title":"PCI express system architecture","author":"Budruk Ravi","unstructured":"Ravi Budruk, Don Anderson, and Tom Shanley. 2004. PCI express system architecture. Addison-Wesley Professional."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/PDP.2007.16"},{"key":"e_1_3_2_1_15_1","volume-title":"Real-Time Imaging VIII","author":"Castro-Pareja Carlos R","unstructured":"Carlos R Castro-Pareja, Jogikal M Jagadeesh, and Raj Shekhar. 2004. FPGA-based acceleration of mutual information calculation for real-time 3D image registration. In Real-Time Imaging VIII, Vol. 5297. SPIE, 212\u2013219."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.2352\/J.ImagingSci.Technol.2005.49.2.art00002"},{"key":"e_1_3_2_1_17_1","unstructured":"Chen Chen Xinkui Zhao Guanjie Cheng Yuesheng Xu Shuiguang Deng and Jianwei Yin. 2025. Next-Gen Computing Systems with Compute Express Link: a Comprehensive Survey. arXiv:2412.20249 [cs.DC] https:\/\/arxiv.org\/abs\/2412.20249"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2021.3124125"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10278-013-9622-7"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/TETC.2022.3157948"},{"key":"e_1_3_2_1_21_1","volume-title":"GPUDirect RDMA and GPUDirect Storage. https:\/\/docs.nvidia.com\/datacenter\/cloud-native\/gpu-operator\/. Accessed","author":"NVIDIA Corporation","year":"2025","unstructured":"NVIDIA Corporation. 2024. GPUDirect RDMA and GPUDirect Storage. https:\/\/docs.nvidia.com\/datacenter\/cloud-native\/gpu-operator\/. Accessed 21 July 2025."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/FCCM.2008.50"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2022.3218898"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3669900"},{"key":"e_1_3_2_1_25_1","unstructured":"ROCm documentation. 2025. ROCm Bandwidth Test documentation. https:\/\/rocm.docs.amd.com\/projects\/rocm_bandwidth_test\/en\/latest\/index.html."},{"key":"e_1_3_2_1_26_1","volume-title":"Implementing Video Monitoring Capabilities by using hardware-based Encoders of the Raspberry Pi Zero 2 W. arXiv preprint arXiv:2507.12487","author":"Ederer Thomas","year":"2025","unstructured":"Thomas Ederer and Igor Ivki\u0107. 2025. Implementing Video Monitoring Capabilities by using hardware-based Encoders of the Raspberry Pi Zero 2 W. arXiv preprint arXiv:2507.12487 (2025)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCD63220.2024.00041"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3651890.3672233"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2024.110424"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3577193.3593724"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/COMST.2018.2862404"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511811685"},{"key":"e_1_3_2_1_33_1","volume-title":"18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24)","author":"He Zhenhao","year":"2024","unstructured":"Zhenhao He, Dario Korolija, Yu Zhu, Benjamin Ramhorst, Tristan Laan, Lucian Petrica, Michaela Blott, and Gustavo Alonso. 2024. ACCL+: an FPGA-Based Collective Engine for Distributed Applications. In 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24). USENIX Association, Santa Clara, CA, 211\u2013231. https:\/\/www.usenix.org\/conference\/osdi24\/presentation\/he"},{"key":"e_1_3_2_1_34_1","unstructured":"Maximilian Jakob Heer Benjamin Ramhorst Yu Zhu Luhao Liu Zhiyi Hu Jonas Dann and Gustavo Alonso. 2025. RoCE BALBOA: Service-enhanced Data Center RDMA for SmartNICs. arXiv:2507.20412 [cs.AR] https:\/\/arxiv.org\/abs\/2507.20412"},{"key":"e_1_3_2_1_35_1","volume-title":"International conference on computational science. Springer, 447\u2013461","author":"Hennigh Oliver","year":"2021","unstructured":"Oliver Hennigh, Susheela Narasimhan, Mohammad Amin Nabian, Akshay Subramaniam, Kaustubh Tangsali, Zhiwei Fang, Max Rietmann, Wonmin Byeon, and Sanjay Choudhry. 2021. NVIDIA SimNet\u2122: An AI-accelerated multi-physics simulation framework. In International conference on computational science. Springer, 447\u2013461."},{"key":"e_1_3_2_1_36_1","volume-title":"Medical image registration. Physics in medicine & biology 46, 3","author":"Hill Derek LG","year":"2001","unstructured":"Derek LG Hill, Philipp G Batchelor, Mark Holden, and David J Hawkes. 2001. Medical image registration. Physics in medicine & biology 46, 3 (2001), R1."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1162\/imag_a_00197"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/3447548.3467139"},{"key":"e_1_3_2_1_39_1","volume-title":"Proceedings of the VLDB Endowment","volume":"18","author":"Jiang Wenqi","year":"2023","unstructured":"Wenqi Jiang, Marco Zeller, Roger Waleffe, Torsten Hoefler, and Gustavo Alonso. 2023. Chameleon: a Heterogeneous and Disaggregated Accelerator System for Retrieval-Augmented Language Models. In Proceedings of the VLDB Endowment, Vol. 18. arXiv:2310.09949 [cs.LG]"},{"key":"e_1_3_2_1_40_1","volume-title":"The ITK Software Guide Book 2: Design and Functionality-Volume 2. Kitware","author":"Johnson Hans J","unstructured":"Hans J Johnson, Matthew M McCormick, and Luis Ibanez. 2015. The ITK Software Guide Book 2: Design and Functionality-Volume 2. Kitware, Inc."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/CANDARW57323.2022.00061"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.2172\/1573446"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMI.2009.2035616"},{"key":"e_1_3_2_1_44_1","volume-title":"14th USENIX Symposium on Operating Systems Design and Implementation (OSDI). 991\u20131010","author":"Korolija Dario","year":"2020","unstructured":"Dario Korolija, Timothy Roscoe, and Gustavo Alonso. 2020. Do OS Abstractions Make Sense on FPGAs?. In 14th USENIX Symposium on Operating Systems Design and Implementation (OSDI). 991\u20131010."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2019.2928289"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/CCGrid.2015.105"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/3626202.3637564"},{"key":"e_1_3_2_1_48_1","volume-title":"Clang: a C language family frontend for LLVM. https:\/\/clang.llvm.org\/. Accessed","author":"LLVM.","year":"2025","unstructured":"LLVM. 2025. Clang: a C language family frontend for LLVM. https:\/\/clang.llvm.org\/. Accessed 22 September 2025."},{"key":"e_1_3_2_1_49_1","volume-title":"2nd USENIX Workshop on Hot Topics in Edge Computing (HotEdge 19)","author":"Lu Sidi","year":"2019","unstructured":"Sidi Lu, Yongtao Yao, and Weisong Shi. 2019. Collaborative learning on the edges: A case study on connected vehicles. In 2nd USENIX Workshop on Hot Topics in Edge Computing (HotEdge 19)."},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/LCA.2022.3189207"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","unstructured":"Gandla Maharnisha Gandla Kumar and R. Arunraj. 2018. Satellite Image Registration and Image Fusion by using Principle Component Analysis. International Journal of Engineering and Technology (UAE) 7 (04 2018) 106\u2013110. 10.14419\/ijet.v7i2.19.15063","DOI":"10.14419\/ijet.v7i2.19.15063"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-32384-3_16"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","unstructured":"Javier Moya Matthias Gabathuler Mario Ruiz and Gustavo Alonso. 2023. fpgasystems\/hacc: ETHZ-HACC. Zenodo. https:\/\/doi.org\/10.5281\/zenodo.8340448. 10.5281\/zenodo.8340448","DOI":"10.5281\/zenodo.8340448"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.7937\/k9\/tcia.2018.pat12tbs"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1145\/3230543.3230560"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1007\/b98874"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICPP.2013.17"},{"key":"e_1_3_2_1_58_1","volume-title":"An efficient method for finding the minimum of a function of several variables without calculating derivatives. The computer journal 7, 2","author":"Powell Michael JD","year":"1964","unstructured":"Michael JD Powell. 1964. An efficient method for finding the minimum of a function of several variables without calculating derivatives. The computer journal 7, 2 (1964), 155\u2013162."},{"key":"e_1_3_2_1_59_1","volume-title":"Numerical recipes: the art of scientific computing","author":"Press William H","unstructured":"William H Press, Saul A Teukolsky, William T Vetterling, and Brian P Flannery. 2007. Numerical recipes: the art of scientific computing. Cambridge university press."},{"key":"e_1_3_2_1_60_1","unstructured":"Android Open Source Project. [n. d.]. Implement DMABUF and GPU memory accounting in Android 12. https:\/\/source.android.com\/docs\/core\/graphics\/implement-dma-buf-gpu-mem."},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2022.3156756"},{"key":"e_1_3_2_1_62_1","volume-title":"Jonas Dann, Luhao Liu, and Gustavo Alonso.","author":"Ramhorst Benjamin","year":"2025","unstructured":"Benjamin Ramhorst, Dario Korolija, Maximilian Jakob Heer, Jonas Dann, Luhao Liu, and Gustavo Alonso. 2025. Coyote v2: Raising the Level of Abstraction for Data Center FPGAs. arXiv:2504.21538 [cs.AR] https:\/\/arxiv.org\/abs\/2504.21538"},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV45572.2020.9093363"},{"key":"e_1_3_2_1_64_1","volume-title":"International Conferences on Computer Graphics, Vision and Mathematics.","author":"Rosner Jakub","year":"2010","unstructured":"Jakub Rosner, Hannes Fassold, Peter Schallauer, and Werner Bailer. 2010. Fast GPU-based image warping and inpainting for frame interpolation. In International Conferences on Computer Graphics, Vision and Mathematics."},{"key":"e_1_3_2_1_65_1","volume-title":"Robot-Assisted Medical Imaging: A Review. Proc","author":"Salcudean Septimiu E","year":"2022","unstructured":"Septimiu E Salcudean, Hamid Moradi, David G Black, and Nassir Navab. 2022. Robot-Assisted Medical Imaging: A Review. Proc. IEEE (2022)."},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2022.3228561"},{"key":"e_1_3_2_1_67_1","first-page":"256","article-title":"Performance review of zero copy techniques","volume":"6","author":"Song Jia","year":"2012","unstructured":"Jia Song and Jim Alves-Foss. 2012. Performance review of zero copy techniques. International Journal of Computer Science and Security (IJCSS) 6, 4 (2012), 256.","journal-title":"International Journal of Computer Science and Security (IJCSS)"},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"publisher","DOI":"10.1109\/FCCM62733.2025.00040"},{"key":"e_1_3_2_1_69_1","doi-asserted-by":"publisher","DOI":"10.1145\/3607928"},{"key":"e_1_3_2_1_70_1","doi-asserted-by":"publisher","DOI":"10.1109\/BioCAS58349.2023.10388589"},{"key":"e_1_3_2_1_71_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2005.251"},{"key":"e_1_3_2_1_72_1","volume-title":"Buffer Sharing and Synchronization (dma-buf). https:\/\/docs.kernel.org\/driver-api\/dma-buf.html. Accessed","author":"Kernel Community The Linux","year":"2025","unstructured":"The Linux Kernel Community. 2025. Buffer Sharing and Synchronization (dma-buf). https:\/\/docs.kernel.org\/driver-api\/dma-buf.html. Accessed 21 July 2025."},{"key":"e_1_3_2_1_73_1","volume-title":"AMDGPU Driver with KFD used by the ROCm project. https:\/\/github.com\/ROCm\/amdgpu. Accessed","author":"Kernel Community The Linux","year":"2025","unstructured":"The Linux Kernel Community. 2025. ROCm\/amdgpu: AMDGPU Driver with KFD used by the ROCm project. https:\/\/github.com\/ROCm\/amdgpu. Accessed 21 July 2025."},{"key":"e_1_3_2_1_74_1","doi-asserted-by":"publisher","DOI":"10.1109\/ReConFig.2013.6732296"},{"key":"e_1_3_2_1_75_1","doi-asserted-by":"publisher","DOI":"10.1145\/3708495"},{"key":"e_1_3_2_1_76_1","doi-asserted-by":"publisher","DOI":"10.1023\/A:1007958904918"},{"key":"e_1_3_2_1_77_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA57654.2024.00083"},{"key":"e_1_3_2_1_78_1","unstructured":"Zeke Wang Jie Zhang Hongjing Huang Yingtao Li Xueying Zhu Mo Sun Zihan Yang De Ma Huajing Tang Gang Pan Fei Wu Bingsheng He and Gustavo Alonso. 2025. FpgaHub: FPGA-Centric Hyper-Heterogeneous Computing Platform for Big Data Analytics. arXiv preprint arXiv:2503.09318."},{"key":"e_1_3_2_1_79_1","volume-title":"2022 USENIX Annual Technical Conference (ATC). 227\u2013241","author":"Wang Zeke","year":"2022","unstructured":"Zeke Wang, Jie Zhang, Zihan Yang, De Ma, Bingsheng He, and Gustavo Alonso. 2022. FpgaNIC: An FPGA-Based Versatile 100Gb SmartNIC for GPUs. In 2022 USENIX Annual Technical Conference (ATC). 227\u2013241."},{"key":"e_1_3_2_1_80_1","unstructured":"W Hwu Wen-mei. 2015. Heterogeneous System Architecture: A new compute platform infrastructure. Morgan Kaufmann."},{"key":"e_1_3_2_1_81_1","volume-title":"Conspirator: SmartNIC-Aided Control Plane for Distributed ML Workloads. In 2024 USENIX Annual Technical Conference (ATC). 765\u2013780","author":"Xiao Yunming","year":"2024","unstructured":"Yunming Xiao, Diman Zad Tootaghaj, Aditya Dhakal, Lianjie Cao, Puneet Sharma, and Aleksandar Kuzmanovic. 2024. Conspirator: SmartNIC-Aided Control Plane for Distributed ML Workloads. In 2024 USENIX Annual Technical Conference (ATC). 765\u2013780."},{"key":"e_1_3_2_1_82_1","volume-title":"OpenFabrics Alliance Workshop. 1\u201315","author":"Xiong Jianxin","year":"2020","unstructured":"Jianxin Xiong. 2020. RDMA with GPU Memory via DMA-Buf. In OpenFabrics Alliance Workshop. 1\u201315."},{"key":"e_1_3_2_1_83_1","doi-asserted-by":"publisher","DOI":"10.1109\/COMST.2024.3450292"},{"key":"e_1_3_2_1_84_1","unstructured":"Yongbo Yu Fuxun Yu Zirui Xu Di Wang Minjia Zhang Ang Li Chenchen Liu Zhi Tian and Xiang Chen. 2025. FedMT: Multi-Task Federated Learning with Competitive GPU Resource Sharing. IEEE Transactions on Computer-Aided Design of Integrated Circuits and Systems (2025)."},{"key":"e_1_3_2_1_85_1","doi-asserted-by":"publisher","DOI":"10.1002\/acm2.70135"},{"key":"e_1_3_2_1_86_1","volume-title":"Image registration methods: a survey. Image and vision computing 21, 11","author":"Zitova Barbara","year":"2003","unstructured":"Barbara Zitova and Jan Flusser. 2003. Image registration methods: a survey. Image and vision computing 21, 11 (2003), 977\u20131000."}],"event":{"name":"EUROSYS '26: 21st European Conference on Computer Systems","location":"McEwan Hall\/The University of Edinburgh Edinburgh Scotland UK","acronym":"EUROSYS '26","sponsor":["SIGOPS ACM Special Interest Group on Operating Systems"]},"container-title":["Proceedings of the 21st European Conference on Computer Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3767295.3803620","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,5]],"date-time":"2026-06-05T12:08:36Z","timestamp":1780661316000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3767295.3803620"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,26]]},"references-count":86,"alternative-id":["10.1145\/3767295.3803620","10.1145\/3767295"],"URL":"https:\/\/doi.org\/10.1145\/3767295.3803620","relation":{},"subject":[],"published":{"date-parts":[[2026,4,26]]},"assertion":[{"value":"2026-04-26","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}