{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T19:22:59Z","timestamp":1784661779102,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":58,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,6,21]]},"DOI":"10.1145\/3695053.3731026","type":"proceedings-article","created":{"date-parts":[[2025,6,20]],"date-time":"2025-06-20T16:43:11Z","timestamp":1750437791000},"page":"664-678","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["Dynamic Load Balancer in Intel Xeon Scalable Processor: Performance Analyses, Enhancements, and Guidelines"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-0619-8634","authenticated-orcid":false,"given":"Jiaqi","family":"Lou","sequence":"first","affiliation":[{"name":"University of Illinois Urbana-Champaign, Urbana, IL, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4224-2812","authenticated-orcid":false,"given":"Srikar","family":"Vanavasam","sequence":"additional","affiliation":[{"name":"University of Illinois Urbana-Champaign, Urbana, IL, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8389-2133","authenticated-orcid":false,"given":"Yifan","family":"Yuan","sequence":"additional","affiliation":[{"name":"Meta, Menlo Park, CA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-2937-5804","authenticated-orcid":false,"given":"Ren","family":"Wang","sequence":"additional","affiliation":[{"name":"Intel Labs, Portland, OR, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0442-5634","authenticated-orcid":false,"given":"Nam Sung","family":"Kim","sequence":"additional","affiliation":[{"name":"University of Illinois Urbana-Champaign, Urbana, IL, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,6,20]]},"reference":[{"key":"e_1_3_3_2_2_2","doi-asserted-by":"crossref","unstructured":"Yehuda Afek Anat Bremler-Barr Shir\u00a0Landau Feibish and Liron Schiff. 2018. Detecting heavy flows in the SDN match and action model. Computer Networks 136 (2018).","DOI":"10.1016\/j.comnet.2018.02.018"},{"key":"e_1_3_3_2_3_2","unstructured":"Jasmin Ajanovic Mahesh Wagh Prashant Sethi Debendra\u00a0Das Sharma David\u00a0J. Harriman Mark\u00a0B. Rosenbluth Ajay\u00a0V. Bhatt Peter Barry Scott\u00a0Dion Rodgers Anil Vasudevan Sridhar Muthrasanallur James Akiyama Robert\u00a0G. Blankenship Ohad Falik Avi Mendelson Ilan Pardo Eran Tamari Eliezer Weissmann and Doron Shamia. 2017. Atomic operations in PCI express. https:\/\/patents.google.com\/patent\/US9535838B2\/en?q=(Atomic+operations+in+PCI+express)&oq=+Atomic+operations+in+PCI+express+"},{"key":"e_1_3_3_2_4_2","volume-title":"20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23)","author":"Bai Wei","year":"2023","unstructured":"Wei Bai, Shanim\u00a0Sainul Abdeen, Ankit Agrawal, Krishan\u00a0Kumar Attre, Paramvir Bahl, Ameya Bhagat, Gowri Bhaskara, Tanya Brokhman, Lei Cao, Ahmad Cheema, et\u00a0al. 2023. Empowering azure storage with RDMA. In 20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23)."},{"key":"e_1_3_3_2_5_2","unstructured":"Bates Stephen and Duer Oren. 2018. Enabling the NVMe\u2122 CMB and PMR Ecosystem. https:\/\/nvmexpress.org\/wp-content\/uploads\/Session-2-Enabling-the-NVMe-CMB-and-PMR-Ecosystem-Eideticom-and-Mell....pdf."},{"key":"e_1_3_3_2_6_2","doi-asserted-by":"crossref","unstructured":"Theophilus Benson Ashok Anand Aditya Akella and Ming Zhang. 2010. Understanding data center traffic characteristics. ACM SIGCOMM Computer Communication Review 40 1 (2010).","DOI":"10.1145\/1672308.1672325"},{"key":"e_1_3_3_2_7_2","doi-asserted-by":"crossref","unstructured":"Haibo Chen Rong Chen Xingda Wei Jiaxin Shi Yanzhe Chen Zhaoguo Wang Binyu Zang and Haibing Guan. 2017. Fast in-memory transaction processing using RDMA and HTM. ACM Transactions on Computer Systems (TOCS) 35 1 (2017) 1\u201337.","DOI":"10.1145\/3092701"},{"key":"e_1_3_3_2_8_2","unstructured":"DPDK. Accessed in 2024. DPDK All Releases. https:\/\/fast.dpdk.org\/rel\/."},{"key":"e_1_3_3_2_9_2","unstructured":"DPDK. Accessed in 2024. Event Device Library. https:\/\/doc.dpdk.org\/guides-18.05\/prog_guide\/eventdev.html."},{"key":"e_1_3_3_2_10_2","unstructured":"DPDK. Accessed in 2024. Packet Distributor Library. https:\/\/doc.dpdk.org\/guides\/prog_guide\/packet_distrib_lib.html."},{"key":"e_1_3_3_2_11_2","unstructured":"Eddie Kohler. Accessed in 2024. Masstree. https:\/\/github.com\/kohler\/masstree-beta."},{"key":"e_1_3_3_2_12_2","doi-asserted-by":"publisher","DOI":"10.1145\/3503222.3507776"},{"key":"e_1_3_3_2_13_2","volume-title":"14th USENIX Symposium on Operating Systems Design and Implementation (OSDI 20)","author":"Fried Joshua","year":"2020","unstructured":"Joshua Fried, Zhenyuan Ruan, Amy Ousterhout, and Adam Belay. 2020. Caladan: Mitigating interference at microsecond timescales. In 14th USENIX Symposium on Operating Systems Design and Implementation (OSDI 20)."},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"publisher","DOI":"10.1145\/3579371.3589082"},{"key":"e_1_3_3_2_15_2","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC59245.2023.00025"},{"key":"e_1_3_3_2_16_2","doi-asserted-by":"crossref","unstructured":"Nicholas Hunt Tom Bergan Luis Ceze and Steven\u00a0D Gribble. 2013. DDOS: taming nondeterminism in distributed systems. ACM SIGPLAN Notices 48 4 (2013) 499\u2013508.","DOI":"10.1145\/2499368.2451170"},{"key":"e_1_3_3_2_17_2","volume-title":"15th USENIX Symposium on Operating Systems Design and Implementation (OSDI 21)","author":"Ibanez Stephen","year":"2021","unstructured":"Stephen Ibanez, Alex Mallery, Serhat Arslan, Theo Jepsen, Muhammad Shahbaz, Changhoon Kim, and Nick McKeown. 2021. The nanoPU: A nanosecond network stack for datacenters. In 15th USENIX Symposium on Operating Systems Design and Implementation (OSDI 21)."},{"key":"e_1_3_3_2_18_2","unstructured":"Intel Corporation. Accessed in 2024. 4th Gen Intel Xeon Processor Scalable Family sapphire rapids. https:\/\/www.intel.com\/content\/www\/us\/en\/developer\/articles\/technical\/fourth-generation-xeon-scalable-family-overview.html."},{"key":"e_1_3_3_2_19_2","unstructured":"Intel Corporation. Accessed in 2024. Intel Dynamic Load Balancer. https:\/\/www.intel.com\/content\/www\/us\/en\/download\/686372\/intel-dynamic-load-balancer.html."},{"key":"e_1_3_3_2_20_2","unstructured":"Intel Corporation. Accessed in 2024. intel\/pcm: Intel\u00ae Performance Counter Monitor (Intel\u00ae PCM). https:\/\/github.com\/intel\/pcm."},{"key":"e_1_3_3_2_21_2","unstructured":"Intel Corporation. Accessed in 2024. Intel\u00ae Ethernet 800 Series - Application Device Queues (ADQ) in a Kubernetes Environment. https:\/\/networkbuilders.intel.com\/docs\/networkbuilders\/intel-ethernet-800-series-application-device-queues-adq-in-a-kubernetes-environment-solution-brief-1664467340.pdf."},{"key":"e_1_3_3_2_22_2","unstructured":"Intel Corporation. Accessed in 2024. Introduction to Intel\u00ae Flow Ethernet Flow Director and Memcached Performance. https:\/\/www.intel.com\/content\/dam\/www\/public\/us\/en\/documents\/white-papers\/intel-ethernet-flow-director.pdf."},{"key":"e_1_3_3_2_23_2","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO61859.2024.00110"},{"key":"e_1_3_3_2_24_2","volume-title":"16th USENIX Symposium on Networked Systems Design and Implementation (NSDI 19)","author":"Kaffes Kostis","year":"2019","unstructured":"Kostis Kaffes, Timothy Chong, Jack\u00a0Tigar Humphries, Adam Belay, David Mazi\u00e8res, and Christos Kozyrakis. 2019. Shinjuku: Preemptive Scheduling for \u03bc second-scale Tail Latency. In 16th USENIX Symposium on Networked Systems Design and Implementation (NSDI 19)."},{"key":"e_1_3_3_2_25_2","first-page":"437","volume-title":"2016 USENIX annual technical conference (USENIX ATC 16)","author":"Kalia Anuj","year":"2016","unstructured":"Anuj Kalia, Michael Kaminsky, and David\u00a0G Andersen. 2016. Design guidelines for high performance RDMA systems. In 2016 USENIX annual technical conference (USENIX ATC 16). 437\u2013450."},{"key":"e_1_3_3_2_26_2","doi-asserted-by":"publisher","DOI":"10.1145\/2749469.2750392"},{"key":"e_1_3_3_2_27_2","doi-asserted-by":"publisher","DOI":"10.1145\/2391229.2391238"},{"key":"e_1_3_3_2_28_2","doi-asserted-by":"publisher","DOI":"10.1145\/3593856.3595890"},{"key":"e_1_3_3_2_29_2","volume-title":"20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23)","author":"Lin Jiaxin","year":"2023","unstructured":"Jiaxin Lin, Adney Cardoza, Tarannum Khan, Yeonju Ro, Brent\u00a0E Stephens, Hassan Wassel, and Aditya Akella. 2023. Ringleader: Efficiently Offloading Intra-Server Orchestration to NICs. In 20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23)."},{"key":"e_1_3_3_2_30_2","doi-asserted-by":"publisher","DOI":"10.1145\/2168836.2168855"},{"key":"e_1_3_3_2_31_2","volume-title":"19th USENIX Symposium on Networked Systems Design and Implementation (NSDI 22)","author":"McClure Sarah","year":"2022","unstructured":"Sarah McClure, Amy Ousterhout, Scott Shenker, and Sylvia Ratnasamy. 2022. Efficient scheduling policies for Microsecond-Scale tasks. In 19th USENIX Symposium on Networked Systems Design and Implementation (NSDI 22)."},{"key":"e_1_3_3_2_32_2","unstructured":"Mellanox. Accessed in 2024. RDMA\/core: Introduce peer memory interface. https:\/\/patchwork.ozlabs.org\/project\/ubuntu-kernel\/patch\/20210830225014.1613762-3-dann.frazier@canonical.com\/##2744167."},{"key":"e_1_3_3_2_33_2","unstructured":"Microsoft. Accessed in 2024. Introduction to Receive Side Scaling. https:\/\/learn.microsoft.com\/en-us\/windows-hardware\/drivers\/network\/introduction-to-receive-side-scaling."},{"key":"e_1_3_3_2_34_2","doi-asserted-by":"publisher","DOI":"10.5555\/AAI28667045"},{"key":"e_1_3_3_2_35_2","unstructured":"Network World. Accessed in 2024. Speed race: Just as 400Gb Ethernet gear rolls out an 800GbE spec is revealed. https:\/\/www.networkworld.com\/article\/3538789\/speed-race-just-as-400gb-ethernet-gear-starts-rolling-out-800gb-ethernet-gets-standardized.html."},{"key":"e_1_3_3_2_36_2","unstructured":"NVIDIA. Accessed in 2024. ConnectX NICs 10\/25\/40\/50\/100\/200 and 400G Ethernet Network Adapters. https:\/\/www.nvidia.com\/en-us\/networking\/ethernet-adapters\/."},{"key":"e_1_3_3_2_37_2","unstructured":"NVIDIA. Accessed in 2024. Developing a Linux Kernel Module using GPUDirect RDMA. https:\/\/docs.nvidia.com\/cuda\/gpudirect-rdma\/."},{"key":"e_1_3_3_2_38_2","unstructured":"NVIDIA. Accessed in 2024. DOCA Documentation v2.10.0. https:\/\/docs.nvidia.com\/doca\/sdk\/doca+dma\/index.html."},{"key":"e_1_3_3_2_39_2","unstructured":"NVIDIA. Accessed in 2024. mlxconfig \u2013 Changing Device Configuration Tool. https:\/\/docs.nvidia.com\/networking\/display\/mftv4240\/mlxconfig+%E2%80%93+changing+device+configuration+tool."},{"key":"e_1_3_3_2_40_2","unstructured":"NVIDIA. Accessed in 2024. NVIDIA BlueField Networking Platform. https:\/\/www.nvidia.com\/en-us\/networking\/products\/data-processing-unit\/."},{"key":"e_1_3_3_2_41_2","unstructured":"NVIDIA. Accessed in 2024. NVIDIA MLX5 Ethernet Driver. https:\/\/doc.dpdk.org\/guides\/nics\/mlx5.html."},{"key":"e_1_3_3_2_42_2","unstructured":"NVIDIA. Accessed in 2024. RDMA Aware Networks Programming User Manual: Transport Modes. https:\/\/docs.nvidia.com\/networking\/display\/rdmaawareprogrammingv17\/transport+modes."},{"key":"e_1_3_3_2_43_2","unstructured":"NVIDIA. Accessed in 2024. Understanding the iommu Linux grub File Configuration. https:\/\/enterprise-support.nvidia.com\/s\/article\/understanding-the-iommu-linux-grub-file-configuration."},{"key":"e_1_3_3_2_44_2","unstructured":"Peter Okech Nicholas\u00a0Mc Guire and William Okelo-Odongo. 2015. Inherent diversity in replicated architectures. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1510.02086 (2015)."},{"key":"e_1_3_3_2_45_2","doi-asserted-by":"crossref","unstructured":"Li Ou Xubin He and Jizhong Han. 2009. An efficient design for fast memory registration in RDMA. Journal of Network and Computer Applications 32 3 (2009).","DOI":"10.1016\/j.jnca.2008.07.008"},{"key":"e_1_3_3_2_46_2","volume-title":"16th USENIX Symposium on Networked Systems Design and Implementation (NSDI 19)","author":"Ousterhout Amy","year":"2019","unstructured":"Amy Ousterhout, Joshua Fried, Jonathan Behrens, Adam Belay, and Hari Balakrishnan. 2019. Shenango: Achieving high CPU efficiency for latency-sensitive datacenter workloads. In 16th USENIX Symposium on Networked Systems Design and Implementation (NSDI 19)."},{"key":"e_1_3_3_2_47_2","unstructured":"PCI-SIG. Accessed in 2024. PCI Express\u00ae Base Specification Revision 4.0. https:\/\/pcisig.com\/specifications."},{"key":"e_1_3_3_2_48_2","unstructured":"Perftest. Accessed in 2024. OFED perftest. https:\/\/github.com\/linux-rdma\/perftest."},{"key":"e_1_3_3_2_49_2","unstructured":"RDMA-Core. Accessed in 2024. RDMA Core Userspace Libraries and Daemons. https:\/\/github.com\/linux-rdma\/rdma-core."},{"key":"e_1_3_3_2_50_2","doi-asserted-by":"publisher","DOI":"10.1145\/3286062.3286081"},{"key":"e_1_3_3_2_51_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA56546.2023.10071135"},{"key":"e_1_3_3_2_52_2","doi-asserted-by":"publisher","DOI":"10.1145\/3373376.3378450"},{"key":"e_1_3_3_2_53_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613424.3614256"},{"key":"e_1_3_3_2_54_2","doi-asserted-by":"publisher","DOI":"10.1145\/3373376.3378528"},{"key":"e_1_3_3_2_55_2","doi-asserted-by":"publisher","DOI":"10.1145\/2967938.2967954"},{"key":"e_1_3_3_2_56_2","volume-title":"17th USENIX Symposium on Operating Systems Design and Implementation (OSDI 23)","author":"Wei Xingda","year":"2023","unstructured":"Xingda Wei, Rongxin Cheng, Yuhan Yang, Rong Chen, and Haibo Chen. 2023. Characterizing Off-path SmartNIC for Accelerating Distributed Systems. In 17th USENIX Symposium on Operating Systems Design and Implementation (OSDI 23)."},{"key":"e_1_3_3_2_57_2","unstructured":"Shinae Woo and KyoungSoo Park. 2012. Scalable TCP session monitoring with symmetric receive-side scaling. KAIST Daejeon Korea Tech. Rep 144 (2012)."},{"key":"e_1_3_3_2_58_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA56546.2023.10071127"},{"key":"e_1_3_3_2_59_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00066"}],"event":{"name":"ISCA '25: Proceedings of the 52nd Annual International Symposium on Computer Architecture","location":"Tokyo Japan","acronym":"SIGARCH '25","sponsor":["SIGARCH ACM Special Interest Group on Computer Architecture"]},"container-title":["Proceedings of the 52nd Annual International Symposium on Computer Architecture"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3695053.3731026","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,21]],"date-time":"2025-06-21T11:01:10Z","timestamp":1750503670000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3695053.3731026"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,20]]},"references-count":58,"alternative-id":["10.1145\/3695053.3731026","10.1145\/3695053"],"URL":"https:\/\/doi.org\/10.1145\/3695053.3731026","relation":{},"subject":[],"published":{"date-parts":[[2025,6,20]]},"assertion":[{"value":"2025-06-20","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}