{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T08:00:45Z","timestamp":1776931245140,"version":"3.51.2"},"publisher-location":"New York, NY, USA","reference-count":74,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,10,17]],"date-time":"2025-10-17T00:00:00Z","timestamp":1760659200000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/100000001","name":"NSF (National Science Foundation)","doi-asserted-by":"publisher","award":["2144796"],"award-info":[{"award-number":["2144796"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,18]]},"DOI":"10.1145\/3725843.3756064","type":"proceedings-article","created":{"date-parts":[[2025,10,17]],"date-time":"2025-10-17T17:19:56Z","timestamp":1760721596000},"page":"1284-1299","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Elk: Exploring the Efficiency of Inter-core Connected AI Chips with Deep Learning Compiler Techniques"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-8171-4970","authenticated-orcid":false,"given":"Yiqi","family":"Liu","sequence":"first","affiliation":[{"name":"University of Illinois Urbana Champaign, Urbana, Illinois, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-0363-9486","authenticated-orcid":false,"given":"Yuqi","family":"Xue","sequence":"additional","affiliation":[{"name":"University of Illinois Urbana Champaign, Urbana, Illinois, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-3018-2273","authenticated-orcid":false,"given":"Noelle","family":"Crawford","sequence":"additional","affiliation":[{"name":"University of Illinois at Urbana-Champaign, Urbana, Illinois, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4495-1997","authenticated-orcid":false,"given":"Jilong","family":"Xue","sequence":"additional","affiliation":[{"name":"Microsoft Research, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1125-671X","authenticated-orcid":false,"given":"Jian","family":"Huang","sequence":"additional","affiliation":[{"name":"University of Illinois at Urbana-Champaign, Urbana, Illinois, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,17]]},"reference":[{"key":"e_1_3_3_2_2_2","unstructured":"2017. Open Neural Network Exchange format. https:\/\/onnx.ai\/."},{"key":"e_1_3_3_2_3_2","unstructured":"2022. NVIDIA Hopper Architecture In-Depth. https:\/\/developer.nvidia.com\/blog\/nvidia-hopper-architecture-in-depth\/."},{"key":"e_1_3_3_2_4_2","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPSW.2019.00103"},{"key":"e_1_3_3_2_5_2","volume-title":"Proceedings of Machine Learning and Systems (MLSys\u201924)","author":"Agrawal Amey","year":"2024","unstructured":"Amey Agrawal, Nitin Kedia, Jayashree Mohan, Ashish Panwar, Nipun Kwatra, Bhargav Gulavani, Ramachandran Ramjee, and Alexey Tumanov. 2024. Vidur: A Large-Scale Simulation Framework For LLM Inference. In Proceedings of Machine Learning and Systems (MLSys\u201924)."},{"key":"e_1_3_3_2_6_2","doi-asserted-by":"crossref","unstructured":"Joshua Ainslie James Lee-Thorp Michiel de Jong Yury Zemlyanskiy Federico Lebr\u00f3n and Sumit Sanghai. 2023. GQA: Training Generalized Multi-Query Transformer Models from Multi-Head Checkpoints. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2305.13245 (2023).","DOI":"10.18653\/v1\/2023.emnlp-main.298"},{"key":"e_1_3_3_2_7_2","doi-asserted-by":"publisher","DOI":"10.1145\/215399.215427"},{"key":"e_1_3_3_2_8_2","unstructured":"Mohamed Bahnas. 2024. Tenstorrent Overview: Products and Software. https:\/\/icl.utk.edu\/newsletter\/presentations\/2024\/mohamed-bahnas-2024-03-22.pdf."},{"key":"e_1_3_3_2_9_2","unstructured":"Jingwei Cai Xuan Wang Mingyu Gao Sen Peng Zijian Zhu Yuchen Wei Zuotong Wu and Kaisheng Ma. 2025. SoMa: Identifying Exploring and Understanding the DRAM Communication Scheduling Space for DNN Accelerators. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2501.12634 (2025)."},{"key":"e_1_3_3_2_10_2","doi-asserted-by":"publisher","DOI":"10.1145\/3579371.3589048"},{"key":"e_1_3_3_2_11_2","unstructured":"Marco Cerliani. 2022. Linear-Tree. https:\/\/github.com\/cerlymarco\/linear-tree."},{"key":"e_1_3_3_2_12_2","doi-asserted-by":"publisher","DOI":"10.1145\/3620666.3651379"},{"key":"e_1_3_3_2_13_2","volume-title":"13th USENIX Symposium on Operating Systems Design and Implementation (OSDI\u201918)","author":"Chen Tianqi","year":"2018","unstructured":"Tianqi Chen, Thierry Moreau, Ziheng Jiang, Lianmin Zheng, Eddie Yan, Haichen Shen, Meghan Cowan, Leyuan Wang, Yuwei Hu, Luis Ceze, Carlos Guestrin, and Arvind Krishnamurthy. 2018. TVM: An Automated End-to-End Optimizing Compiler for Deep Learning. In 13th USENIX Symposium on Operating Systems Design and Implementation (OSDI\u201918)."},{"key":"e_1_3_3_2_14_2","unstructured":"Peizhuang Cong Aomufei Yuan Shimao Chen Yuxuan Tian Bowen Ye and Tong Yang. 2024. Prediction Is All MoE Needs: Expert Load Distribution Goes from Fluctuating to Stabilizing. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2404.16914 (2024)."},{"key":"e_1_3_3_2_15_2","unstructured":"Tri Dao Daniel\u00a0Y. Fu Stefano Ermon Atri Rudra and Christopher R\u00e9. 2022. FlashAttention: Fast and Memory-Efficient Exact Attention with IO-Awareness. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2205.14135 (2022)."},{"key":"e_1_3_3_2_16_2","doi-asserted-by":"publisher","DOI":"10.1145\/3695053.3731002"},{"key":"e_1_3_3_2_17_2","doi-asserted-by":"publisher","DOI":"10.1145\/3575693.3576933"},{"key":"e_1_3_3_2_18_2","doi-asserted-by":"publisher","DOI":"10.1063\/1.4981999"},{"key":"e_1_3_3_2_19_2","unstructured":"Google. 2023. XLA. https:\/\/www.tensorflow.org\/xla."},{"key":"e_1_3_3_2_20_2","unstructured":"Graphcore. 2022. Tile Vertex ISA. https:\/\/docs.graphcore.ai\/projects\/isa\/en\/latest\/_static\/Tile-Vertex-ISA_1.2.3.pdf."},{"key":"e_1_3_3_2_21_2","unstructured":"Graphcore. 2024. Next Generation IPU Systems: IPU-M2000 + IPU-POD4. https:\/\/www.graphcore.ai\/products\/mk2\/ipu-m2000-ipu-pod4."},{"key":"e_1_3_3_2_22_2","unstructured":"Graphcore. 2024. PopLibs API reference. https:\/\/docs.graphcore.ai\/projects\/poplar-api\/en\/latest\/poplibs_api.html."},{"key":"e_1_3_3_2_23_2","unstructured":"Congjie He Yeqi Huang Pei Mu Ziming Miao Jilong Xue Lingxiao Ma Fan Yang and Luo Mai. 2025. WaferLLM: A Wafer-Scale LLM Inference System. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2502.04563 (2025)."},{"key":"e_1_3_3_2_24_2","unstructured":"Emmanuel Hebrard. 2012. Scheduling and SAT. Nantes (2012). https:\/\/homepages.laas.fr\/ehebrard\/papers\/prescpaior2012.pdf"},{"key":"e_1_3_3_2_25_2","doi-asserted-by":"crossref","unstructured":"Charles Hong Sahil Bhatia Altan Haan Shengjun\u00a0Kris Dong Dima Nikiforov Alvin Cheung and Yakun\u00a0Sophia Shao. 2024. LLM-Aided Compilation for Tensor Accelerators. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2408.03408 (2024).","DOI":"10.1109\/LAD62341.2024.10691748"},{"key":"e_1_3_3_2_26_2","doi-asserted-by":"publisher","DOI":"10.1145\/3620666.3651383"},{"key":"e_1_3_3_2_27_2","doi-asserted-by":"publisher","DOI":"10.1145\/3503222.3507778"},{"key":"e_1_3_3_2_28_2","unstructured":"Zhe Jia Blake Tillman Marco Maggioni and Daniele\u00a0Paolo Scarpazza. 2020. Dissecting the Graphcore IPU Architecture via Microbenchmarking. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1912.03413 (2020)."},{"key":"e_1_3_3_2_29_2","doi-asserted-by":"publisher","DOI":"10.1145\/3579371.3589350"},{"key":"e_1_3_3_2_30_2","volume-title":"2021 IEEE Hot Chips 33 Symposium (HCS\u201921)","author":"Knowles Simon","year":"2021","unstructured":"Simon Knowles. 2021. Graphcore Colossus Mk2 IPU. In 2021 IEEE Hot Chips 33 Symposium (HCS\u201921)."},{"key":"e_1_3_3_2_31_2","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613165"},{"key":"e_1_3_3_2_32_2","volume-title":"Advances in Neural Information Processing Systems (NeurIPS\u201922)","author":"Li Dacheng","year":"2022","unstructured":"Dacheng Li, Hongyi Wang, Eric Xing, and Hao Zhang. 2022. AMP: Automatically Finding Model Parallel Strategies with Heterogeneity Awareness. In Advances in Neural Information Processing Systems (NeurIPS\u201922)."},{"key":"e_1_3_3_2_33_2","unstructured":"Shang Li Zhiyuan Yang Dhiraj Reddy Ankur Srivastava and Bruce Jacob. 2020. DRAMsim3: A Cycle-Accurate Thermal-Capable DRAM Simulator. IEEE Computer Architecture Letters (2020)."},{"key":"e_1_3_3_2_34_2","doi-asserted-by":"publisher","DOI":"10.1109\/HCS52781.2021.9567153"},{"key":"e_1_3_3_2_35_2","doi-asserted-by":"publisher","DOI":"10.1145\/3694715.3695955"},{"key":"e_1_3_3_2_36_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2017.29"},{"key":"e_1_3_3_2_37_2","unstructured":"Xinhao Luo Zihan Liu Yangjie Zhou Shihan Fang Ziyu Huang Yu Feng Chen Zhang Shixuan Sun Zhenzhe Zheng Jingwen Leng and Minyi Guo. 2025. ClusterFusion: Expanding Operator Fusion Scope for LLM Inference via Cluster-Level Collective Primitive. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2508.18850 (2025)."},{"key":"e_1_3_3_2_38_2","unstructured":"Meta. 2024. Our next-generation Meta Training and Inference Accelerator. https:\/\/ai.meta.com\/blog\/next-generation-meta-training-inference-accelerator-AI-MTIA\/."},{"key":"e_1_3_3_2_39_2","doi-asserted-by":"crossref","unstructured":"Xupeng Miao Yujie Wang Youhe Jiang Chunan Shi Xiaonan Nie Hailin Zhang and Bin Cui. 2022. Galvatron: Efficient Transformer Training over Multiple GPUs Using Automatic Parallelism. Proceedings of the VLDB Endowment (2022).","DOI":"10.14778\/3570690.3570697"},{"key":"e_1_3_3_2_40_2","unstructured":"Micron. 2024. HBM3E. https:\/\/www.micron.com\/products\/memory\/hbm\/hbm3e."},{"key":"e_1_3_3_2_41_2","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476209"},{"key":"e_1_3_3_2_42_2","unstructured":"NVIDIA. 2024. Blackwell Architecture for Generative AI. https:\/\/www.nvidia.com\/en-us\/data-center\/technologies\/blackwell-architecture\/."},{"key":"e_1_3_3_2_43_2","unstructured":"NVIDIA H100 Tensor Core GPU. 2024. https:\/\/www.nvidia.com\/en-us\/data-center\/h100\/."},{"key":"e_1_3_3_2_44_2","unstructured":"Dylan Patel and Daniel Nishball. 2023. Groq Inference Tokenomics: Speed But At What Cost. https:\/\/www.semianalysis.com\/p\/groq-inference-tokenomics-speed-but."},{"key":"e_1_3_3_2_45_2","doi-asserted-by":"crossref","unstructured":"William Peebles and Saining Xie. 2023. Scalable Diffusion Models with Transformers. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2212.09748 (2023).","DOI":"10.1109\/ICCV51070.2023.00387"},{"key":"e_1_3_3_2_46_2","unstructured":"Reiner Pope Sholto Douglas Aakanksha Chowdhery Jacob Devlin James Bradbury Anselm Levskaya Jonathan Heek Kefan Xiao Shivani Agrawal and Jeff Dean. 2022. Efficiently Scaling Transformer Inference. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2211.05102 (2022)."},{"key":"e_1_3_3_2_47_2","doi-asserted-by":"crossref","unstructured":"Raghu Prabhakar Ram Sivaramakrishnan Darshan Gandhi Yun Du Mingran Wang Xiangyu Song Kejie Zhang Tianren Gao Angela Wang Karen Li Yongning Sheng Joshua Brot Denis Sokolov Apurv Vivek Calvin Leung Arjun Sabnis Jiayu Bai Tuowen Zhao Mark Gottscho David Jackson Mark Luttrell Manish\u00a0K. Shah Edison Chen Kaizhao Liang Swayambhoo Jain Urmish Thakker Dawei Huang Sumti Jairath Kevin\u00a0J. Brown and Kunle Olukotun. 2024. SambaNova SN40L: Scaling the AI Memory Wall with Dataflow and Composition of Experts. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2405.07518 (2024).","DOI":"10.1109\/MICRO61859.2024.00100"},{"key":"e_1_3_3_2_48_2","doi-asserted-by":"publisher","DOI":"10.1145\/3079856.3080256"},{"key":"e_1_3_3_2_49_2","unstructured":"PyTorch. 2024. Building Models with PyTorch. https:\/\/pytorch.org\/tutorials\/beginner\/introyt\/modelsyt_tutorial.html."},{"key":"e_1_3_3_2_50_2","doi-asserted-by":"publisher","DOI":"10.1145\/3470496.3527382"},{"key":"e_1_3_3_2_51_2","volume-title":"17th USENIX Symposium on Operating Systems Design and Implementation (OSDI\u201923)","author":"Shi Yining","year":"2023","unstructured":"Yining Shi, Zhi Yang, Jilong Xue, Lingxiao Ma, Yuqing Xia, Ziming Miao, Yuxiao Guo, Fan Yang, and Lidong Zhou. 2023. Welder: Scheduling Deep Learning Memory Access via Tile-graph. In 17th USENIX Symposium on Operating Systems Design and Implementation (OSDI\u201923)."},{"key":"e_1_3_3_2_52_2","unstructured":"Gemma Team Thomas Mesnard Cassidy Hardin Robert Dadashi Surya Bhupatiraju Shreya Pathak Laurent Sifre Morgane Rivi\u00e8re Mihir\u00a0Sanjay Kale Juliette Love Pouya Tafti L\u00e9onard Hussenot Pier\u00a0Giuseppe Sessa Aakanksha Chowdhery Adam Roberts Aditya Barua Alex Botev Alex Castro-Ros Ambrose Slone Am\u00e9lie H\u00e9liou Andrea Tacchetti Anna Bulanova Antonia Paterson Beth Tsai Bobak Shahriari Charline\u00a0Le Lan Christopher\u00a0A. Choquette-Choo Cl\u00e9ment Crepy Daniel Cer Daphne Ippolito David Reid Elena Buchatskaya Eric Ni Eric Noland Geng Yan George Tucker George-Christian Muraru Grigory Rozhdestvenskiy Henryk Michalewski Ian Tenney Ivan Grishchenko Jacob Austin James Keeling Jane Labanowski Jean-Baptiste Lespiau Jeff Stanway Jenny Brennan Jeremy Chen Johan Ferret Justin Chiu Justin Mao-Jones Katherine Lee Kathy Yu Katie Millican Lars\u00a0Lowe Sjoesund Lisa Lee Lucas Dixon Machel Reid Maciej Miku\u0142a Mateo Wirth Michael Sharman Nikolai Chinaev Nithum Thain Olivier Bachem Oscar Chang Oscar Wahltinez Paige Bailey Paul Michel Petko Yotov Rahma Chaabouni Ramona Comanescu Reena Jana Rohan Anil Ross McIlroy Ruibo Liu Ryan Mullins Samuel\u00a0L Smith Sebastian Borgeaud Sertan Girgin Sholto Douglas Shree Pandya Siamak Shakeri Soham De Ted Klimenko Tom Hennigan Vlad Feinberg Wojciech Stokowiec Yu hui Chen Zafarali Ahmed Zhitao Gong Tris Warkentin Ludovic Peran Minh Giang Cl\u00e9ment Farabet Oriol Vinyals Jeff Dean Koray Kavukcuoglu Demis Hassabis Zoubin Ghahramani Douglas Eck Joelle Barral Fernando Pereira Eli Collins Armand Joulin Noah Fiedel Evan Senter Alek Andreev and Kathleen Kenealy. 2024. Gemma: Open Models Based on Gemini Research and Technology. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2403.08295 (2024)."},{"key":"e_1_3_3_2_53_2","unstructured":"Tenstorrent. 2023. Meet Grayskull. https:\/\/tenstorrent.com\/grayskull\/."},{"key":"e_1_3_3_2_54_2","unstructured":"Hugo Touvron Louis Martin Kevin Stone Peter Albert Amjad Almahairi Yasmine Babaei Nikolay Bashlykov Soumya Batra Prajjwal Bhargava Shruti Bhosale Dan Bikel Lukas Blecher Cristian\u00a0Canton Ferrer Moya Chen Guillem Cucurull David Esiobu Jude Fernandes Jeremy Fu Wenyin Fu Brian Fuller Cynthia Gao Vedanuj Goswami Naman Goyal Anthony Hartshorn Saghar Hosseini Rui Hou Hakan Inan Marcin Kardas Viktor Kerkez Madian Khabsa Isabel Kloumann Artem Korenev Punit\u00a0Singh Koura Marie-Anne Lachaux Thibaut Lavril Jenya Lee Diana Liskovich Yinghai Lu Yuning Mao Xavier Martinet Todor Mihaylov Pushkar Mishra Igor Molybog Yixin Nie Andrew Poulton Jeremy Reizenstein Rashi Rungta Kalyan Saladi Alan Schelten Ruan Silva Eric\u00a0Michael Smith Ranjan Subramanian Xiaoqing\u00a0Ellen Tan Binh Tang Ross Taylor Adina Williams Jian\u00a0Xiang Kuan Puxin Xu Zheng Yan Iliyan Zarov Yuchen Zhang Angela Fan Melanie Kambadur Sharan Narang Aurelien Rodriguez Robert Stojnic Sergey Edunov and Thomas Scialom. 2023. Llama 2: Open Foundation and Fine-Tuned Chat Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2307.09288 (2023)."},{"key":"e_1_3_3_2_55_2","volume-title":"2024 USENIX Annual Technical Conference (USENIX ATC\u201924)","author":"Um Taegeon","year":"2024","unstructured":"Taegeon Um, Byungsoo Oh, Minyoung Kang, Woo-Yeon Lee, Goeun Kim, Dongseob Kim, Youngtaek Kim, Mohd Muzzammil, and Myeongjae Jeon. 2024. Metis: Fast Automatic Distributed Training on Heterogeneous GPUs. In 2024 USENIX Annual Technical Conference (USENIX ATC\u201924)."},{"key":"e_1_3_3_2_56_2","doi-asserted-by":"crossref","unstructured":"Mario Vanhoucke and Jos\u00e9 Coelho. 2016. An approach using SAT solvers for the RCPSP with logical constraints. European Journal of Operational Research (2016).","DOI":"10.1016\/j.ejor.2015.08.044"},{"key":"e_1_3_3_2_57_2","unstructured":"Nicolas Vasilache Oleksandr Zinenko Theodoros Theodoridis Priya Goyal Zachary DeVito William\u00a0S. Moses Sven Verdoolaege Andrew Adams and Albert Cohen. 2018. Tensor Comprehensions: Framework-Agnostic High-Performance Machine Learning Abstractions. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1802.04730 (2018)."},{"key":"e_1_3_3_2_58_2","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan\u00a0N. Gomez Lukasz Kaiser and Illia Polosukhin. 2023. Attention Is All You Need. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1706.03762 (2023)."},{"key":"e_1_3_3_2_59_2","volume-title":"18th USENIX Symposium on Operating Systems Design and Implementation (OSDI\u201924)","author":"Wang Lei","year":"2024","unstructured":"Lei Wang, Lingxiao Ma, Shijie Cao, Quanlu Zhang, Jilong Xue, Yining Shi, Ningxin Zheng, Ziming Miao, Fan Yang, Ting Cao, Yuqing Yang, and Mao Yang. 2024. Ladder: Enabling Efficient Low-Precision Deep Learning Computing through Hardware-aware Tensor Transformation. In 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI\u201924)."},{"key":"e_1_3_3_2_60_2","doi-asserted-by":"publisher","DOI":"10.1145\/3567955.3567959"},{"key":"e_1_3_3_2_61_2","doi-asserted-by":"crossref","unstructured":"Samuel Williams Andrew Waterman and David Patterson. 2009. Roofline: An Insightful Visual Performance Model for Multicore Architectures. Communications of the ACM (2009).","DOI":"10.2172\/1407078"},{"key":"e_1_3_3_2_62_2","unstructured":"Yuqi Xue and Jian Huang. 2025. ReGate: Enabling Power Gating in Neural Processing Units. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2508.02536 (2025)."},{"key":"e_1_3_3_2_63_2","doi-asserted-by":"publisher","DOI":"10.1145\/3593856.3595912"},{"key":"e_1_3_3_2_64_2","doi-asserted-by":"publisher","DOI":"10.1145\/3579371.3589059"},{"key":"e_1_3_3_2_65_2","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO61859.2024.00011"},{"key":"e_1_3_3_2_66_2","doi-asserted-by":"publisher","DOI":"10.1145\/3613424.3614309"},{"key":"e_1_3_3_2_67_2","unstructured":"Susan Zhang Stephen Roller Naman Goyal Mikel Artetxe Moya Chen Shuohui Chen Christopher Dewan Mona Diab Xian Li Xi\u00a0Victoria Lin Todor Mihaylov Myle Ott Sam Shleifer Kurt Shuster Daniel Simig Punit\u00a0Singh Koura Anjali Sridhar Tianlu Wang and Luke Zettlemoyer. 2022. OPT: Open Pre-trained Transformer Language Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2205.01068 (2022)."},{"key":"e_1_3_3_2_68_2","doi-asserted-by":"publisher","DOI":"10.1145\/3307650.3322249"},{"key":"e_1_3_3_2_69_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA52012.2021.00085"},{"key":"e_1_3_3_2_70_2","volume-title":"17th USENIX Symposium on Operating Systems Design and Implementation (OSDI\u201923)","author":"Zhao Jie","year":"2023","unstructured":"Jie Zhao, Siyuan Feng, Xiaoqiang Dan, Fei Liu, Chengke Wang, Sheng Yuan, Wenyuan Lv, and Qikai Xie. 2023. Effectively Scheduling Computational Graphs of Deep Neural Networks toward Their Domain-Specific Accelerators. In 17th USENIX Symposium on Operating Systems Design and Implementation (OSDI\u201923)."},{"key":"e_1_3_3_2_71_2","series-title":"(OSDI\u201920)","volume-title":"Proceedings of the 14th USENIX Conference on Operating Systems Design and Implementation","author":"Zheng Lianmin","year":"2020","unstructured":"Lianmin Zheng, Chengfan Jia, Minmin Sun, Zhao Wu, Cody\u00a0Hao Yu, Ameer Haj-Ali, Yida Wang, Jun Yang, Danyang Zhuo, Koushik Sen, Joseph\u00a0E. Gonzalez, and Ion Stoica. 2020. Ansor: generating high-performance tensor programs for deep learning. In Proceedings of the 14th USENIX Conference on Operating Systems Design and Implementation(OSDI\u201920)."},{"key":"e_1_3_3_2_72_2","volume-title":"16th USENIX Symposium on Operating Systems Design and Implementation (OSDI\u201922)","author":"Zheng Lianmin","year":"2022","unstructured":"Lianmin Zheng, Zhuohan Li, Hao Zhang, Yonghao Zhuang, Zhifeng Chen, Yanping Huang, Yida Wang, Yuanzhong Xu, Danyang Zhuo, Eric\u00a0P Xing, et\u00a0al. 2022. Alpa: Automating Inter-and Intra-Operator Parallelism for Distributed Deep Learning. In 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI\u201922)."},{"key":"e_1_3_3_2_73_2","doi-asserted-by":"publisher","DOI":"10.1145\/3470496.3527440"},{"key":"e_1_3_3_2_74_2","doi-asserted-by":"publisher","DOI":"10.1145\/3503222.3507723"},{"key":"e_1_3_3_2_75_2","volume-title":"16th USENIX Symposium on Operating Systems Design and Implementation (OSDI\u201922)","author":"Zhu Hongyu","year":"2022","unstructured":"Hongyu Zhu, Ruofan Wu, Yijia Diao, Shanbin Ke, Haoyu Li, Chen Zhang, Jilong Xue, Lingxiao Ma, Yuqing Xia, Wei Cui, Fan Yang, Mao Yang, Lidong Zhou, Asaf Cidon, and Gennady Pekhimenko. 2022. ROLLER: Fast and Efficient Tensor Compilation for Deep Learning. In 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI\u201922)."}],"event":{"name":"MICRO 2025: 58th IEEE\/ACM International Symposium on Microarchitecture","location":"Seoul Korea","acronym":"MICRO 2025","sponsor":["SIGMICRO ACM Special Interest Group on Microarchitectural Research and Processing"]},"container-title":["Proceedings of the 58th IEEE\/ACM International Symposium on Microarchitecture"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3725843.3756064","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3725843.3756064","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,26]],"date-time":"2026-01-26T21:44:16Z","timestamp":1769463856000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3725843.3756064"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,17]]},"references-count":74,"alternative-id":["10.1145\/3725843.3756064","10.1145\/3725843"],"URL":"https:\/\/doi.org\/10.1145\/3725843.3756064","relation":{},"subject":[],"published":{"date-parts":[[2025,10,17]]},"assertion":[{"value":"2025-10-17","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}