{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,29]],"date-time":"2026-01-29T03:33:08Z","timestamp":1769657588137,"version":"3.49.0"},"publisher-location":"New York, NY, USA","reference-count":39,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,1,28]]},"DOI":"10.1145\/3774934.3786444","type":"proceedings-article","created":{"date-parts":[[2026,1,28]],"date-time":"2026-01-28T15:25:57Z","timestamp":1769613957000},"page":"635-647","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["MetaAttention: A Unified and Performant Attention Framework across Hardware Backends"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-0420-0875","authenticated-orcid":false,"given":"Feiyang","family":"Chen","sequence":"first","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-9436-674X","authenticated-orcid":false,"given":"Yu","family":"Cheng","sequence":"additional","affiliation":[{"name":"Peking University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-2313-5348","authenticated-orcid":false,"given":"Lei","family":"Wang","sequence":"additional","affiliation":[{"name":"Peking University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-4413-5500","authenticated-orcid":false,"given":"Yuqing","family":"Xia","sequence":"additional","affiliation":[{"name":"Microsoft Research, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7466-2128","authenticated-orcid":false,"given":"Ziming","family":"Miao","sequence":"additional","affiliation":[{"name":"Microsoft Research, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-9524-5476","authenticated-orcid":false,"given":"Lingxiao","family":"Ma","sequence":"additional","affiliation":[{"name":"Microsoft Research, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0378-060X","authenticated-orcid":false,"given":"Fan","family":"Yang","sequence":"additional","affiliation":[{"name":"Microsoft Research, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4495-1997","authenticated-orcid":false,"given":"Jilong","family":"Xue","sequence":"additional","affiliation":[{"name":"Microsoft Research, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8219-4499","authenticated-orcid":false,"given":"Zhi","family":"Yang","sequence":"additional","affiliation":[{"name":"Peking University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-6455-3898","authenticated-orcid":false,"given":"Mao","family":"Yang","sequence":"additional","affiliation":[{"name":"Microsoft Research, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4983-6047","authenticated-orcid":false,"given":"Xingda","family":"Wei","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9720-0361","authenticated-orcid":false,"given":"Haibo","family":"Chen","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2026,1,28]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"[n. d.]. NVIDIA TensorRT. https:\/\/developer.nvidia.com\/tensorrt"},{"key":"e_1_3_2_1_2_1","unstructured":"[n. d.]. PyTorch. https:\/\/pytorch.org\/"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3620665.3640366"},{"key":"e_1_3_2_1_4_1","volume-title":"Longformer: The Long-Document Transformer. arxiv:2004.05150. arxiv:2004.05150","author":"Beltagy Iz","year":"2020","unstructured":"Iz Beltagy, Matthew E. Peters, and Arman Cohan. 2020. Longformer: The Long-Document Transformer. arxiv:2004.05150. arxiv:2004.05150"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","unstructured":"Feiyang Chen. 2025. Artifact for: MetaAttention: A Unified and Performant Attention Framework Across Hardware Backends. https:\/\/doi.org\/10.5281\/zenodo.17701680 10.5281\/zenodo.17701680","DOI":"10.5281\/zenodo.17701680"},{"key":"e_1_3_2_1_6_1","volume-title":"TVM: An Automated End-to-End Optimizing Compiler for Deep Learning. In 13th USENIX Symposium on Operating Systems Design and Implementation (OSDI 18)","author":"Chen Tianqi","year":"2018","unstructured":"Tianqi Chen, Thierry Moreau, Ziheng Jiang, Lianmin Zheng, Eddie Yan, Haichen Shen, Meghan Cowan, Leyuan Wang, Yuwei Hu, Luis Ceze, Carlos Guestrin, and Arvind Krishnamurthy. 2018. TVM: An Automated End-to-End Optimizing Compiler for Deep Learning. In 13th USENIX Symposium on Operating Systems Design and Implementation (OSDI 18). USENIX Association, Carlsbad, CA. 578\u2013594. isbn:978-1-931971-47-8 https:\/\/www.usenix.org\/conference\/osdi18\/presentation\/chen"},{"key":"e_1_3_2_1_7_1","volume-title":"CUTLASS: CUDA Templates for Linear Algebra Subroutines. https:\/\/github.com\/NVIDIA\/cutlass","author":"NVIDIA Corporation","year":"2024","unstructured":"NVIDIA Corporation. 2024. CUTLASS: CUDA Templates for Linear Algebra Subroutines. https:\/\/github.com\/NVIDIA\/cutlass"},{"key":"e_1_3_2_1_8_1","unstructured":"Tri Dao. 2023. Flashattention-2: Faster attention with better parallelism and work partitioning. arXiv preprint arXiv:2307.08691."},{"key":"e_1_3_2_1_9_1","first-page":"16344","article-title":"Flashattention: Fast and memory-efficient exact attention with io-awareness","volume":"35","author":"Dao Tri","year":"2022","unstructured":"Tri Dao, Dan Fu, Stefano Ermon, Atri Rudra, and Christopher R\u00e9. 2022. Flashattention: Fast and memory-efficient exact attention with io-awareness. Advances in Neural Information Processing Systems, 35 (2022), 16344\u201316359.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_10_1","unstructured":"Tri Dao and Albert Gu. 2024. Transformers are SSMs: Generalized models and efficient algorithms through structured state space duality. arXiv preprint arXiv:2405.21060."},{"key":"e_1_3_2_1_11_1","unstructured":"DeepSeek-AI Aixin Liu Bei Feng Bin Wang Bingxuan Wang Bo Liu Chenggang Zhao Chengqi Dengr Chong Ruan Damai Dai Daya Guo Dejian Yang Deli Chen Dongjie Ji Erhang Li Fangyun Lin Fuli Luo Guangbo Hao Guanting Chen Guowei Li et al. 2024. DeepSeek-V2: A Strong Economical and Efficient Mixture-of-Experts Language Model. arxiv:2405.04434. arxiv:2405.04434"},{"key":"e_1_3_2_1_12_1","volume-title":"Flex Attention: A Programming Model for Generating Optimized Attention Kernels. arxiv:2412.05496. arxiv:2412.05496","author":"Dong Juechu","year":"2024","unstructured":"Juechu Dong, Boyuan Feng, Driss Guessous, Yanbo Liang, and Horace He. 2024. Flex Attention: A Programming Model for Generating Optimized Attention Kernels. arxiv:2412.05496. arxiv:2412.05496"},{"key":"e_1_3_2_1_13_1","unstructured":"Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Amy Yang Angela Fan et al. 2024. The llama 3 herd of models. arXiv preprint arXiv:2407.21783."},{"key":"e_1_3_2_1_14_1","volume-title":"Ting Cao, Fan Yang, and Mao Yang.","author":"Gao Yizhao","year":"2024","unstructured":"Yizhao Gao, Zhichen Zeng, Dayou Du, Shijie Cao, Hayden Kwok-Hay So, Ting Cao, Fan Yang, and Mao Yang. 2024. SeerAttention: Learning Intrinsic Sparse Attention in Your LLMs. arxiv:2410.13276. arxiv:2410.13276"},{"key":"e_1_3_2_1_15_1","first-page":"680","article-title":"Alcop: Automatic load-compute pipelining in deep learning compiler for ai-gpus","volume":"5","author":"Huang Guyue","year":"2023","unstructured":"Guyue Huang, Yang Bai, Liu Liu, Yuke Wang, Bei Yu, Yufei Ding, and Yuan Xie. 2023. Alcop: Automatic load-compute pipelining in deep learning compiler for ai-gpus. Proceedings of Machine Learning and Systems, 5 (2023), 680\u2013694.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_16_1","unstructured":"Shengyu Liu Jiashi Li. 2025. FlashMLA: Efficient MLA decoding kernels. https:\/\/github.com\/deepseek-ai\/FlashMLA"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3650200.3656626"},{"key":"e_1_3_2_1_18_1","unstructured":"Maxim Milakov and Natalia Gimelshein. 2018. Online normalizer calculation for softmax. arxiv:1805.02867. arxiv:1805.02867"},{"key":"e_1_3_2_1_19_1","unstructured":"Hao Peng Nikolaos Pappas Dani Yogatama Roy Schwartz Noah A. Smith and Lingpeng Kong. 2021. Random Feature Attention. arxiv:2103.02143. arxiv:2103.02143"},{"key":"e_1_3_2_1_20_1","unstructured":"Jason Ramapuram Federico Danieli Eeshan Dhekane Floris Weers Dan Busbridge Pierre Ablin Tatiana Likhomanenko Jagrit Digani Zijin Gu Amitis Shidani and Russ Webb. 2024. Theory Analysis and Best Practices for Sigmoid Self-Attention. arxiv:2409.04431. arxiv:2409.04431"},{"key":"e_1_3_2_1_21_1","first-page":"68658","article-title":"Flashattention-3: Fast and accurate attention with asynchrony and low-precision","volume":"37","author":"Shah Jay","year":"2024","unstructured":"Jay Shah, Ganesh Bikshandi, Ying Zhang, Vijay Thakkar, Pradeep Ramani, and Tri Dao. 2024. Flashattention-3: Fast and accurate attention with asynchrony and low-precision. Advances in Neural Information Processing Systems, 37 (2024), 68658\u201368685.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_22_1","volume-title":"17th USENIX Symposium on Operating Systems Design and Implementation (OSDI 23)","author":"Shi Yining","year":"2023","unstructured":"Yining Shi, Zhi Yang, Jilong Xue, Lingxiao Ma, Yuqing Xia, Ziming Miao, Yuxiao Guo, Fan Yang, and Lidong Zhou. 2023. Welder: Scheduling deep learning memory access via tile-graph. In 17th USENIX Symposium on Operating Systems Design and Implementation (OSDI 23). 701\u2013718."},{"key":"e_1_3_2_1_23_1","unstructured":"Yutao Sun Li Dong Shaohan Huang Shuming Ma Yuqing Xia Jilong Xue Jianyong Wang and Furu Wei. 2023. Retentive network: A successor to transformer for large language models. arXiv preprint arXiv:2307.08621."},{"key":"e_1_3_2_1_24_1","first-page":"7339","article-title":"You only cache once: Decoder-decoder architectures for language models","volume":"37","author":"Sun Yutao","year":"2024","unstructured":"Yutao Sun, Li Dong, Yi Zhu, Shaohan Huang, Wenhui Wang, Shuming Ma, Quanlu Zhang, Jianyong Wang, and Furu Wei. 2024. You only cache once: Decoder-decoder architectures for language models. Advances in Neural Information Processing Systems, 37 (2024), 7339\u20137361.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/3315508.3329973"},{"key":"e_1_3_2_1_26_1","volume-title":"\u0141 ukasz Kaiser, and Illia Polosukhin","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141 ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems, 30 (2017)."},{"key":"e_1_3_2_1_27_1","volume-title":"TRL: Transformer Reinforcement Learning. https:\/\/github.com\/huggingface\/trl","author":"von Werra Leandro","year":"2020","unstructured":"Leandro von Werra, Younes Belkada, Lewis Tunstall, Edward Beeching, Tristan Thrush, Nathan Lambert, Shengyi Huang, Kashif Rasul, and Quentin Gallou\u00e9dec. 2020. TRL: Transformer Reinforcement Learning. https:\/\/github.com\/huggingface\/trl"},{"key":"e_1_3_2_1_28_1","unstructured":"Hongyu Wang Shuming Ma Li Dong Shaohan Huang Huaijie Wang Lingxiao Ma Fan Yang Ruiping Wang Yi Wu and Furu Wei. 2023. BitNet: Scaling 1-bit Transformers for Large Language Models. arxiv:2310.11453. arxiv:2310.11453"},{"key":"e_1_3_2_1_29_1","unstructured":"Lei Wang Yu Cheng Yining Shi Zhengju Tang Zhiwen Mo Wenhao Xie Lingxiao Ma Yuqing Xia Jilong Xue Fan Yang and Zhi Yang. 2025. TileLang: A Composable Tiled Programming Model for AI Systems. arxiv:2504.17577. arxiv:2504.17577"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-demos.6"},{"key":"e_1_3_2_1_31_1","unstructured":"Mitchell Wortsman Jaehoon Lee Justin Gilmer and Simon Kornblith. 2023. Replacing softmax with ReLU in Vision Transformers. arxiv:2309.08586. arxiv:2309.08586"},{"key":"e_1_3_2_1_32_1","unstructured":"Songlin Yang Bailin Wang Yikang Shen Rameswar Panda and Yoon Kim. 2024. Gated Linear Attention Transformers with Hardware-Efficient Training. arxiv:2312.06635. arxiv:2312.06635"},{"key":"e_1_3_2_1_33_1","volume-title":"FLA: A Triton-Based Library for Hardware-Efficient Implementations of Linear Attention Mechanism. https:\/\/github.com\/fla-org\/flash-linear-attention","author":"Yang Songlin","year":"2024","unstructured":"Songlin Yang and Yu Zhang. 2024. FLA: A Triton-Based Library for Hardware-Efficient Implementations of Linear Attention Mechanism. https:\/\/github.com\/fla-org\/flash-linear-attention"},{"key":"e_1_3_2_1_34_1","unstructured":"Tianzhu Ye Li Dong Yuqing Xia Yutao Sun Yi Zhu Gao Huang and Furu Wei. 2024. Differential Transformer. arxiv:2410.05258v2. arxiv:2410.05258v2"},{"key":"e_1_3_2_1_35_1","unstructured":"Zihao Ye Lequn Chen Ruihang Lai Wuwei Lin Yineng Zhang Stephanie Wang Tianqi Chen Baris Kasikci Vinod Grover Arvind Krishnamurthy and Luis Ceze. 2025. FlashInfer: Efficient and Customizable Attention Engine for LLM Inference Serving. arxiv:2501.01005v2. arxiv:2501.01005v2"},{"key":"e_1_3_2_1_36_1","volume-title":"Big Bird: Transformers for Longer Sequences. arxiv:2007.14062. arxiv:2007.14062","author":"Zaheer Manzil","year":"2021","unstructured":"Manzil Zaheer, Guru Guruganesh, Avinava Dubey, Joshua Ainslie, Chris Alberti, Santiago Ontanon, Philip Pham, Anirudh Ravula, Qifan Wang, Li Yang, and Amr Ahmed. 2021. Big Bird: Transformers for Longer Sequences. arxiv:2007.14062. arxiv:2007.14062"},{"key":"e_1_3_2_1_37_1","volume-title":"Ansor: Generating High-Performance Tensor Programs for Deep Learning. In 14th USENIX Symposium on Operating Systems Design and Implementation (OSDI 20)","author":"Zheng Lianmin","year":"2020","unstructured":"Lianmin Zheng, Chengfan Jia, Minmin Sun, Zhao Wu, Cody Hao Yu, Ameer Haj-Ali, Yida Wang, Jun Yang, Danyang Zhuo, Koushik Sen, Joseph E. Gonzalez, and Ion Stoica. 2020. Ansor: Generating High-Performance Tensor Programs for Deep Learning. In 14th USENIX Symposium on Operating Systems Design and Implementation (OSDI 20). USENIX Association, 863\u2013879. isbn:978-1-939133-19-9 https:\/\/www.usenix.org\/conference\/osdi20\/presentation\/zheng"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/3710848.3710864"},{"key":"e_1_3_2_1_39_1","volume-title":"Erland Hilman Fuadi, and Alham Fikri Aji","author":"Zuhri Zayd M. K.","year":"2025","unstructured":"Zayd M. K. Zuhri, Erland Hilman Fuadi, and Alham Fikri Aji. 2025. Softpick: No Attention Sink, No Massive Activations with Rectified Softmax. arxiv:2504.20966. arxiv:2504.20966"}],"event":{"name":"PPoPP '26: 31st ACM SIGPLAN Annual Symposium on Principles and Practice of Parallel Programming","location":"Sydney NSW Australia","acronym":"PPoPP '26","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing","SIGPLAN ACM Special Interest Group on Programming Languages"]},"container-title":["Proceedings of the 31st ACM SIGPLAN Annual Symposium on Principles and Practice of Parallel Programming"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3774934.3786444","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,28]],"date-time":"2026-01-28T15:27:15Z","timestamp":1769614035000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3774934.3786444"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,1,28]]},"references-count":39,"alternative-id":["10.1145\/3774934.3786444","10.1145\/3774934"],"URL":"https:\/\/doi.org\/10.1145\/3774934.3786444","relation":{},"subject":[],"published":{"date-parts":[[2026,1,28]]},"assertion":[{"value":"2026-01-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}