{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,5,8]],"date-time":"2025-05-08T04:10:28Z","timestamp":1746677428992,"version":"3.40.5"},"reference-count":38,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2023,12,27]],"date-time":"2023-12-27T00:00:00Z","timestamp":1703635200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,12,27]],"date-time":"2023-12-27T00:00:00Z","timestamp":1703635200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Sci. China Inf. Sci."],"published-print":{"date-parts":[[2024,1]]},"DOI":"10.1007\/s11432-022-3727-6","type":"journal-article","created":{"date-parts":[[2024,1,5]],"date-time":"2024-01-05T05:01:48Z","timestamp":1704430908000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Building a domain-specific compiler for emerging processors with a reusable approach"],"prefix":"10.1007","volume":"67","author":[{"given":"Mingzhen","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yi","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bangduo","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hailong","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhongzhi","family":"Luan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Depei","family":"Qian","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,12,27]]},"reference":[{"key":"3727_CR1","doi-asserted-by":"publisher","first-page":"072001","DOI":"10.1007\/s11432-016-5588-7","volume":"59","author":"H H Fu","year":"2016","unstructured":"Fu H H, Liao J F, Yang J Z, et al. The Sunway TaihuLight supercomputer: system and applications. Sci China Inf Sci, 2016, 59: 072001","journal-title":"Sci China Inf Sci"},{"key":"3727_CR2","doi-asserted-by":"publisher","first-page":"708","DOI":"10.1109\/TPDS.2020.3030548","volume":"32","author":"M Z Li","year":"2021","unstructured":"Li M Z, Liu Y, Liu X Y, et al. The deep learning compiler: a comprehensive survey. IEEE Trans Parallel Distrib Syst, 2021, 32: 708\u2013727","journal-title":"IEEE Trans Parallel Distrib Syst"},{"key":"3727_CR3","unstructured":"Leary C, Wang T. XLA: TensorFlow, compiled. TensorFlow Dev Summit, 2017. https:\/\/developers.googleblog.com\/2017\/03\/xla-tensorflow-compiled.html"},{"key":"3727_CR4","unstructured":"Chen T Q, Moreau T, Jiang Z H, et al. TVM: an automated end-to-end optimizing compiler for deep learning. In: Proceedings of USENIX Symposium on Operating Systems Design and Implementation, Carlsbad, 2018. 578\u2013594"},{"key":"3727_CR5","doi-asserted-by":"crossref","unstructured":"Bondhugula U, Hartono A, Ramanujam J, et al. A practical automatic polyhedral parallelizer and locality optimizer. In: Proceedings of ACM SIGPLAN Conference on Programming Language Design and Implementation, New York, 2008. 101\u2013113","DOI":"10.1145\/1375581.1375595"},{"key":"3727_CR6","doi-asserted-by":"crossref","unstructured":"Gysi T, Osuna C, Fuhrer O, et al. STELLA: a domain-specific tool for structured grid methods in weather and climate models. In: Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis, New York, 2015. 1\u201312","DOI":"10.1145\/2807591.2807627"},{"key":"3727_CR7","doi-asserted-by":"crossref","unstructured":"Lattner C, Adve V. LLVM: a compilation framework for lifelong program analysis & transformation. In: Proceedings of International Symposium on Code Generation and Optimization, San Jose, 2004. 75\u201386","DOI":"10.1109\/CGO.2004.1281665"},{"key":"3727_CR8","doi-asserted-by":"crossref","unstructured":"Lattner C, Amini M, Bondhugula U, et al. MLIR: scaling compiler infrastructure for domain specific computation. In: Proceedings of International Symposium on Code Generation and Optimization, Seoul, 2021. 2\u201314","DOI":"10.1109\/CGO51591.2021.9370308"},{"key":"3727_CR9","unstructured":"Vasilache N, Zinenko O, Bik A J C. Composable and modular code generation in MLIR: a structured and retargetable approach to tensor compiler construction. 2022. ArXiv:2202.03293"},{"key":"3727_CR10","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3544559","volume":"19","author":"A J C Bik","year":"2022","unstructured":"Bik A J C, Koanantakool P, Shpeisman T, et al. Compiler support for sparse tensor computations in MLIR. ACM Trans Archit Code Optim, 2022, 19: 1\u201325","journal-title":"ACM Trans Archit Code Optim"},{"key":"3727_CR11","doi-asserted-by":"crossref","unstructured":"Tian R Q, Guo L Z, Li, J J, et al. A high performance sparse tensor algebra compiler in MLIR. In: Proceedings of Workshop on the LLVM Compiler Infrastructure in HPC, St. Louis, 2021. 27\u201338","DOI":"10.1109\/LLVMHPC54804.2021.00009"},{"key":"3727_CR12","doi-asserted-by":"crossref","unstructured":"Jeong G, Kestor G, Chatarasi P, et al. Union: a unified HW-SW co-design ecosystem in MLIR for evaluating tensor operations on spatial accelerators. In: Proceedings of the 30th International Conference on Parallel Architectures and Compilation Techniques, Atlanta, 2021. 30\u201344","DOI":"10.1109\/PACT52795.2021.00010"},{"key":"3727_CR13","doi-asserted-by":"crossref","unstructured":"Li M Z, Liu Y, Hu Y M, et al. Automatic code generation and optimization of large-scale stencil computation on many-core processors. In: Proceedings of the International Conference on Parallel Processing, Lemont, 2021. 1\u201312","DOI":"10.1145\/3472456.3473517"},{"key":"3727_CR14","doi-asserted-by":"crossref","unstructured":"Yang C, Xue W, Fu H H, et al. 10M-core scalable fully-implicit solver for nonhydrostatic atmospheric dynamics. In: Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis, Salt Lake City, 2016. 57\u201368","DOI":"10.1109\/SC.2016.5"},{"key":"3727_CR15","doi-asserted-by":"crossref","unstructured":"Chen B W, Fu H H, Wei Y W, et al. Simulating the Wenchuan earthquake with accurate surface topography on Sunway TaihuLight. In: Proceedings of the International Conference for High Performance Computing, Networking, Storage, and Analysis, Dallas, 2018. 517\u2013528","DOI":"10.1109\/SC.2018.00043"},{"key":"3727_CR16","doi-asserted-by":"crossref","unstructured":"Duan X H, Gao P, Zhang T J, et al. Redesigning LAMMPS for peta-scale and hundred-billion-atom simulation on Sunway TaihuLight. In: Proceedings of the International Conference for High Performance Computing, Networking, Storage, and Analysis, Dallas, 2018. 148\u2013159","DOI":"10.1109\/SC.2018.00015"},{"key":"3727_CR17","doi-asserted-by":"crossref","unstructured":"Liu Y, Liu X, Li F, et al. Closing the \u201cQuantum Supremacy\u201d gap: achieving real-time simulation of a random quantum circuit using a new Sunway supercomputer. In: Proceedings of the International Conference for High Performance Computing, Networking, Storage, and Analysis, New York, 2021. 1\u201312","DOI":"10.1145\/3458817.3487399"},{"key":"3727_CR18","doi-asserted-by":"crossref","unstructured":"Liu C X, Xie B W, Liu X, et al. Towards efficient SpMV on Sunway Manycore architectures. In: Proceedings of the International Conference on Supercomputing, Beijing, 2018. 363\u2013373","DOI":"10.1145\/3205289.3205313"},{"key":"3727_CR19","doi-asserted-by":"crossref","unstructured":"Li M Z, Liu Y, Yang H L, et al. Multi-role SpTRSV on Sunway many-core architecture. In: Proceedings of International Conference on High Performance Computing and Communications, Exeter, 2018. 594\u2013601","DOI":"10.1109\/HPCC\/SmartCity\/DSS.2018.00109"},{"key":"3727_CR20","doi-asserted-by":"crossref","unstructured":"Wang X L, Liu W F, Xue W, et al. SwSpTRSV: a fast sparse triangular solve with sparse level tile layout on Sunway architectures. In: Proceedings of the ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, Vienna, 2018. 338\u2013353","DOI":"10.1145\/3178487.3178513"},{"key":"3727_CR21","doi-asserted-by":"publisher","first-page":"1636","DOI":"10.1109\/TPDS.2019.2953852","volume":"31","author":"M Z Li","year":"2020","unstructured":"Li M Z, Liu Y, Yang H L, et al. Accelerating sparse Cholesky factorization on Sunway Manycore architecture. IEEE Trans Parallel Distrib Syst, 2020, 31: 1636\u20131650","journal-title":"IEEE Trans Parallel Distrib Syst"},{"key":"3727_CR22","doi-asserted-by":"crossref","unstructured":"Fang J R, Fu H H, Zhao W L, et al. swDNN: a library for accelerating deep learning applications on Sunway TaihuLight. In: Proceedings of International Parallel and Distributed Processing Symposium, Orlando, 2017. 615\u2013624","DOI":"10.1109\/IPDPS.2017.20"},{"key":"3727_CR23","doi-asserted-by":"publisher","first-page":"4533","DOI":"10.1007\/s11227-020-03444-2","volume":"77","author":"Q C Han","year":"2021","unstructured":"Han Q C, Yang H L, Dun M, et al. Towards efficient tile low-rank GEMM computation on Sunway many-core processors. J Supercomput, 2021, 77: 4533\u20134564","journal-title":"J Supercomput"},{"key":"3727_CR24","doi-asserted-by":"publisher","first-page":"1020","DOI":"10.1109\/TETC.2018.2881265","volume":"9","author":"X G Zhong","year":"2018","unstructured":"Zhong X G, Li M Z, Yang H L, et al. swMR: a framework for accelerating MapReduce applications on Sunway Taihulight. IEEE Trans Emerg Top Comput, 2018, 9: 1020\u20131030","journal-title":"IEEE Trans Emerg Top Comput"},{"key":"3727_CR25","unstructured":"Zerrell T, Bruestle J. Stripe: tensor compilation via the nested polyhedral model. 2019. ArXiv:1903.06498"},{"key":"3727_CR26","unstructured":"Jin T, Bercea G T, Le T D. Compiling ONNX neural network models using MLIR. 2020. ArXiv:2008.08272"},{"key":"3727_CR27","unstructured":"Katel N, Khandelwal V, Bondhugula U. High performance GPU code generation for matrix-matrix multiplication using MLIR: some early results. 2021. ArXiv:2108.13191"},{"key":"3727_CR28","doi-asserted-by":"crossref","unstructured":"Komisarczyk K, Chelini L, Vadivel K, et al. PET-to-MLIR: a polyhedral front-end for MLIR. In: Proceedings of Euromicro Conference on Digital System Design, Kranj, 2020. 551\u2013556","DOI":"10.1109\/DSD51259.2020.00091"},{"key":"3727_CR29","unstructured":"Majumder K, Bondhugula U. HIR: an MLIR-based intermediate representation for hardware accelerator description. 2021. ArXiv:2103.00194"},{"key":"3727_CR30","unstructured":"Zhao R Z, Cheng J Y. Phism: polyhedral high-level synthesis in MLIR. 2021. ArXiv:2103.15103"},{"key":"3727_CR31","doi-asserted-by":"crossref","unstructured":"Yount C, Tobin J, Breuer A, et al. YASK-Yet another stencil kernel: a framework for HPC stencil code-generation and tuning. In: Proceedings of International Workshop on Domain-Specific Languages and High-Level Frameworks for High Performance Computing, Salt Lake City, 2016. 30\u201339","DOI":"10.1109\/WOLFHPC.2016.08"},{"key":"3727_CR32","doi-asserted-by":"crossref","unstructured":"Maruyama N, Nomura T, Sato K, et al. Physis: an implicitly parallel programming model for stencil computations on large-scale GPU-accelerated supercomputers. In: Proceedings of International Conference for High Performance Computing, Networking, Storage and Analysis, Seattle, 2011. 1\u201312","DOI":"10.1145\/2063384.2063398"},{"key":"3727_CR33","doi-asserted-by":"crossref","unstructured":"Ragan-Kelley J, Barnes C, Adams A, et al. Halide: a language and compiler for optimizing parallelism, locality, and recomputation in image processing pipelines. In: Proceedings of the ACM SIGPLAN Conference on Programming Language Design and Implementation, Seattle, 2013. 519\u2013530","DOI":"10.1145\/2499370.2462176"},{"key":"3727_CR34","doi-asserted-by":"crossref","unstructured":"Rawat P S, Vaidya M, Sukumaran-Rajam A, et al. On optimizing complex stencils on GPUs. In: Proceedings of International Parallel and Distributed Processing Symposium, Rio de Janeiro, 2019. 641\u2013652","DOI":"10.1109\/IPDPS.2019.00073"},{"key":"3727_CR35","doi-asserted-by":"crossref","unstructured":"Hagedorn B, Stoltzfus L, Steuwer M, et al. High performance stencil code generation with lift. In: Proceedings of the International Symposium on Code Generation and Optimization, Vienna, 2018. 100\u2013112","DOI":"10.1145\/3168824"},{"key":"3727_CR36","doi-asserted-by":"crossref","unstructured":"Ansel J, Kamil S, Veeramachaneni K, et al. OpenTuner: an extensible framework for program autotuning. In: Proceedings of the International Conference on Parallel Architectures and Compilation Techniques, Edmonton, 2014. 303\u2013316","DOI":"10.1145\/2628071.2628092"},{"key":"3727_CR37","doi-asserted-by":"crossref","unstructured":"Sun Q X, Liu Y, Yang H L, et al. csTuner: scalable auto-tuning framework for complex stencil computation on GPUs. In: Proceedings of International Conference on Cluster Computing, Portland, 2021. 192\u2013203","DOI":"10.1109\/Cluster48925.2021.00037"},{"key":"3727_CR38","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3469030","volume":"18","author":"T Gysi","year":"2021","unstructured":"Gysi T, M\u00fcller C, Zinenko O, et al. Domain-specific multi-level IR rewriting for GPU: the open earth compiler for GPU-accelerated climate simulation. ACM Trans Archit Code Optim, 2021, 18: 1\u201323","journal-title":"ACM Trans Archit Code Optim"}],"container-title":["Science China Information Sciences"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11432-022-3727-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11432-022-3727-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11432-022-3727-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,7]],"date-time":"2025-05-07T14:14:28Z","timestamp":1746627268000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11432-022-3727-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,12,27]]},"references-count":38,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2024,1]]}},"alternative-id":["3727"],"URL":"https:\/\/doi.org\/10.1007\/s11432-022-3727-6","relation":{},"ISSN":["1674-733X","1869-1919"],"issn-type":[{"type":"print","value":"1674-733X"},{"type":"electronic","value":"1869-1919"}],"subject":[],"published":{"date-parts":[[2023,12,27]]},"assertion":[{"value":"1 April 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 November 2022","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 February 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 December 2023","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"112101"}}