{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,31]],"date-time":"2025-10-31T07:59:57Z","timestamp":1761897597388,"version":"build-2065373602"},"publisher-location":"Cham","reference-count":27,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032024350","type":"print"},{"value":"9783032024367","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,11,1]],"date-time":"2025-11-01T00:00:00Z","timestamp":1761955200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,11,1]],"date-time":"2025-11-01T00:00:00Z","timestamp":1761955200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-02436-7_13","type":"book-chapter","created":{"date-parts":[[2025,10,31]],"date-time":"2025-10-31T06:49:55Z","timestamp":1761893395000},"page":"195-211","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["PEAK: Generating High-Performance Schedules in\u00a0MLIR"],"prefix":"10.1007","author":[{"given":"Amir Mohammad","family":"Tavakkoli","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sameeran","family":"Joshi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shreya","family":"Singh","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yufan","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"P.","family":"Sadayappan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mary","family":"Hall","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,11,1]]},"reference":[{"key":"13_CR1","unstructured":"Iree: Intermediate representation execution environment. https:\/\/openxla.github.io\/iree\/. Accessed 4 Sep 2023"},{"key":"13_CR2","unstructured":"Schedule primitives in tvm. https:\/\/tvm.apache.org\/docs\/how_to\/work_with_schedules\/schedule_primitives.html. Accessed 2 Sep 2023"},{"key":"13_CR3","unstructured":"Transform dialect in mlir. https:\/\/mlir.llvm.org\/docs\/Dialects\/Transform\/. Accessed 2 Sep 2023"},{"issue":"4","key":"13_CR4","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3306346.3322967","volume":"38","author":"A Adams","year":"2019","unstructured":"Adams, A., et al.: Learning to optimize halide with tree search and random programs. ACM Trans. Graph. (TOG) 38(4), 1\u201312 (2019)","journal-title":"ACM Trans. Graph. (TOG)"},{"key":"13_CR5","doi-asserted-by":"crossref","unstructured":"Anderson, L., Adams, A., Ma, K., Li, T.M., Jin, T., Ragan-Kelley, J.: Efficient automatic scheduling of imaging and vision pipelines for the gpu. Proc. ACM on Program. Lang. 5(OOPSLA), 1\u201328 (2021)","DOI":"10.1145\/3485486"},{"key":"13_CR6","doi-asserted-by":"crossref","unstructured":"Bansal, M., Hsu, O., Olukotun, K., Kjolstad, F.: Mosaic: an interoperable compiler for tensor algebra. Proc. ACM on Program. Lang. 7(PLDI), 394\u2013419 (2023)","DOI":"10.1145\/3591236"},{"key":"13_CR7","unstructured":"Chen, C., Chame, J., Hall, M.: Chill: A framework for composing high-level loop transformations. Tech. rep, Citeseer (2008)"},{"key":"13_CR8","unstructured":"Chen, T., et al.: Tvm: an automated end-to-end optimizing compiler for deep learning (2018)"},{"key":"13_CR9","unstructured":"Chen, T., et al.: Learning to optimize tensor programs. Adv. Neural Inf. Process. Syst. 31 (2018)"},{"key":"13_CR10","unstructured":"Chetlur, S., et al.: cuDNN: efficient primitives for deep learning. arXiv preprint arXiv:1410.0759 (2014)"},{"key":"13_CR11","doi-asserted-by":"crossref","unstructured":"Hagedorn, B., Elliott, A.S., Barthels, H., Bodik, R., Grover, V.: Fireiron: a data-movement-aware scheduling language for gpus. In: Proceedings of the ACM International Conference on Parallel Architectures and Compilation Techniques, pp. 71\u201382 (2020)","DOI":"10.1145\/3410463.3414632"},{"key":"13_CR12","unstructured":"Hagedorn, B., Lenfers, J., Koehler, T., Gorlatch, S., Steuwer, M.: A language for describing optimization strategies. arXiv preprint arXiv:2002.02268 (2020)"},{"key":"13_CR13","unstructured":"Haj-Ali, A., et al.: Protuner: tuning programs with monte carlo tree search. arXiv preprint arXiv:2005.13685 (2020)"},{"key":"13_CR14","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"13_CR15","doi-asserted-by":"crossref","unstructured":"Ikarashi, Y., Bernstein, G.L., Reinking, A., Genc, H., Ragan-Kelley, J.: Exocompilation for productive programming of hardware accelerators. In: Proceedings of the 43rd ACM SIGPLAN International Conference on Programming Language Design and Implementation, pp. 703\u2013718 (2022)","DOI":"10.1145\/3519939.3523446"},{"key":"13_CR16","unstructured":"Kerr, A., Merrill, D., Demouth, J., Tran, J.: Cutlass: fast linear algebra in cuda C++. https:\/\/devblogs.nvidia.com\/cutlass-linear-algebra-cuda"},{"key":"13_CR17","doi-asserted-by":"publisher","unstructured":"Lattner, C., et al.: Mlir: scaling compiler infrastructure for domain specific computation. In: 2021 IEEE\/ACM International Symposium on Code Generation and Optimization (CGO), pp. 2\u201314 (2021). https:\/\/doi.org\/10.1109\/CGO51591.2021.9370308","DOI":"10.1109\/CGO51591.2021.9370308"},{"key":"13_CR18","unstructured":"Nvida: Basic linear algebra on nvidia gpus. https:\/\/docs.nvidia.com\/cuda\/cublas\/index.html\/"},{"issue":"6","key":"13_CR19","doi-asserted-by":"publisher","first-page":"519","DOI":"10.1145\/2499370.2462176","volume":"48","author":"J Ragan-Kelley","year":"2013","unstructured":"Ragan-Kelley, J., Barnes, C., Adams, A., Paris, S., Durand, F., Amarasinghe, S.: Halide: a language and compiler for optimizing parallelism, locality, and recomputation in image processing pipelines. Acm Sigplan Notices 48(6), 519\u2013530 (2013)","journal-title":"Acm Sigplan Notices"},{"key":"13_CR20","doi-asserted-by":"crossref","unstructured":"Redmon, J., Divvala, S., Girshick, R., Farhadi, A.: You only look once: unified, real-time object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 779\u2013788 (2016)","DOI":"10.1109\/CVPR.2016.91"},{"key":"13_CR21","doi-asserted-by":"crossref","unstructured":"Tiwari, A., Chen, C., Chame, J., Hall, M., Hollingsworth, J.K.: A scalable auto-tuning framework for compiler optimization. In: 2009 IEEE International Symposium on Parallel & Distributed Processing, pp. 1\u201312. IEEE (2009)","DOI":"10.1109\/IPDPS.2009.5161054"},{"key":"13_CR22","unstructured":"Vasilache, N., et al.: Tensor comprehensions: framework-agnostic high-performance machine learning abstractions. arXiv preprint arXiv:1802.04730 (2018)"},{"key":"13_CR23","doi-asserted-by":"crossref","unstructured":"Wu, X., et al.: Ytopt: autotuning scientific applications for energy efficiency at large scales. arXiv preprint arXiv:2303.16245 (2023)","DOI":"10.1002\/cpe.8322"},{"key":"13_CR24","doi-asserted-by":"crossref","unstructured":"Xu, Y., Yuan, Q., Barton, E.C., Li, R., Sadayappan, P., Sukumaran-Rajam, A.: Effective performance modeling and domain-specific compiler optimization of cnns for gpus. In: Proceedings of the International Conference on Parallel Architectures and Compilation Techniques, pp. 252\u2013264 (2022)","DOI":"10.1145\/3559009.3569674"},{"key":"13_CR25","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Yang, M., Baghdadi, R., Kamil, S., Shun, J., Amarasinghe, S.: Graphit: a high-performance dsl for graph analytics. arXiv preprint arXiv:1805.00923 (2018)","DOI":"10.1145\/3276491"},{"key":"13_CR26","unstructured":"Zheng, L., et\u00a0al.: Ansor: generating $$\\{$$High-Performance$$\\}$$ tensor programs for deep learning. In: 14th USENIX Symposium on Operating Systems Design and Implementation (OSDI 20), pp. 863\u2013879 (2020)"},{"key":"13_CR27","doi-asserted-by":"crossref","unstructured":"Zheng, S., Liang, Y., Wang, S., Chen, R., Sheng, K.: Flextensor: an automatic schedule exploration and optimization framework for tensor computation on heterogeneous system. In: Proceedings of the Twenty-Fifth International Conference on Architectural Support for Programming Languages and Operating Systems, pp. 859\u2013873 (2020)","DOI":"10.1145\/3373376.3378508"}],"container-title":["Lecture Notes in Computer Science","Languages and Compilers for Parallel Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-02436-7_13","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,31]],"date-time":"2025-10-31T06:50:04Z","timestamp":1761893404000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-02436-7_13"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,1]]},"ISBN":["9783032024350","9783032024367"],"references-count":27,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-02436-7_13","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,11,1]]},"assertion":[{"value":"1 November 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"LCPC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Workshop on Languages and Compilers for Parallel Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Lexington, KY","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"USA","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2023","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"11 October 2023","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"13 October 2023","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"36","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"lcpc2023","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.lcpcworkshop.org\/LCPC23\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}