{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T03:28:27Z","timestamp":1783740507550,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":55,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,3,30]],"date-time":"2025-03-30T00:00:00Z","timestamp":1743292800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/100000001","name":"NSF (National Science Foundation)","doi-asserted-by":"publisher","award":["CNS-2441284"],"award-info":[{"award-number":["CNS-2441284"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,3,30]]},"DOI":"10.1145\/3721146.3721943","type":"proceedings-article","created":{"date-parts":[[2025,4,1]],"date-time":"2025-04-01T17:42:05Z","timestamp":1743529325000},"page":"82-89","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Verifying Semantic Equivalence of Large Models with Equality Saturation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7850-9769","authenticated-orcid":false,"given":"Kahfi S.","family":"Zulkifli","sequence":"first","affiliation":[{"name":"University of Virginia, Charlottesville, Virginia, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8343-4845","authenticated-orcid":false,"given":"Wenbo","family":"Qian","sequence":"additional","affiliation":[{"name":"Northeastern University, Boston, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0335-1151","authenticated-orcid":false,"given":"Shaowei","family":"Zhu","sequence":"additional","affiliation":[{"name":"Amazon Web Services, Seattle, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-2144-8931","authenticated-orcid":false,"given":"Yuan","family":"Zhou","sequence":"additional","affiliation":[{"name":"Amazon Web Services, Seattle, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0164-0849","authenticated-orcid":false,"given":"Zhen","family":"Zhang","sequence":"additional","affiliation":[{"name":"Amazon Web Services, Seattle, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-3056-5729","authenticated-orcid":false,"given":"Chang","family":"Lou","sequence":"additional","affiliation":[{"name":"University of Virginia, Charlottesville, Virginia, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,4]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Cuda toolkit. https:\/\/developer.nvidia.com\/cuda-toolkit."},{"key":"e_1_3_2_1_2_1","unstructured":"Mlir-hlo: A standalone \"hlo\" mlir-based compiler. https:\/\/github.com\/tensorflow\/mlir-hlo."},{"key":"e_1_3_2_1_3_1","unstructured":"Pytorch. https:\/\/pytorch.org\/."},{"key":"e_1_3_2_1_4_1","unstructured":"Tensorflow: An end-to-end platform for machine learning. https:\/\/www.tensorflow.org\/."},{"key":"e_1_3_2_1_5_1","unstructured":"Transformers neuron. https:\/\/github.com\/aws-neuron\/transformers-neuronx."},{"key":"e_1_3_2_1_6_1","unstructured":"Xla (accelerated linear algebra). https:\/\/openxla.org\/xla."},{"key":"e_1_3_2_1_7_1","volume-title":"https:\/\/github.com\/Lightning-AI\/pytorch-lightning\/issues\/9237","author":"Manual","year":"2021","unstructured":"Manual optimization does not synchronize gradients in ddp. https:\/\/github.com\/Lightning-AI\/pytorch-lightning\/issues\/9237, 2021."},{"key":"e_1_3_2_1_8_1","volume-title":"https:\/\/github.com\/kohya-ss\/sd-scripts\/issues\/924","year":"2023","unstructured":"[bug] gradients not synchronized. https:\/\/github.com\/kohya-ss\/sd-scripts\/issues\/924, 2023."},{"key":"e_1_3_2_1_9_1","volume-title":"moving model to cpu and back to gpu breaks gradient synchronization. https:\/\/github.com\/pytorch\/pytorch\/issues\/104336","author":"Ddp","year":"2023","unstructured":"Ddp: moving model to cpu and back to gpu breaks gradient synchronization. https:\/\/github.com\/pytorch\/pytorch\/issues\/104336, 2023."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3600006.3613142"},{"key":"e_1_3_2_1_11_1","volume-title":"Proc. ACM Program. Lang., 9(POPL)","author":"Arora J.","year":"2025","unstructured":"J. Arora, S. Lu, D. Jain, T. Xu, F. Houshmand, P. M. Phothilimthana, M. Lesani, P. Narayanan, K. S. Murthy, R. Bodik, A. Sabne, and C. Mendis. Tensorright: Automated verification of tensor graph rewrites. Proc. ACM Program. Lang., 9(POPL), Jan. 2025."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"crossref","first-page":"472","DOI":"10.1145\/3492321.3519584","volume-title":"Proceedings of the Seventeenth European Conference on Computer Systems, EuroSys '22","author":"Athlur S.","year":"2022","unstructured":"S. Athlur, N. Saran, M. Sivathanu, R. Ramjee, and N. Kwatra. Varuna: scalable, low-cost training of massive deep learning models. In Proceedings of the Seventeenth European Conference on Computer Systems, EuroSys '22, page 472--487, Rennes, France, 2022."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.5555\/3495724.3495883"},{"key":"e_1_3_2_1_14_1","first-page":"578","volume-title":"13th USENIX Symposium on Operating Systems Design and Implementation (OSDI 18)","author":"Chen T.","year":"2018","unstructured":"T. Chen, T. Moreau, Z. Jiang, L. Zheng, E. Yan, H. Shen, M. Cowan, L. Wang, Y. Hu, L. Ceze, C. Guestrin, and A. Krishnamurthy. TVM: An automated End-to-End optimizing compiler for deep learning. In 13th USENIX Symposium on Operating Systems Design and Implementation (OSDI 18), pages 578--594. USENIX Association, Oct. 2018."},{"key":"e_1_3_2_1_15_1","first-page":"563","volume-title":"18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24)","author":"Choudhury A.","year":"2024","unstructured":"A. Choudhury, Y. Wang, T. Pelkonen, K. Srinivasan, A. Jain, S. Lin, D. David, S. Soleimanifard, M. Chen, A. Yadav, R. Tijoriwala, D. Samoylov, and C. Tang. MAST: Global scheduling of ML training across Geo-Distributed datacenters at hyperscale. In 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24), pages 563--580. USENIX Association, July 2024."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3314221.3314596"},{"key":"e_1_3_2_1_17_1","volume-title":"Proc. ACM Program. Lang., 6(OOPSLA1)","author":"Cl\u00e9ment B.","year":"2022","unstructured":"B. Cl\u00e9ment and A. Cohen. End-to-end translation validation for the halide language. Proc. ACM Program. Lang., 6(OOPSLA1), Apr. 2022."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"crossref","first-page":"127","DOI":"10.1007\/978-3-319-71237-6_7","volume-title":"Programming Languages and Systems: 15th Asian Symposium, APLAS 2017, Suzhou, China, November 27--29, 2017, Proceedings 15","author":"Dahiya M.","year":"2017","unstructured":"M. Dahiya and S. Bansal. Black-box equivalence checking across compiler optimizations. In Programming Languages and Systems: 15th Asian Symposium, APLAS 2017, Suzhou, China, November 27--29, 2017, Proceedings 15, pages 127--147. Springer, 2017."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICSE48619.2023.00024"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/2987550.2987583"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3578356.3592577"},{"key":"e_1_3_2_1_22_1","volume-title":"Proc. ACM Program. Lang., 5(OOPSLA)","author":"Herklotz Y.","year":"2021","unstructured":"Y. Herklotz, J. D. Pollard, N. Ramanathan, and J. Wickerson. Formal verification of high-level synthesis. Proc. ACM Program. Lang., 5(OOPSLA), Oct. 2021."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3102980.3103005"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/2402.322396"},{"key":"e_1_3_2_1_25_1","volume-title":"SOSP '23","author":"Yang Z.","unstructured":"Jang, Z. Yang, Z. Zhang, X. Jin, and M. Chowdhury. Resilient distributed training of large models using pipeline templates. SOSP '23."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"crossref","first-page":"47","DOI":"10.1145\/3341301.3359630","volume-title":"Proceedings of the 27th ACM Symposium on Operating Systems Principles, SOSP '19","author":"Jia Z.","year":"2019","unstructured":"Z. Jia, O. Padon, J. Thomas, T. Warszawski, M. Zaharia, and A. Aiken. Taso: optimizing deep learning computation with automatic generation of graph substitutions. In Proceedings of the 27th ACM Symposium on Operating Systems Principles, SOSP '19, page 47--62, Huntsville, Ontario, Canada, 2019."},{"key":"e_1_3_2_1_27_1","first-page":"1","volume":"54","author":"Khan S.","year":"2022","unstructured":"S. Khan, M. Naseer, M. Hayat, S. W. Zamir, F. S. Khan, and M. Shah. Transformers in vision: A survey. ACM computing surveys (CSUR), 54(10s):1--41, 2022.","journal-title":"Transformers in vision: A survey. ACM computing surveys (CSUR)"},{"key":"e_1_3_2_1_28_1","volume-title":"ASPLOS '23","author":"Lin J.","unstructured":"Liu, J. Lin, F. Ruffy, C. Tan, J. Li, A. Panda, and L. Zhang. Generating diverse and valid test cases for deep learning compilers. ASPLOS '23."},{"key":"e_1_3_2_1_29_1","volume-title":"Proc. ACM Program. Lang., 8(PLDI)","author":"Liu A.","year":"2024","unstructured":"A. Liu, G. Bernstein, A. Chlipala, and J. Ragan-Kelley. A verified compiler for a functional tensor language. Proc. ACM Program. Lang., 8(PLDI), June 2024."},{"key":"e_1_3_2_1_30_1","volume-title":"Proc. ACM Program. Lang., 6(OOPSLA1)","author":"Liu J.","year":"2022","unstructured":"J. Liu, Y. Wei, S. Yang, Y. Deng, and L. Zhang. Coverage-guided tensor compiler fuzzing with joint ir-pass mutation. Proc. ACM Program. Lang., 6(OOPSLA1), Apr. 2022."},{"key":"e_1_3_2_1_31_1","first-page":"91","volume-title":"Proceedings of the 16th USENIX Symposium on Operating Systems Design and Implementation, OSDI '22","author":"Lou C.","year":"2022","unstructured":"C. Lou, Y. Jing, and P. Huang. Demystifying and checking silent semantic violations in large distributed systems. In Proceedings of the 16th USENIX Symposium on Operating Systems Design and Implementation, OSDI '22, pages 91--107. USENIX Association, July 2022."},{"key":"e_1_3_2_1_32_1","volume-title":"Understanding silent data corruption in llm training","author":"Ma J.","year":"2025","unstructured":"J. Ma, H. Pei, L. Lausen, and G. Karypis. Understanding silent data corruption in llm training, 2025."},{"key":"e_1_3_2_1_33_1","first-page":"203","volume-title":"19th USENIX Conference on File and Storage Technologies (FAST 21)","author":"Mohan J.","year":"2021","unstructured":"J. Mohan, A. Phanishayee, and V. Chidambaram. CheckFreq: Frequent, Fine-Grained DNN checkpointing. In 19th USENIX Conference on File and Storage Technologies (FAST 21), pages 203--216. USENIX Association, Feb. 2021."},{"key":"e_1_3_2_1_34_1","first-page":"1","volume-title":"Proceedings of the 26th Symposium on Operating Systems Principles, SOSP '17","author":"Pei K.","year":"2017","unstructured":"K. Pei, Y. Cao, J. Yang, and S. Jana. Deepxplore: Automated whitebox testing of deep learning systems. In Proceedings of the 26th Symposium on Operating Systems Principles, SOSP '17, page 1--18, Shanghai, China, 2017."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICSE.2019.00107"},{"key":"e_1_3_2_1_36_1","volume-title":"MLSys","author":"Pienaar J. A.","year":"2021","unstructured":"J. A. Pienaar, M. Phothilimthana, M. Willsey, R. Wang, S. Roy, and Y. Yang. Equality saturation for tensor graph superoptimization. In MLSys, 2021."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3626202.3637563"},{"key":"e_1_3_2_1_38_1","volume-title":"Enhanced capabilities in large language modeling. https:\/\/ai.meta.com\/llama","author":"Research M. A.","year":"2023","unstructured":"M. A. Research. Llama 3.1: Enhanced capabilities in large language modeling. https:\/\/ai.meta.com\/llama, 2023. Version 3.1."},{"key":"e_1_3_2_1_39_1","first-page":"968","volume-title":"Proceedings of the 29th ACM Joint Meeting on European Software Engineering Conference and Symposium on the Foundations of Software Engineering, ESEC\/FSE 2021","author":"Shen Q.","year":"2021","unstructured":"Q. Shen, H. Ma, J. Chen, Y. Tian, S.-C. Cheung, and X. Chen. A comprehensive study of deep learning compiler bugs. In Proceedings of the 29th ACM Joint Meeting on European Software Engineering Conference and Symposium on the Foundations of Software Engineering, ESEC\/FSE 2021, page 968--980, Athens, Greece, 2021."},{"key":"e_1_3_2_1_40_1","volume-title":"Silent bugs in deep learning frameworks: An empirical study of keras and tensorflow. CoRR, abs\/2112.13314","author":"Tambon F.","year":"2021","unstructured":"F. Tambon, A. Nikanjam, L. An, F. Khomh, and G. Antoniol. Silent bugs in deep learning frameworks: An empirical study of keras and tensorflow. CoRR, abs\/2112.13314, 2021."},{"key":"e_1_3_2_1_41_1","volume-title":"POPL '09","author":"Stepp M.","unstructured":"Tate, M. Stepp, Z. Tatlock, and S. Lerner. Equality saturation: a new approach to optimization. POPL '09."},{"key":"e_1_3_2_1_42_1","volume-title":"A next-generation deep learning search engine. https:\/\/www.deepseek.ai","author":"Team D.","year":"2023","unstructured":"D. Team. Deepseek v3: A next-generation deep learning search engine. https:\/\/www.deepseek.ai, 2023. Version 3."},{"key":"e_1_3_2_1_43_1","first-page":"497","volume-title":"20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23)","author":"Thorpe J.","year":"2023","unstructured":"J. Thorpe, P. Zhao, J. Eyolfson, Y. Qiao, Z. Jia, M. Zhang, R. Netravali, and G. H. Xu. Bamboo: Making preemptible instances resilient for affordable training of large DNNs. In 20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23), pages 497--513. USENIX Association, Apr. 2023."},{"key":"e_1_3_2_1_44_1","first-page":"303","volume-title":"Proceedings of the 40th International Conference on Software Engineering, ICSE '18","author":"Tian Y.","year":"2018","unstructured":"Y. Tian, K. Pei, S. Jana, and B. Ray. Deeptest: automated testing of deep-neural-network-driven autonomous cars. In Proceedings of the 40th International Conference on Software Engineering, ICSE '18, page 303--314, Gothenburg, Sweden, 2018."},{"key":"e_1_3_2_1_45_1","volume-title":"TOPLAS '12","author":"Janssens G.","unstructured":"Verdoolaege, G. Janssens, and M. Bruynooghe. Equivalence checking of static affine programs using widening to handle recurrences. TOPLAS '12."},{"key":"e_1_3_2_1_46_1","volume-title":"SOSP '23","author":"Jia Z.","unstructured":"Wang, Z. Jia, S. Zheng, Z. Zhang, X. Fu, T. S. E. Ng, and Y. Wang. Fast failure recovery in distributed training with in-memory checkpoints. SOSP '23."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"crossref","first-page":"1555","DOI":"10.1109\/ASE56229.2023.00120","volume-title":"2023 38th IEEE\/ACM International Conference on Automated Software Engineering (ASE)","author":"Wang H.","year":"2023","unstructured":"H. Wang, J. Chen, C. Xie, S. Liu, Z. Wang, Q. Shen, and Y. Zhao. Mlirsmith: Random program generation for fuzzing mlir compiler infrastructure. In 2023 38th IEEE\/ACM International Conference on Automated Software Engineering (ASE), pages 1555--1566, 2023."},{"key":"e_1_3_2_1_48_1","first-page":"37","volume-title":"15th USENIX Symposium on Operating Systems Design and Implementation (OSDI 21)","author":"Wang H.","year":"2021","unstructured":"H. Wang, J. Zhai, M. Gao, Z. Ma, S. Tang, L. Zheng, Y. Li, K. Rong, Y. Chen, and Z. Jia. PET: Optimizing tensor programs with partially equivalent transformations and automated corrections. In 15th USENIX Symposium on Operating Systems Design and Implementation (OSDI 21), pages 37--54. USENIX Association, July 2021."},{"key":"e_1_3_2_1_49_1","first-page":"798","volume-title":"Proceedings of the 44th International Conference on Software Engineering, ICSE '22","author":"Wang J.","year":"2022","unstructured":"J. Wang, T. Lutellier, S. Qian, H. V. Pham, and L. Tan. Eagle: creating equivalent graphs to test deep learning libraries. In Proceedings of the 44th International Conference on Software Engineering, ICSE '22, page 798--810, Pittsburgh, Pennsylvania, 2022."},{"key":"e_1_3_2_1_50_1","volume-title":"SOSP '23","author":"Wang S.","unstructured":"S. Wang, G. Zhang, J. Wei, Y. Wang, J. Wu, and Q. Luo. Understanding silent data corruptions in a large production cpu population. SOSP '23."},{"key":"e_1_3_2_1_51_1","volume-title":"Mirage: A multi-level superoptimizer for tensor programs","author":"Wu M.","year":"2024","unstructured":"M. Wu, X. Cheng, S. Liu, C. Shi, J. Ji, K. Ao, P. Velliengiri, X. Miao, O. Padon, and Z. Jia. Mirage: A multi-level superoptimizer for tensor programs, 2024."},{"key":"e_1_3_2_1_52_1","first-page":"5772","volume-title":"Proceedings of the Twenty-Eighth International Joint Conference on Artificial Intelligence, IJCAI-19","author":"Xie X.","year":"2019","unstructured":"X. Xie, L. Ma, H. Wang, Y. Li, Y. Liu, and X. Li. Diffchaser: Detecting disagreements for deep neural networks. In Proceedings of the Twenty-Eighth International Joint Conference on Artificial Intelligence, IJCAI-19, pages 5772--5778. International Joint Conferences on Artificial Intelligence Organization, 7 2019."},{"key":"e_1_3_2_1_53_1","volume-title":"Proc. ACM Program. Lang., 7(PLDI)","author":"Zhang Y.","year":"2023","unstructured":"Y. Zhang, Y. R. Wang, O. Flatt, D. Cao, P. Zucker, E. Rosenthal, Z. Tatlock, and M. Willsey. Better together: Unifying datalog and equality saturation. Proc. ACM Program. Lang., 7(PLDI), June 2023."},{"key":"e_1_3_2_1_54_1","first-page":"559","volume-title":"16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22)","author":"Zheng L.","year":"2022","unstructured":"L. Zheng, Z. Li, H. Zhang, Y. Zhuang, Z. Chen, Y. Huang, Y. Wang, Y. Xu, D. Zhuo, E. P. Xing, J. E. Gonzalez, and I. Stoica. Alpa: Automating inter- and Intra-Operator parallelism for distributed deep learning. In 16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22), pages 559--578. USENIX Association, July 2022."},{"key":"e_1_3_2_1_55_1","volume-title":"Proc. ACM Program. Lang., 8(OOPSLA2)","author":"Zhou C.","year":"2024","unstructured":"C. Zhou, B. Qian, G. Go, Q. Zhang, S. Li, and Y. Jiang. Polyjuice: Detecting mis-compilation bugs in tensor compilers with equality saturation based rewriting. Proc. ACM Program. Lang., 8(OOPSLA2), Oct. 2024."}],"event":{"name":"EuroMLSys '25: 5th Workshop on Machine Learning and Systems","location":"World Trade Center Rotterdam Netherlands","acronym":"EuroMLSys '25","sponsor":["SIGOPS ACM Special Interest Group on Operating Systems"]},"container-title":["Proceedings of the 5th Workshop on Machine Learning and Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3721146.3721943","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3721146.3721943","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3721146.3721943","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:57:39Z","timestamp":1750298259000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3721146.3721943"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,3,30]]},"references-count":55,"alternative-id":["10.1145\/3721146.3721943","10.1145\/3721146"],"URL":"https:\/\/doi.org\/10.1145\/3721146.3721943","relation":{},"subject":[],"published":{"date-parts":[[2025,3,30]]},"assertion":[{"value":"2025-04-01","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}