{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,1]],"date-time":"2025-10-01T15:31:15Z","timestamp":1759332675995,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":75,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,10,8]],"date-time":"2022-10-08T00:00:00Z","timestamp":1665187200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["CNS-2147909"],"award-info":[{"award-number":["CNS-2147909"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"name":"DARPA","award":["the Real Time Machine Learning"],"award-info":[{"award-number":["the Real Time Machine Learning"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,10,8]]},"DOI":"10.1145\/3559009.3569651","type":"proceedings-article","created":{"date-parts":[[2023,1,27]],"date-time":"2023-01-27T14:02:50Z","timestamp":1674828170000},"page":"517-529","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["Collage"],"prefix":"10.1145","author":[{"given":"Byungsoo","family":"Jeon","sequence":"first","affiliation":[{"name":"Carnegie Mellon University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sunghyun","family":"Park","sequence":"additional","affiliation":[{"name":"OctoML"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Peiyuan","family":"Liao","sequence":"additional","affiliation":[{"name":"Carnegie Mellon University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sheng","family":"Xu","sequence":"additional","affiliation":[{"name":"Amazon Web Services"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tianqi","family":"Chen","sequence":"additional","affiliation":[{"name":"Carnegie Mellon University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhihao","family":"Jia","sequence":"additional","affiliation":[{"name":"Carnegie Mellon University"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,1,27]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"[n.d.]. Apple Neural Engine (ANE). https:\/\/www.apple.com\/newsroom\/2020\/11\/apple-unleashes-m1\/. Accessed: 2021-08-25.  [n.d.]. Apple Neural Engine (ANE). https:\/\/www.apple.com\/newsroom\/2020\/11\/apple-unleashes-m1\/. Accessed: 2021-08-25."},{"key":"e_1_3_2_1_2_1","unstructured":"[n.d.]. DLPack: Open In Memory Tensor Structure. https:\/\/github.com\/dmlc\/dlpack. Accessed: 2022-04-05.  [n.d.]. DLPack: Open In Memory Tensor Structure. https:\/\/github.com\/dmlc\/dlpack. Accessed: 2022-04-05."},{"key":"e_1_3_2_1_3_1","unstructured":"[n.d.]. Intel OneDNN. https:\/\/software.intel.com\/content\/www\/us\/en\/develop\/tools\/oneapi\/components\/onednn.html. Accessed: 2021-09-27.  [n.d.]. Intel OneDNN. https:\/\/software.intel.com\/content\/www\/us\/en\/develop\/tools\/oneapi\/components\/onednn.html. Accessed: 2021-09-27."},{"key":"e_1_3_2_1_4_1","unstructured":"[n.d.]. Intel OpenVINO. https:\/\/software.intel.com\/content\/www\/us\/en\/develop\/tools\/openvino-toolkit.html. Accessed: 2021-09-27.  [n.d.]. Intel OpenVINO. https:\/\/software.intel.com\/content\/www\/us\/en\/develop\/tools\/openvino-toolkit.html. Accessed: 2021-09-27."},{"key":"e_1_3_2_1_5_1","unstructured":"[n.d.]. NVIDIA cuBLAS. https:\/\/developer.nvidia.com\/cublas. Accessed: 2021-08-05.  [n.d.]. NVIDIA cuBLAS. https:\/\/developer.nvidia.com\/cublas. Accessed: 2021-08-05."},{"volume-title":"NVIDIA Deep Learning Accelerator (NVDLA)","key":"e_1_3_2_1_6_1","unstructured":"[n.d.]. NVIDIA Deep Learning Accelerator (NVDLA) . http:\/\/nvdla.org\/. Accessed: 2021-08-25. [n.d.]. NVIDIA Deep Learning Accelerator (NVDLA). http:\/\/nvdla.org\/. Accessed: 2021-08-25."},{"key":"e_1_3_2_1_7_1","unstructured":"[n.d.]. NVIDIA TensorRT. https:\/\/developer.nvidia.com\/tensorrt. Accessed: 2021-08-05.  [n.d.]. NVIDIA TensorRT. https:\/\/developer.nvidia.com\/tensorrt. Accessed: 2021-08-05."},{"key":"e_1_3_2_1_8_1","unstructured":"[n.d.]. NVIDIA TensorRT Deserialization. https:\/\/docs.nvidia.com\/deeplearning\/tensorrt\/developer-guide\/index.html. Accessed: 2022-06-09.  [n.d.]. NVIDIA TensorRT Deserialization. https:\/\/docs.nvidia.com\/deeplearning\/tensorrt\/developer-guide\/index.html. Accessed: 2022-06-09."},{"key":"e_1_3_2_1_9_1","unstructured":"[n.d.]. Tensorflow XLA. https:\/\/www.tensorflow.org\/xla. Accessed: 2021-09-27.  [n.d.]. Tensorflow XLA. https:\/\/www.tensorflow.org\/xla. Accessed: 2021-09-27."},{"key":"e_1_3_2_1_10_1","volume-title":"Tensorflow: A system for large-scale machine learning. In 12th {USENIX} symposium on operating systems design and implementation ({OSDI} 16). 265--283.","author":"Abadi Mart\u00edn","year":"2016","unstructured":"Mart\u00edn Abadi , Paul Barham , Jianmin Chen , Zhifeng Chen , Andy Davis , Jeffrey Dean , Matthieu Devin , Sanjay Ghemawat , Geoffrey Irving , Michael Isard , 2016 . Tensorflow: A system for large-scale machine learning. In 12th {USENIX} symposium on operating systems design and implementation ({OSDI} 16). 265--283. Mart\u00edn Abadi, Paul Barham, Jianmin Chen, Zhifeng Chen, Andy Davis, Jeffrey Dean, Matthieu Devin, Sanjay Ghemawat, Geoffrey Irving, Michael Isard, et al. 2016. Tensorflow: A system for large-scale machine learning. In 12th {USENIX} symposium on operating systems design and implementation ({OSDI} 16). 265--283."},{"key":"e_1_3_2_1_11_1","volume-title":"NeurIPS ML for Systems Workshop.","author":"Abdolrashidi Amirali","year":"2019","unstructured":"Amirali Abdolrashidi , Qiumin Xu , Shibo Wang , Sudip Roy , and Yanqi Zhou . 2019 . Learning to fuse . In NeurIPS ML for Systems Workshop. Amirali Abdolrashidi, Qiumin Xu, Shibo Wang, Sudip Roy, and Yanqi Zhou. 2019. Learning to fuse. In NeurIPS ML for Systems Workshop."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3306346.3322967"},{"key":"e_1_3_2_1_13_1","volume-title":"NIPS Machine Learning for Systems Workshop.","author":"Addanki Ravichandra","year":"2018","unstructured":"Ravichandra Addanki , Shaileshh Bojja Venkatakrishnan , Shreyan Gupta , Hongzi Mao , and Mohammad Alizadeh . 2018 . Placeto: Efficient progressive device placement optimization . In NIPS Machine Learning for Systems Workshop. Ravichandra Addanki, Shaileshh Bojja Venkatakrishnan, Shreyan Gupta, Hongzi Mao, and Mohammad Alizadeh. 2018. Placeto: Efficient progressive device placement optimization. In NIPS Machine Learning for Systems Workshop."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/2628071.2628092"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/2858788.2688521"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.5555\/3314872.3314896"},{"key":"e_1_3_2_1_17_1","volume-title":"Optimizing GPU Deep Learning Operators with Polyhedral Scheduling Constraint Injection. In 2022 IEEE\/ACM International Symposium on Code Generation and Optimization (CGO). IEEE, 313--324","author":"Bastoul Cedric","year":"2022","unstructured":"Cedric Bastoul , Zhen Zhang , Harenome Razanajato , Nelson Lossing , Adilla Susungi , Javier de Juan , Etienne Filhol , Baptiste Jarry , Gianpietro Consolaro , and Renwei Zhang . 2022 . Optimizing GPU Deep Learning Operators with Polyhedral Scheduling Constraint Injection. In 2022 IEEE\/ACM International Symposium on Code Generation and Optimization (CGO). IEEE, 313--324 . Cedric Bastoul, Zhen Zhang, Harenome Razanajato, Nelson Lossing, Adilla Susungi, Javier de Juan, Etienne Filhol, Baptiste Jarry, Gianpietro Consolaro, and Renwei Zhang. 2022. Optimizing GPU Deep Learning Operators with Polyhedral Scheduling Constraint Injection. In 2022 IEEE\/ACM International Symposium on Code Generation and Optimization (CGO). IEEE, 313--324."},{"key":"e_1_3_2_1_18_1","volume-title":"On optimizing operator fusion plans for large-scale machine learning in systemml. arXiv preprint arXiv:1801.00829","author":"Boehm Matthias","year":"2018","unstructured":"Matthias Boehm , Berthold Reinwald , Dylan Hutchison , Alexandre V Evfimievski , and Prithviraj Sen . 2018. On optimizing operator fusion plans for large-scale machine learning in systemml. arXiv preprint arXiv:1801.00829 ( 2018 ). Matthias Boehm, Berthold Reinwald, Dylan Hutchison, Alexandre V Evfimievski, and Prithviraj Sen. 2018. On optimizing operator fusion plans for large-scale machine learning in systemml. arXiv preprint arXiv:1801.00829 (2018)."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.5555\/3314872.3314902"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3410463.3414635"},{"key":"e_1_3_2_1_21_1","unstructured":"Tianqi Chen Thierry Moreau Ziheng Jiang Lianmin Zheng Eddie Yan Haichen Shen Meghan Cowan Leyuan Wang Yuwei Hu Luis Ceze etal 2018. {TVM}: An automated end-to-end optimizing compiler for deep learning. In 13th {USENIX} Symposium on Operating Systems Design and Implementation ({OSDI} 18). 578--594.  Tianqi Chen Thierry Moreau Ziheng Jiang Lianmin Zheng Eddie Yan Haichen Shen Meghan Cowan Leyuan Wang Yuwei Hu Luis Ceze et al. 2018. {TVM}: An automated end-to-end optimizing compiler for deep learning. In 13th {USENIX} Symposium on Operating Systems Design and Implementation ({OSDI} 18). 578--594."},{"key":"e_1_3_2_1_22_1","volume-title":"Learning to optimize tensor programs. arXiv preprint arXiv:1805.08166","author":"Chen Tianqi","year":"2018","unstructured":"Tianqi Chen , Lianmin Zheng , Eddie Yan , Ziheng Jiang , Thierry Moreau , Luis Ceze , Carlos Guestrin , and Arvind Krishnamurthy . 2018. Learning to optimize tensor programs. arXiv preprint arXiv:1805.08166 ( 2018 ). Tianqi Chen, Lianmin Zheng, Eddie Yan, Ziheng Jiang, Thierry Moreau, Luis Ceze, Carlos Guestrin, and Arvind Krishnamurthy. 2018. Learning to optimize tensor programs. arXiv preprint arXiv:1805.08166 (2018)."},{"key":"e_1_3_2_1_23_1","volume-title":"cudnn: Efficient primitives for deep learning. arXiv preprint arXiv:1410.0759","author":"Chetlur Sharan","year":"2014","unstructured":"Sharan Chetlur , Cliff Woolley , Philippe Vandermersch , Jonathan Cohen , John Tran , Bryan Catanzaro , and Evan Shelhamer . 2014. cudnn: Efficient primitives for deep learning. arXiv preprint arXiv:1410.0759 ( 2014 ). Sharan Chetlur, Cliff Woolley, Philippe Vandermersch, Jonathan Cohen, John Tran, Bryan Catanzaro, and Evan Shelhamer. 2014. cudnn: Efficient primitives for deep learning. arXiv preprint arXiv:1410.0759 (2014)."},{"key":"e_1_3_2_1_24_1","unstructured":"Scott Cyphers Arjun K Bansal Anahita Bhiwandiwalla Jayaram Bobba Matthew Brookhart Avijit Chakraborty Will Constable Christian Convey Leona Cook Omar Kanawi etal 2018. Intel ngraph: An intermediate representation compiler and executor for deep learning. arXiv preprint arXiv:1801.08058 (2018).  Scott Cyphers Arjun K Bansal Anahita Bhiwandiwalla Jayaram Bobba Matthew Brookhart Avijit Chakraborty Will Constable Christian Convey Leona Cook Omar Kanawi et al. 2018. Intel ngraph: An intermediate representation compiler and executor for deep learning. arXiv preprint arXiv:1801.08058 (2018)."},{"key":"e_1_3_2_1_25_1","volume-title":"Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805","author":"Devlin Jacob","year":"2018","unstructured":"Jacob Devlin , Ming-Wei Chang , Kenton Lee , and Kristina Toutanova . 2018 . Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018). Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2018. Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)."},{"key":"e_1_3_2_1_26_1","volume-title":"International Conference on Machine Learning. PMLR","author":"Diamos Greg","year":"2016","unstructured":"Greg Diamos , Shubho Sengupta , Bryan Catanzaro , Mike Chrzanowski , Adam Coates , Erich Elsen , Jesse Engel , Awni Hannun , and Sanjeev Satheesh . 2016 . Persistent rnns: Stashing recurrent weights on-chip . In International Conference on Machine Learning. PMLR , 2024--2033. Greg Diamos, Shubho Sengupta, Bryan Catanzaro, Mike Chrzanowski, Adam Coates, Erich Elsen, Jesse Engel, Awni Hannun, and Sanjeev Satheesh. 2016. Persistent rnns: Stashing recurrent weights on-chip. In International Conference on Machine Learning. PMLR, 2024--2033."},{"key":"e_1_3_2_1_27_1","volume-title":"SPOOF: Sum-Product Optimization and Operator Fusion for Large-Scale Machine Learning.. In CIDR.","author":"Elgamal Tarek","year":"2017","unstructured":"Tarek Elgamal , Shangyu Luo , Matthias Boehm , Alexandre V Evfimievski , Shirish Tatikonda , Berthold Reinwald , and Prithviraj Sen . 2017 . SPOOF: Sum-Product Optimization and Operator Fusion for Large-Scale Machine Learning.. In CIDR. Tarek Elgamal, Shangyu Luo, Matthias Boehm, Alexandre V Evfimievski, Shirish Tatikonda, Berthold Reinwald, and Prithviraj Sen. 2017. SPOOF: Sum-Product Optimization and Operator Fusion for Large-Scale Machine Learning.. In CIDR."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.14778\/3407790.3407857"},{"key":"e_1_3_2_1_29_1","volume-title":"Proceedings of Machine Learning and Systems 3","author":"Fegade Pratik","year":"2021","unstructured":"Pratik Fegade , Tianqi Chen , Phillip Gibbons , and Todd Mowry . 2021 . Cortex: A Compiler for Recursive Deep Learning Models . Proceedings of Machine Learning and Systems 3 (2021). Pratik Fegade, Tianqi Chen, Phillip Gibbons, and Todd Mowry. 2021. Cortex: A Compiler for Recursive Deep Learning Models. Proceedings of Machine Learning and Systems 3 (2021)."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.5555\/2503308.2503311"},{"key":"e_1_3_2_1_31_1","volume-title":"International Conference on Machine Learning. PMLR, 1676--1684","author":"Gao Yuanxiang","year":"2018","unstructured":"Yuanxiang Gao , Li Chen , and Baochun Li . 2018 . Spotlight: Optimizing device placement for training deep neural networks . In International Conference on Machine Learning. PMLR, 1676--1684 . Yuanxiang Gao, Li Chen, and Baochun Li. 2018. Spotlight: Optimizing device placement for training deep neural networks. In International Conference on Machine Learning. PMLR, 1676--1684."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/3422622"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00685"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/PACT52795.2021.00010"},{"key":"e_1_3_2_1_36_1","unstructured":"Zhihao Jia Sina Lin Charles R Qi and Alex Aiken. 2018. Exploring Hidden Dimensions in Parallelizing Convolutional Neural Networks.. In ICML. 2279--2288.  Zhihao Jia Sina Lin Charles R Qi and Alex Aiken. 2018. Exploring Hidden Dimensions in Parallelizing Convolutional Neural Networks.. In ICML. 2279--2288."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359630"},{"key":"e_1_3_2_1_38_1","volume-title":"Optimizing dnn computation with relaxed graph substitutions. SysML 2019","author":"Jia Zhihao","year":"2019","unstructured":"Zhihao Jia , James Thomas , Tod Warszawski , Mingyu Gao , Matei Zaharia , and Alex Aiken . 2019. Optimizing dnn computation with relaxed graph substitutions. SysML 2019 ( 2019 ). Zhihao Jia, James Thomas, Tod Warszawski, Mingyu Gao, Matei Zaharia, and Alex Aiken. 2019. Optimizing dnn computation with relaxed graph substitutions. SysML 2019 (2019)."},{"key":"e_1_3_2_1_39_1","volume-title":"Proceedings of Machine Learning and Systems, A. Talwalkar, V. Smith, and M. Zaharia (Eds.)","volume":"1","author":"Jia Zhihao","year":"2019","unstructured":"Zhihao Jia , Matei Zaharia , and Alex Aiken . 2019 . Beyond Data and Model Parallelism for Deep Neural Networks .. In Proceedings of Machine Learning and Systems, A. Talwalkar, V. Smith, and M. Zaharia (Eds.) , Vol. 1 . 1--13. https:\/\/proceedings.mlsys.org\/paper\/2019\/file\/c74d97b01eae257e44aa9d5bade97baf-Paper.pdf Zhihao Jia, Matei Zaharia, and Alex Aiken. 2019. Beyond Data and Model Parallelism for Deep Neural Networks.. In Proceedings of Machine Learning and Systems, A. Talwalkar, V. Smith, and M. Zaharia (Eds.), Vol. 1. 1--13. https:\/\/proceedings.mlsys.org\/paper\/2019\/file\/c74d97b01eae257e44aa9d5bade97baf-Paper.pdf"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/3079856.3080246"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3453483.3454038"},{"key":"e_1_3_2_1_42_1","volume-title":"Proc. Workshop ML Syst. NeurIPS. 1--6.","author":"Kaufman Samuel","year":"2019","unstructured":"Samuel Kaufman , Phitchaya Mangpo Phothilimthana , and Mike Burrows . 2019 . Learned TPU cost model for XLA tensor programs . In Proc. Workshop ML Syst. NeurIPS. 1--6. Samuel Kaufman, Phitchaya Mangpo Phothilimthana, and Mike Burrows. 2019. Learned TPU cost model for XLA tensor programs. In Proc. Workshop ML Syst. NeurIPS. 1--6."},{"key":"e_1_3_2_1_43_1","unstructured":"Jehandad Khan Paul Fultz Artem Tamazov Daniel Lowell Chao Liu Michael Melesse Murali Nandhimandalam Kamil Nasyrov Ilya Perminov Tejash Shah etal 2019. MIOpen: An open source library for deep learning primitives. arXiv preprint arXiv:1910.00078 (2019).  Jehandad Khan Paul Fultz Artem Tamazov Daniel Lowell Chao Liu Michael Melesse Murali Nandhimandalam Kamil Nasyrov Ilya Perminov Tejash Shah et al. 2019. MIOpen: An open source library for deep learning primitives. arXiv preprint arXiv:1910.00078 (2019)."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/3133901"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/CGO51591.2021.9370308"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/CGO53902.2022.9741270"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/CGO51591.2021.9370324"},{"key":"e_1_3_2_1_48_1","volume-title":"14th USENIX Symposium on Operating Systems Design and Implementation (OSDI 20)","author":"Ma Lingxiao","year":"2020","unstructured":"Lingxiao Ma , Zhiqiang Xie , Zhi Yang , Jilong Xue , Youshan Miao , Wei Cui , Wenxiang Hu , Fan Yang , Lintao Zhang , and Lidong Zhou . 2020 . Rammer: Enabling Holistic Deep Learning Compiler Optimizations with rTasks . In 14th USENIX Symposium on Operating Systems Design and Implementation (OSDI 20) . USENIX Association, 881--897. https:\/\/www.usenix.org\/conference\/osdi20\/presentation\/ma Lingxiao Ma, Zhiqiang Xie, Zhi Yang, Jilong Xue, Youshan Miao, Wei Cui, Wenxiang Hu, Fan Yang, Lintao Zhang, and Lidong Zhou. 2020. Rammer: Enabling Holistic Deep Learning Compiler Optimizations with rTasks. In 14th USENIX Symposium on Operating Systems Design and Implementation (OSDI 20). USENIX Association, 881--897. https:\/\/www.usenix.org\/conference\/osdi20\/presentation\/ma"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/3410463.3414647"},{"key":"e_1_3_2_1_50_1","volume-title":"International Conference on Learning Representations.","author":"Mirhoseini Azalia","year":"2018","unstructured":"Azalia Mirhoseini , Anna Goldie , Hieu Pham , Benoit Steiner , Quoc V Le , and Jeff Dean . 2018 . A hierarchical model for device placement . In International Conference on Learning Representations. Azalia Mirhoseini, Anna Goldie, Hieu Pham, Benoit Steiner, Quoc V Le, and Jeff Dean. 2018. A hierarchical model for device placement. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_51_1","volume-title":"International Conference on Machine Learning. PMLR, 2430--2439","author":"Mirhoseini Azalia","year":"2017","unstructured":"Azalia Mirhoseini , Hieu Pham , Quoc V Le , Benoit Steiner , Rasmus Larsen , Yuefeng Zhou , Naveen Kumar , Mohammad Norouzi , Samy Bengio , and Jeff Dean . 2017 . Device placement optimization with reinforcement learning . In International Conference on Machine Learning. PMLR, 2430--2439 . Azalia Mirhoseini, Hieu Pham, Quoc V Le, Benoit Steiner, Rasmus Larsen, Yuefeng Zhou, Naveen Kumar, Mohammad Norouzi, Samy Bengio, and Jeff Dean. 2017. Device placement optimization with reinforcement learning. In International Conference on Machine Learning. PMLR, 2430--2439."},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359646"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1145\/3453483.3454083"},{"key":"e_1_3_2_1_54_1","volume-title":"Reinforced Genetic Algorithm Learning for Optimizing Computation Graphs. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=rkxDoJBYPB","author":"Paliwal Aditya","year":"2020","unstructured":"Aditya Paliwal , Felix Gimeno , Vinod Nair , Yujia Li , Miles Lubin , Pushmeet Kohli , and Oriol Vinyals . 2020 . Reinforced Genetic Algorithm Learning for Optimizing Computation Graphs. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=rkxDoJBYPB Aditya Paliwal, Felix Gimeno, Vinod Nair, Yujia Li, Miles Lubin, Pushmeet Kohli, and Oriol Vinyals. 2020. Reinforced Genetic Algorithm Learning for Optimizing Computation Graphs. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=rkxDoJBYPB"},{"key":"e_1_3_2_1_55_1","volume-title":"Pytorch: An imperative style, high-performance deep learning library. Advances in neural information processing systems 32","author":"Paszke Adam","year":"2019","unstructured":"Adam Paszke , Sam Gross , Francisco Massa , Adam Lerer , James Bradbury , Gregory Chanan , Trevor Killeen , Zeming Lin , Natalia Gimelshein , Luca Antiga , 2019 . Pytorch: An imperative style, high-performance deep learning library. Advances in neural information processing systems 32 (2019), 8026--8037. Adam Paszke, Sam Gross, Francisco Massa, Adam Lerer, James Bradbury, Gregory Chanan, Trevor Killeen, Zeming Lin, Natalia Gimelshein, Luca Antiga, et al. 2019. Pytorch: An imperative style, high-performance deep learning library. Advances in neural information processing systems 32 (2019), 8026--8037."},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1109\/PACT52795.2021.00008"},{"key":"e_1_3_2_1_57_1","volume-title":"Unsupervised representation learning with deep convolutional generative adversarial networks. arXiv preprint arXiv:1511.06434","author":"Radford Alec","year":"2015","unstructured":"Alec Radford , Luke Metz , and Soumith Chintala . 2015. Unsupervised representation learning with deep convolutional generative adversarial networks. arXiv preprint arXiv:1511.06434 ( 2015 ). Alec Radford, Luke Metz, and Soumith Chintala. 2015. Unsupervised representation learning with deep convolutional generative adversarial networks. arXiv preprint arXiv:1511.06434 (2015)."},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1145\/2499370.2462176"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1109\/PACT.2019.00035"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1145\/3211346.3211348"},{"key":"e_1_3_2_1_61_1","volume-title":"Glow: Graph lowering compiler techniques for neural networks. arXiv preprint arXiv:1805.00907","author":"Rotem Nadav","year":"2018","unstructured":"Nadav Rotem , Jordan Fix , Saleem Abdulrasool , Garret Catron , Summer Deng , Roman Dzhabarov , Nick Gibson , James Hegeman , Meghan Lele , Roman Levenstein , 2018 . Glow: Graph lowering compiler techniques for neural networks. arXiv preprint arXiv:1805.00907 (2018). Nadav Rotem, Jordan Fix, Saleem Abdulrasool, Garret Catron, Summer Deng, Roman Dzhabarov, Nick Gibson, James Hegeman, Meghan Lele, Roman Levenstein, et al. 2018. Glow: Graph lowering compiler techniques for neural networks. arXiv preprint arXiv:1805.00907 (2018)."},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.1109\/IISWC53511.2021.00030"},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1145\/3297858.3304072"},{"key":"e_1_3_2_1_64_1","first-page":"15451","article-title":"Efficient algorithms for device placement of dnn graph operators","volume":"33","author":"Tarnawski Jakub M","year":"2020","unstructured":"Jakub M Tarnawski , Amar Phanishayee , Nikhil Devanur , Divya Mahajan , and Fanny Nina Paravecino . 2020 . Efficient algorithms for device placement of dnn graph operators . Advances in Neural Information Processing Systems 33 (2020), 15451 -- 15463 . Jakub M Tarnawski, Amar Phanishayee, Nikhil Devanur, Divya Mahajan, and Fanny Nina Paravecino. 2020. Efficient algorithms for device placement of dnn graph operators. Advances in Neural Information Processing Systems 33 (2020), 15451--15463.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.1145\/2908080.2908105"},{"key":"e_1_3_2_1_66_1","volume-title":"Tensor comprehensions: Framework-agnostic high-performance machine learning abstractions. arXiv preprint arXiv:1802.04730","author":"Vasilache Nicolas","year":"2018","unstructured":"Nicolas Vasilache , Oleksandr Zinenko , Theodoros Theodoridis , Priya Goyal , Zachary DeVito , William S Moses , Sven Verdoolaege , Andrew Adams , and Albert Cohen . 2018. Tensor comprehensions: Framework-agnostic high-performance machine learning abstractions. arXiv preprint arXiv:1802.04730 ( 2018 ). Nicolas Vasilache, Oleksandr Zinenko, Theodoros Theodoridis, Priya Goyal, Zachary DeVito, William S Moses, Sven Verdoolaege, Andrew Adams, and Albert Cohen. 2018. Tensor comprehensions: Framework-agnostic high-performance machine learning abstractions. arXiv preprint arXiv:1802.04730 (2018)."},{"volume-title":"High-Performance Computing on the Intel\u00ae Xeon Phi\u2122","author":"Wang Endong","key":"e_1_3_2_1_67_1","unstructured":"Endong Wang , Qing Zhang , Bo Shen , Guangyong Zhang , Xiaowei Lu , Qing Wu , and Yajuan Wang . 2014. Intel math kernel library . In High-Performance Computing on the Intel\u00ae Xeon Phi\u2122 . Springer , 167--188. Endong Wang, Qing Zhang, Bo Shen, Guangyong Zhang, Xiaowei Lu, Qing Wu, and Yajuan Wang. 2014. Intel math kernel library. In High-Performance Computing on the Intel\u00ae Xeon Phi\u2122. Springer, 167--188."},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.634"},{"key":"e_1_3_2_1_69_1","volume-title":"Yisu Remy Wang, Max Willsey, Sudip Roy, and Jacques Pienaar.","author":"Yang Yichen","year":"2021","unstructured":"Yichen Yang , Phitchaya Mangpo Phothilimtha , Yisu Remy Wang, Max Willsey, Sudip Roy, and Jacques Pienaar. 2021 . Equality Saturation for Tensor Graph Superoptimization . arXiv:cs.AI\/2101.01332 Yichen Yang, Phitchaya Mangpo Phothilimtha, Yisu Remy Wang, Max Willsey, Sudip Roy, and Jacques Pienaar. 2021. Equality Saturation for Tensor Graph Superoptimization. arXiv:cs.AI\/2101.01332"},{"key":"e_1_3_2_1_70_1","volume-title":"Ansor: Generating High-Performance Tensor Programs for Deep Learning. In 14th USENIX Symposium on Operating Systems Design and Implementation (OSDI).","author":"Zheng Lianmin","year":"2020","unstructured":"Lianmin Zheng , Chengfan Jia , Minmin Sun , Zhao Wu , Cody Hao Yu , Ameer Haj-Ali , Yida Wang , Jun Yang , Danyang Zhuo , Koushik Sen , Joseph E. Gonzalez , and Ion Stoica . 2020 . Ansor: Generating High-Performance Tensor Programs for Deep Learning. In 14th USENIX Symposium on Operating Systems Design and Implementation (OSDI). Lianmin Zheng, Chengfan Jia, Minmin Sun, Zhao Wu, Cody Hao Yu, Ameer Haj-Ali, Yida Wang, Jun Yang, Danyang Zhuo, Koushik Sen, Joseph E. Gonzalez, and Ion Stoica. 2020. Ansor: Generating High-Performance Tensor Programs for Deep Learning. In 14th USENIX Symposium on Operating Systems Design and Implementation (OSDI)."},{"key":"e_1_3_2_1_71_1","volume-title":"Alpa: Automating Inter-and Intra-Operator Parallelism for Distributed Deep Learning. arXiv preprint arXiv:2201.12023","author":"Zheng Lianmin","year":"2022","unstructured":"Lianmin Zheng , Zhuohan Li , Hao Zhang , Yonghao Zhuang , Zhifeng Chen , Yanping Huang , Yida Wang , Yuanzhong Xu , Danyang Zhuo , Joseph E Gonzalez , 2022 . Alpa: Automating Inter-and Intra-Operator Parallelism for Distributed Deep Learning. arXiv preprint arXiv:2201.12023 (2022). Lianmin Zheng, Zhuohan Li, Hao Zhang, Yonghao Zhuang, Zhifeng Chen, Yanping Huang, Yida Wang, Yuanzhong Xu, Danyang Zhuo, Joseph E Gonzalez, et al. 2022. Alpa: Automating Inter-and Intra-Operator Parallelism for Distributed Deep Learning. arXiv preprint arXiv:2201.12023 (2022)."},{"key":"e_1_3_2_1_72_1","doi-asserted-by":"publisher","DOI":"10.1145\/3373376.3378508"},{"key":"e_1_3_2_1_73_1","volume-title":"Fusionstitching: boosting memory intensive computations for deep learning workloads. arXiv preprint arXiv:2009.10924","author":"Zheng Zhen","year":"2020","unstructured":"Zhen Zheng , Pengzhan Zhao , Guoping Long , Feiwen Zhu , Kai Zhu , Wenyi Zhao , Lansong Diao , Jun Yang , and Wei Lin . 2020. Fusionstitching: boosting memory intensive computations for deep learning workloads. arXiv preprint arXiv:2009.10924 ( 2020 ). Zhen Zheng, Pengzhan Zhao, Guoping Long, Feiwen Zhu, Kai Zhu, Wenyi Zhao, Lansong Diao, Jun Yang, and Wei Lin. 2020. Fusionstitching: boosting memory intensive computations for deep learning workloads. arXiv preprint arXiv:2009.10924 (2020)."},{"key":"e_1_3_2_1_74_1","volume-title":"Gdp: Generalized device placement for dataflow graphs. arXiv preprint arXiv:1910.01578","author":"Zhou Yanqi","year":"2019","unstructured":"Yanqi Zhou , Sudip Roy , Amirali Abdolrashidi , Daniel Wong , Peter C Ma , Qiumin Xu , Ming Zhong , Hanxiao Liu , Anna Goldie , Azalia Mirhoseini , 2019 . Gdp: Generalized device placement for dataflow graphs. arXiv preprint arXiv:1910.01578 (2019). Yanqi Zhou, Sudip Roy, Amirali Abdolrashidi, Daniel Wong, Peter C Ma, Qiumin Xu, Ming Zhong, Hanxiao Liu, Anna Goldie, Azalia Mirhoseini, et al. 2019. Gdp: Generalized device placement for dataflow graphs. arXiv preprint arXiv:1910.01578 (2019)."},{"key":"e_1_3_2_1_75_1","volume-title":"Le","author":"Zoph Barret","year":"2018","unstructured":"Barret Zoph , Vijay Vasudevan , Jonathon Shlens , and Quoc V . Le . 2018 . Learning Transferable Architectures for Scalable Image Recognition . https:\/\/arxiv.org\/pdf\/1707.07012.pdf Barret Zoph, Vijay Vasudevan, Jonathon Shlens, and Quoc V. Le. 2018. Learning Transferable Architectures for Scalable Image Recognition. https:\/\/arxiv.org\/pdf\/1707.07012.pdf"}],"event":{"name":"PACT '22: International Conference on Parallel Architectures and Compilation Techniques","sponsor":["SIGARCH ACM Special Interest Group on Computer Architecture","IFIP WG 10.3 IFIP WG 10.3","IEEE CS"],"location":"Chicago Illinois","acronym":"PACT '22"},"container-title":["Proceedings of the International Conference on Parallel Architectures and Compilation Techniques"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3559009.3569651","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3559009.3569651","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T19:02:38Z","timestamp":1750186958000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3559009.3569651"}},"subtitle":["Seamless Integration of Deep Learning Backends with Automatic Placement"],"short-title":[],"issued":{"date-parts":[[2022,10,8]]},"references-count":75,"alternative-id":["10.1145\/3559009.3569651","10.1145\/3559009"],"URL":"https:\/\/doi.org\/10.1145\/3559009.3569651","relation":{},"subject":[],"published":{"date-parts":[[2022,10,8]]},"assertion":[{"value":"2023-01-27","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}