{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,2]],"date-time":"2026-06-02T23:13:33Z","timestamp":1780442013545,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":27,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,7,10]],"date-time":"2022-07-10T00:00:00Z","timestamp":1657411200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"FMitF: Track I: Robust Enforcement of Customizable Resource Constraints in Heterogeneous Embedded Systems","award":["2124010"],"award-info":[{"award-number":["2124010"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,7,10]]},"DOI":"10.1145\/3489517.3530572","type":"proceedings-article","created":{"date-parts":[[2022,8,23]],"date-time":"2022-08-23T23:19:29Z","timestamp":1661296769000},"page":"1069-1074","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":39,"title":["AxoNN"],"prefix":"10.1145","author":[{"given":"Ismet","family":"Dagli","sequence":"first","affiliation":[{"name":"Colorado School of Mines"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Alexander","family":"Cieslewicz","sequence":"additional","affiliation":[{"name":"Colorado School of Mines"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jedidiah","family":"McClurg","sequence":"additional","affiliation":[{"name":"Colorado School of Mines"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mehmet E.","family":"Belviranli","sequence":"additional","affiliation":[{"name":"Colorado School of Mines"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2022,8,23]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"crossref","unstructured":"Rajkishore Barik Naila Farooqui Brian T. Lewis Chunling Hu and Tatiana Shpeisman. 2016. A black-box approach to energy-aware scheduling on integrated CPU-GPU systems. In CGO.","DOI":"10.1145\/2854038.2854052"},{"key":"e_1_3_2_1_2_1","volume-title":"TVM: An Automated End-to-End Optimizing Compiler for Deep Learning. In OSDI.","author":"Chen Tianqi","year":"2018","unstructured":"Tianqi Chen. 2018. TVM: An Automated End-to-End Optimizing Compiler for Deep Learning. In OSDI."},{"key":"e_1_3_2_1_3_1","volume-title":"Multi-accelerator Neural Network Inference in Diversely Heterogeneous Embedded Systems. In RSDHA Workshop.","author":"Dagli Ismet","unstructured":"Ismet Dagli and Mehmet E. Belviranli. 2021. Multi-accelerator Neural Network Inference in Diversely Heterogeneous Embedded Systems. In RSDHA Workshop."},{"key":"e_1_3_2_1_4_1","volume-title":"Co-scheduling on fused CPU-GPU architectures with shared last level caches","author":"Damschen Marvin","unstructured":"Marvin Damschen, Frank Mueller, and J\u00f6rg Henkel. 2018. Co-scheduling on fused CPU-GPU architectures with shared last level caches. In IEEE TCAD."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"crossref","unstructured":"Maria Angelica Davila Guzman. 2019. Cooperative CPU GPU and FPGA heterogeneous execution with EngineCL. In The Journal of Supercomputing.","DOI":"10.1007\/s11227-019-02768-y"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"crossref","unstructured":"Li Han Yiqin Gao Jing Liu Yves Robert and Frederic Vivien. 2020. Energy-Aware Strategies for Reliability-Oriented Real-Time Task Allocation on Heterogeneous Platforms. In ICPP.","DOI":"10.1145\/3404397.3404419"},{"key":"e_1_3_2_1_7_1","unstructured":"Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Deep Residual Learning for Image Recognition. In CVPR."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"crossref","unstructured":"Sitao Huang. 2019. Analysis and Modeling of Collaborative Execution Strategies for Heterogeneous CPU-FPGA Architectures. In ICPE.","DOI":"10.1145\/3297663.3310305"},{"key":"e_1_3_2_1_9_1","unstructured":"Yanping Huang. 2019. GPipe: Efficient training of giant neural networks using pipeline parallelism. In NIPS."},{"key":"e_1_3_2_1_10_1","volume-title":"Deep Learning Inference Parallelization on Heterogeneous Processors with TensorRT","author":"Jeong EunJin","unstructured":"EunJin Jeong, Jangryul Kim, Samnieng Tan, Jaeseong Lee, and Soonhoi Ha. 2021. Deep Learning Inference Parallelization on Heterogeneous Processors with TensorRT. In IEEE Embedded Systems Letters."},{"key":"e_1_3_2_1_11_1","volume-title":"Scheduling of Deep Learning Applications Onto Heterogeneous Processors in an Embedded Device","author":"Kang Duseok","unstructured":"Duseok Kang, Jinwoo Oh, Jongwoo Choi, Youngmin Yi, and Soonhoi Ha. 2020. Scheduling of Deep Learning Applications Onto Heterogeneous Processors in an Embedded Device. In IEEE Access."},{"key":"e_1_3_2_1_12_1","volume-title":"GAMMA: Automating the HW Mapping of DNN Models on Accelerators via Genetic Algorithm. In ICCAD.","author":"Kao Sheng-Chun","year":"2020","unstructured":"Sheng-Chun Kao and Tushar Krishna. 2020. GAMMA: Automating the HW Mapping of DNN Models on Accelerators via Genetic Algorithm. In ICCAD."},{"key":"e_1_3_2_1_13_1","volume-title":"Hinton","author":"Krizhevsky Alex","year":"2012","unstructured":"Alex Krizhevsky, Ilya Sutskever, and Geoffrey E. Hinton. 2012. ImageNet Classification with Deep Convolutional Neural Networks. In NIPS."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"crossref","unstructured":"Svetlana Minakova and Erqian Tang. 2020. Combining Task- and Data-Level Parallelism for High-Throughput CNN Inference on Embedded CPUs-GPUs MPSoCs. In Embedded Computer Systems: Architectures Modeling and Simulation.","DOI":"10.1007\/978-3-030-60939-9_2"},{"key":"e_1_3_2_1_15_1","volume-title":"Malony","author":"Haque Monil Mohammad Alaul","year":"2020","unstructured":"Mohammad Alaul Haque Monil, Mehmet E. Belviranli, Seyong Lee, Jeffrey S. Vetter, and Allen D. Malony. 2020. MEPHESTO: Modeling Energy-Performance in Heterogeneous SoCs and Their Trade-Offs. In PACT."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"crossref","unstructured":"Deepak Narayanan Aaron Harlap Amar Phanishayee Vivek Seshadri Nikhil R. Devanur Gregory R. Ganger Phillip B. Gibbons and Matei Zaharia. 2019. PipeDream: Generalized Pipeline Parallelism for DNN Training. In SOSP.","DOI":"10.1145\/3341301.3359646"},{"key":"e_1_3_2_1_17_1","unstructured":"Deepak Narayanan Keshav Santhanam Fiodar Kazhamiaka Amar Phanishayee and Matei Zaharia. 2020. Heterogeneity-Aware Cluster Scheduling Policies for Deep Learning Workloads. In OSDI."},{"key":"e_1_3_2_1_18_1","unstructured":"NVIDIA. 2021. TensorRT. https:\/\/developer.nvidia.com\/tensorrt"},{"key":"e_1_3_2_1_19_1","unstructured":"Jay H. Park. 2020. HetPipe: Enabling Large DNN Training on (Whimpy) Heterogeneous GPU Clusters through Integration of Pipelined Model Parallelism and Data Parallelism. In USENIX ATC."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"crossref","unstructured":"Elham Shamsa Anil Kanduri Amir M. Rahmani Pasi Liljeberg Axel Jantsch and Nikil Dutt. 2019. Goal-Driven Autonomy for Efficient On-chip Resource Management: Transforming Objectives to Goals. In DATE.","DOI":"10.23919\/DATE.2019.8715134"},{"key":"e_1_3_2_1_21_1","unstructured":"Karen Simonyan and Andrew Zisserman. 2015. Very Deep Convolutional Networks for Large-Scale Image Recognition. In ICLR."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"crossref","unstructured":"Christian Szegedy. 2015. Going deeper with convolutions. In CVPR.","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"crossref","unstructured":"Stavros Tzilis Pedro Trancoso and Ioannis Sourdis. 2019. Energy-Efficient Runtime Management of Heterogeneous Multicores Using Online Projection. TACO.","DOI":"10.1145\/3293446"},{"key":"e_1_3_2_1_24_1","unstructured":"Hsin-I Wu Da-Yi Guo Hsu-Hsun Chin and Ren-Song Tsay. 2020. A Pipeline-Based Scheduler for Optimizing Latency of Convolution Neural Network Inference over Heterogeneous Multicore Systems. In AICAS."},{"key":"e_1_3_2_1_25_1","unstructured":"Hongzhi Xu Renfa Li Chen Pan and Keqin Li. 2019. Minimizing energy consumption with reliability goal on heterogeneous embedded systems. In JPDC."},{"key":"e_1_3_2_1_26_1","volume-title":"Xipeng Shen, and Jeffrey Vetter.","author":"Xu Yuanchao","year":"2021","unstructured":"Yuanchao Xu, Mehmet Esat Belviranli, Xipeng Shen, and Jeffrey Vetter. 2021. PCCS: Processor-Centric Contention-Aware Slowdown Model for Heterogeneous System-on-Chips. In MICRO-54."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.sysarc.2017.01.002"}],"event":{"name":"DAC '22: 59th ACM\/IEEE Design Automation Conference","location":"San Francisco California","acronym":"DAC '22","sponsor":["SIGDA ACM Special Interest Group on Design Automation","IEEE CEDA"]},"container-title":["Proceedings of the 59th ACM\/IEEE Design Automation Conference"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3489517.3530572","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3489517.3530572","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T19:02:18Z","timestamp":1750186938000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3489517.3530572"}},"subtitle":["energy-aware execution of neural network inference on multi-accelerator heterogeneous SoCs"],"short-title":[],"issued":{"date-parts":[[2022,7,10]]},"references-count":27,"alternative-id":["10.1145\/3489517.3530572","10.1145\/3489517"],"URL":"https:\/\/doi.org\/10.1145\/3489517.3530572","relation":{},"subject":[],"published":{"date-parts":[[2022,7,10]]},"assertion":[{"value":"2022-08-23","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}