{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,19]],"date-time":"2026-06-19T16:35:38Z","timestamp":1781886938321,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":72,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,2,20]],"date-time":"2024-02-20T00:00:00Z","timestamp":1708387200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/100000001","name":"NSF (National Science Foundation)","doi-asserted-by":"publisher","award":["CCF-2124010"],"award-info":[{"award-number":["CCF-2124010"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/100000125","name":"National Institute for Occupational Safety and Health","doi-asserted-by":"publisher","award":["75D30119C05413"],"award-info":[{"award-number":["75D30119C05413"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/100000125","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,3,2]]},"DOI":"10.1145\/3627535.3638502","type":"proceedings-article","created":{"date-parts":[[2024,2,20]],"date-time":"2024-02-20T14:22:41Z","timestamp":1708438961000},"page":"243-256","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":16,"title":["Shared Memory-contention-aware Concurrent DNN Execution for Diversely Heterogeneous System-on-Chips"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-1460-6906","authenticated-orcid":false,"given":"Ismet","family":"Dagli","sequence":"first","affiliation":[{"name":"Computer Science Department, Colorado School of Mines, Golden, United States of America"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9434-9833","authenticated-orcid":false,"given":"Mehmet E.","family":"Belviranli","sequence":"additional","affiliation":[{"name":"Computer Science Department, Colorado School of Mines, Golden, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,2,20]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"NVIDIA Deep Learning Accelerator. 2023. http:\/\/nvdla.org\/ (accessed on 08\/04\/2023)."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2016.7783725"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1186\/s40537-021-00444-8"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/2925426.2926271"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/DAC18074.2021.9586280"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/PACT52795.2021.00019"},{"key":"e_1_3_2_1_7_1","volume-title":"13th USENIX Symposium on Operating Systems Design and Implementation (OSDI 18)","author":"Chen Tianqi","year":"2018","unstructured":"Tianqi Chen, Thierry Moreau, Ziheng Jiang, Lianmin Zheng, Eddie Yan, Haichen Shen, Meghan Cowan, Leyuan Wang, Yuwei Hu, Luis Ceze, et al. 2018. {TVM}: An automated {End-to-End} optimizing compiler for deep learning. In 13th USENIX Symposium on Operating Systems Design and Implementation (OSDI 18). 578--594."},{"key":"e_1_3_2_1_8_1","unstructured":"NVIDIA Nsight Compute. 2022. https:\/\/docs.nvidia.com\/nsight-compute\/NsightCompute\/index.html (accessed on 08\/04\/2023)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.350"},{"key":"e_1_3_2_1_10_1","volume-title":"Multi-accelerator Neural Network Inference in Diversely Heterogeneous Embedded Systems. In 2021 IEEE\/ACM Redefining Scalability for Diversely Heterogeneous Architectures Workshop (RSDHA). IEEE, 1--7.","author":"Dagli Ismet","year":"2021","unstructured":"Ismet Dagli and Mehmet E Belviranli. 2021. Multi-accelerator Neural Network Inference in Diversely Heterogeneous Embedded Systems. In 2021 IEEE\/ACM Redefining Scalability for Diversely Heterogeneous Architectures Workshop (RSDHA). IEEE, 1--7."},{"key":"e_1_3_2_1_11_1","volume-title":"Proceedings of the 59th ACM\/IEEE Design Automation Conference (DAC).","author":"Dagli Ismet","unstructured":"Ismet Dagli, Alexander Cieslewicz, Jedidiah McClurg, and Mehmet E. Belviranli. 2022. AxoNN: Energy-Aware Execution of Neural Network Inference on Multi-Accelerator Heterogeneous SoCs. In Proceedings of the 59th ACM\/IEEE Design Automation Conference (DAC)."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3589010.3594889"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-78800-3_24"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/2749469.2750389"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1002\/rob.21918"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.robot.2017.04.004"},{"key":"e_1_3_2_1_17_1","unstructured":"Gurobi Optimization LLC. 2023. Gurobi Optimizer Reference Manual. https:\/\/www.gurobi.com"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3445814.3446762"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2019.00047"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3489517.3530445"},{"key":"e_1_3_2_1_22_1","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition(CVPR).","author":"Huang Gao","unstructured":"Gao Huang, Zhuang Liu, Laurens van der Maaten, and Kilian Q Weinberger. 2017. Densely Connected Convolutional Networks. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition(CVPR)."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCCN52240.2021.9522156"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/LES.2021.3087707"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/3498361.3538932"},{"key":"e_1_3_2_1_26_1","volume-title":"Caffe: Convolutional Architecture for Fast Feature Embedding. arXiv preprint arXiv:1408.5093","author":"Jia Yangqing","year":"2014","unstructured":"Yangqing Jia, Evan Shelhamer, Jeff Donahue, Sergey Karayev, Jonathan Long, Ross Girshick, Sergio Guadarrama, and Trevor Darrell. 2014. Caffe: Convolutional Architecture for Fast Feature Embedding. arXiv preprint arXiv:1408.5093 (2014)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.5555\/3488766.3488792"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2020.2977496"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3400302.3415639"},{"key":"e_1_3_2_1_30_1","volume-title":"MAGMA: An Optimization Framework for Mapping Multiple DNNs on Multiple Accelerator Cores. In 2022 IEEE International Symposium on High-Performance Computer Architecture (HPCA). IEEE, 814--830","author":"Kao Sheng-Chun","year":"2022","unstructured":"Sheng-Chun Kao and Tushar Krishna. 2022. MAGMA: An Optimization Framework for Mapping Multiple DNNs on Multiple Accelerator Cores. In 2022 IEEE International Symposium on High-Performance Computer Architecture (HPCA). IEEE, 814--830."},{"key":"e_1_3_2_1_31_1","unstructured":"Sheng-Chun Kao Suvinay Subramanian Gaurav Agrawal and Tushar Krishna. 2023. FLAT: An Optimized Dataflow for Mitigating Attention Performance Bottlenecks. published in arxiv will appear in Proceedings of the International Conference on Architectural Support for Programming Languages and Operating Systems (ASPLOS) (2023)."},{"key":"e_1_3_2_1_32_1","volume-title":"OmniBoost: Boosting Throughput of Heterogeneous Embedded Devices under Multi-DNN Workload. arXiv preprint arXiv:2307.03290","author":"Karatzas Andreas","year":"2023","unstructured":"Andreas Karatzas and Iraklis Anagnostopoulos. 2023. OmniBoost: Boosting Throughput of Heterogeneous Embedded Devices under Multi-DNN Workload. arXiv preprint arXiv:2307.03290 (2023)."},{"key":"e_1_3_2_1_33_1","volume-title":"Adaptive Execution for Multi-Tenant Deep Neural Networks. In 2023 IEEE International Symposium on High-Performance Computer Architecture (HPCA). IEEE, 828--841","author":"Kim Seah","year":"2023","unstructured":"Seah Kim, Hasan Genc, Vadim Vadimovich Nikiforov, Krste Asanovi\u0107, Borivoje Nikoli\u0107, and Yakun Sophia Shao. 2023. MoCA: Memory-Centric, Adaptive Execution for Multi-Tenant Deep Neural Networks. In 2023 IEEE International Symposium on High-Performance Computer Architecture (HPCA). IEEE, 828--841."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISPASS55109.2022.00023"},{"key":"e_1_3_2_1_35_1","volume-title":"Automatic Domain-Specific SoC Design for Autonomous Unmanned Aerial Vehicles. In 2022 55th IEEE\/ACM International Symposium on Microarchitecture (MICRO). IEEE, 300--317","author":"Krishnan Srivatsan","year":"2022","unstructured":"Srivatsan Krishnan, Zishen Wan, Kshitij Bhardwaj, Paul Whatmough, Aleksandra Faust, Sabrina Neuman, Gu-Yeon Wei, David Brooks, and Vijay Janapa Reddi. 2022. Automatic Domain-Specific SoC Design for Autonomous Unmanned Aerial Vehicles. In 2022 55th IEEE\/ACM International Symposium on Microarchitecture (MICRO). IEEE, 300--317."},{"key":"e_1_3_2_1_36_1","volume-title":"Imagenet classification with deep convolutional neural networks. Advances in neural information processing systems(NeurIPS) 25","author":"Krizhevsky Alex","year":"2012","unstructured":"Alex Krizhevsky, Ilya Sutskever, and Geoffrey E Hinton. 2012. Imagenet classification with deep convolutional neural networks. Advances in neural information processing systems(NeurIPS) 25 (2012)."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA51647.2021.00016"},{"key":"e_1_3_2_1_38_1","volume-title":"Proceedings, Part V 13","author":"Lin Tsung-Yi","year":"2014","unstructured":"Tsung-Yi Lin, Michael Maire, Serge Belongie, James Hays, Pietro Perona, Deva Ramanan, Piotr Doll\u00e1r, and C Lawrence Zitnick. 2014. Microsoft coco: Common objects in context. In Computer Vision-ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6--12, 2014, Proceedings, Part V 13. Springer, 740--755."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/MDAT.2022.3202997"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2018.00035"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3410463.3414671"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3341301.3359646"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.5555\/3488766.3488793"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/3453483.3454083"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/BEC49624.2020.9276943"},{"key":"e_1_3_2_1_46_1","unstructured":"NVIDIA. 2023. AI-Powered Autonomous Machines at Scale | NVIDIA Jetson AGX Xavier. https:\/\/www.nvidia.com\/en-us\/autonomous-machines\/embedded-systems\/jetson-agx-xavier\/. (accessed on 08\/04\/2023)."},{"key":"e_1_3_2_1_47_1","unstructured":"NVIDIA. 2023. Next-level AI performance for next-gen robotics | NVIDIA Jetson Orin AGX. https:\/\/www.nvidia.com\/en-us\/autonomous-machines\/embedded-systems\/jetson-orin\/. (accessed on 08\/04\/2023)."},{"key":"e_1_3_2_1_48_1","unstructured":"NVIDIA. 2023. TensorRT. https:\/\/developer.nvidia.com\/tensorrt (accessed on 08\/04\/2023)."},{"key":"e_1_3_2_1_49_1","unstructured":"NVIDIA. 2023. TensorRT IProfiler. https:\/\/docs.nvidia.com\/deeplearning\/tensorrt\/api\/c_api\/classnvinfer1_1_1_i_profiler.html (accessed on 08\/04\/2023)."},{"key":"e_1_3_2_1_50_1","volume-title":"Proceedings of the 2020 USENIX Conference on Usenix Annual Technical Conference. 307--321","author":"Park Jay H","year":"2020","unstructured":"Jay H Park, Gyeongchan Yun, Chang M Yi, Nguyen T Nguyen, Seungmin Lee, Jaesik Choi, Sam H Noh, and Young-ri Choi. 2020. Hetpipe: Enabling large DNN training on (whimpy) heterogeneous GPU clusters through integration of pipelined model parallelism and data parallelism. In Proceedings of the 2020 USENIX Conference on Usenix Annual Technical Conference. 307--321."},{"key":"e_1_3_2_1_51_1","volume-title":"Mehmet Esat Belviranli, and Didem Unat","author":"Qararyah Fareed","year":"2021","unstructured":"Fareed Qararyah, Mohamed Wahib, Do\u011fa Dikbay\u0131r, Mehmet Esat Belviranli, and Didem Unat. 2021. A computational-graph partitioning method for training memory-constrained DNNs. Parallel computing 104 (2021), 102792."},{"key":"e_1_3_2_1_52_1","unstructured":"Qualcomm. 2023. Neural Processing SDK for AI. https:\/\/developer.qualcomm.com\/software\/qualcomm-neural-processing-sdk (accessed on 08\/04\/2023)."},{"key":"e_1_3_2_1_53_1","unstructured":"Qualcomm. 2023. Snapdragon 865 Mobile Hardware Development Kit. https:\/\/stage.developer.qualcomm.com\/hardware\/snapdragon-865-hdk. (accessed on 08\/04\/2023)."},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/JSEN.2020.3041615"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/COASE.2018.8560344"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"crossref","unstructured":"Olga Russakovsky Jia Deng Hao Su Jonathan Krause Sanjeev Satheesh Sean Ma Zhiheng Huang Andrej Karpathy Aditya Khosla Michael Bernstein et al. 2015. Imagenet large scale visual recognition challenge. International journal of computer vision 115 (2015) 211--252.","DOI":"10.1007\/s11263-015-0816-y"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-21690-4_27"},{"key":"e_1_3_2_1_58_1","volume-title":"International Conference on Learning Representations(ICLR).","author":"Simonyan Karen","year":"2014","unstructured":"Karen Simonyan and Andrew Zisserman. 2014. Very deep convolutional networks for large-scale image recognition. In International Conference on Learning Representations(ICLR)."},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.5555\/3298023.3298188"},{"key":"e_1_3_2_1_60_1","volume-title":"Alemi","author":"Szegedy Christian","year":"2017","unstructured":"Christian Szegedy, Sergey Ioffe, Vincent Vanhoucke, and Alexander A. Alemi. 2017. Inception-v4, Inception-ResNet and the Impact of Residual Connections on Learning. In AAAI."},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"e_1_3_2_1_62_1","unstructured":"Tesla. 2023. Tesla Autopilot AI. https:\/\/www.tesla.com\/AI"},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1145\/3293446"},{"key":"e_1_3_2_1_64_1","doi-asserted-by":"publisher","DOI":"10.1145\/3508352.3561111"},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.1007\/s41095-020-0162-z"},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1109\/AICAS48895.2020.9073977"},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"publisher","DOI":"10.1145\/3466752.3480101"},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503222.3507767"},{"key":"e_1_3_2_1_69_1","doi-asserted-by":"publisher","DOI":"10.1145\/3489517.3530509"},{"key":"e_1_3_2_1_70_1","doi-asserted-by":"publisher","DOI":"10.1109\/IV47402.2020.9304602"},{"key":"e_1_3_2_1_71_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2017.124"},{"key":"e_1_3_2_1_72_1","doi-asserted-by":"publisher","DOI":"10.1145\/3577949.3577965"}],"event":{"name":"PPoPP '24: 29th ACM SIGPLAN Annual Symposium on Principles and Practice of Parallel Programming","location":"Edinburgh United Kingdom","acronym":"PPoPP '24","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing","SIGPLAN ACM Special Interest Group on Programming Languages"]},"container-title":["Proceedings of the 29th ACM SIGPLAN Annual Symposium on Principles and Practice of Parallel Programming"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3627535.3638502","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3627535.3638502","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T22:50:04Z","timestamp":1750287004000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3627535.3638502"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,2,20]]},"references-count":72,"alternative-id":["10.1145\/3627535.3638502","10.1145\/3627535"],"URL":"https:\/\/doi.org\/10.1145\/3627535.3638502","relation":{},"subject":[],"published":{"date-parts":[[2024,2,20]]},"assertion":[{"value":"2024-02-20","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}