{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T14:24:43Z","timestamp":1784643883904,"version":"3.55.0"},"reference-count":39,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"12","license":[{"start":{"date-parts":[[2025,12,1]],"date-time":"2025-12-01T00:00:00Z","timestamp":1764547200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,12,1]],"date-time":"2025-12-01T00:00:00Z","timestamp":1764547200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,12,1]],"date-time":"2025-12-01T00:00:00Z","timestamp":1764547200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100006639","name":"NUDT Foundation","doi-asserted-by":"publisher","award":["ZK2023-16"],"award-info":[{"award-number":["ZK2023-16"]}],"id":[{"id":"10.13039\/100006639","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U23A20301"],"award-info":[{"award-number":["U23A20301"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100004761","name":"Natural Science Foundation of Hainan Province","doi-asserted-by":"publisher","award":["2024JJ6470"],"award-info":[{"award-number":["2024JJ6470"]}],"id":[{"id":"10.13039\/501100004761","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Key Laboratory of Advanced Microprocessor Chips and Systems"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Comput.-Aided Des. Integr. Circuits Syst."],"published-print":{"date-parts":[[2025,12]]},"DOI":"10.1109\/tcad.2025.3568347","type":"journal-article","created":{"date-parts":[[2025,5,8]],"date-time":"2025-05-08T13:38:49Z","timestamp":1746711529000},"page":"4752-4764","source":"Crossref","is-referenced-by-count":1,"title":["MAP-SIM: A DNN-Specific Mapping Optimization Framework for Shared-Memory CPU-Systolic Array Architectures"],"prefix":"10.1109","volume":"44","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9496-4748","authenticated-orcid":false,"given":"Yuhang","family":"Li","sequence":"first","affiliation":[{"name":"College of Computer Science and Technology, National University of Defense Technology, Changsha, Hunan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5875-3297","authenticated-orcid":false,"given":"Mei","family":"Wen","sequence":"additional","affiliation":[{"name":"College of Computer Science and Technology, National University of Defense Technology, Changsha, Hunan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6233-6800","authenticated-orcid":false,"given":"Junzhong","family":"Shen","sequence":"additional","affiliation":[{"name":"College of Computer Science and Technology, National University of Defense Technology, Changsha, Hunan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1552-8396","authenticated-orcid":false,"given":"Zhaoyun","family":"Chen","sequence":"additional","affiliation":[{"name":"College of Computer Science and Technology, National University of Defense Technology, Changsha, Hunan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5786-3171","authenticated-orcid":false,"given":"Yang","family":"Shi","sequence":"additional","affiliation":[{"name":"College of Computer Science and Technology, National University of Defense Technology, Changsha, Hunan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7030-1990","authenticated-orcid":false,"given":"Tianyu","family":"Wang","sequence":"additional","affiliation":[{"name":"College of Computer Science and Software Engineering, Shenzhen University, Shenzhen, Guangdong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2173-2847","authenticated-orcid":false,"given":"Zili","family":"Shao","sequence":"additional","affiliation":[{"name":"Department of Computer Science and Engineering, The Chinese University of Hong Kong, Hong Kong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2015.50"},{"key":"ref2","volume-title":"NVIDIA grace CPU","year":"2023"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.23919\/DATE58400.2024.10546765"},{"key":"ref4","first-page":"1","article-title":"Deep compression: Compressing deep neural networks with pruning, trained Quantization and Huffman coding","volume-title":"Proc. Int. Conf. Learn. Represent. (ICLR)","author":"Han"},{"key":"ref5","volume-title":"NVIDIA TensorRT: Programmer\u2019s Guide","year":"2023"},{"key":"ref6","first-page":"56","article-title":"Resource partitioning analysis in Intel\u2019s edge AI stack","volume-title":"Proc. ACM Symp. Edge Comput.","author":"Patel"},{"key":"ref7","volume-title":"Intel SSE Programming Reference","year":"2025"},{"key":"ref8","volume-title":"Intel AVX Programming Reference","year":"2025"},{"key":"ref9","volume-title":"ARM NEON Programming Guide","year":"2025"},{"key":"ref10","volume-title":"Nvidia Jetson AGX Orin Series","author":"Karumbunathan","year":"2022"},{"key":"ref11","volume-title":"Intelligent scheduling for simultaneous CPU-GPU applications","author":"Cheng","year":"2017"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA51647.2021.00016"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1145\/3613424.3614285"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1007\/978-981-96-1525-4_21"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ISPASS.2019.00042"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2014.24"},{"key":"ref17","doi-asserted-by":"crossref","first-page":"756","DOI":"10.1007\/s11227-014-1200-3","article-title":"Strategies for maximizing utilization on multi-CPU and multi-GPU heterogeneous architectures","volume":"70","author":"Navarro","year":"2014","journal-title":"J. Supercomput."},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ISPASS57527.2023.00028"},{"key":"ref19","article-title":"Accelerating deep learning on heterogenous architectures","author":"Nandakumar","year":"2022"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/TC.2022.3207137"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1145\/3665314.3670841"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/ISPASS57527.2023.00051"},{"key":"ref23","first-page":"1","article-title":"Scalable fast multipole methods on distributed heterogeneous architectures","volume-title":"Proc. Int. Conf. High Perform. Comput., Netw., Storage Anal.","author":"Hu"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-04580-6_9"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/DSD60849.2023.00015"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA56546.2023.10071095"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA52012.2021.00062"},{"key":"ref28","volume-title":"Integer Set Library: Manual, Version 0.10","author":"Verdoolaege","year":"2011"},{"issue":"1","key":"ref29","doi-asserted-by":"crossref","first-page":"20","DOI":"10.1145\/216585.216588","article-title":"Hitting the memory wall: Implications of the obvious","volume":"23","author":"Wulf","year":"1995","journal-title":"SIGARCH Comput. Archit. News"},{"key":"ref30","first-page":"1","article-title":"lmbench: Portable tools for performance analysis","volume-title":"Proc. USENIX Annu. Tech. Conf.","author":"McVoy"},{"key":"ref31","volume-title":"AMD uProf User Guide","year":"2023"},{"key":"ref32","volume-title":"perf: Linux Profiling With Performance Counters","year":"2025"},{"key":"ref33","first-page":"1","article-title":"SCALE-Sim: Systolic CNN accelerator simulator","volume-title":"Proc. IEEE ISPASS","author":"Samajdar"},{"key":"ref34","first-page":"1","article-title":"ImageNet classification with deep convolutional neural networks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Krizhevsky"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.91"},{"key":"ref36","article-title":"Very deep convolutional networks for large-scale image recognition","author":"Simonyan","year":"2014","journal-title":"arXiv:1409.1556"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.23919\/DATE58400.2024.10546765"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2020.2985963"}],"container-title":["IEEE Transactions on Computer-Aided Design of Integrated Circuits and Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/43\/11263962\/10993473.pdf?arnumber=10993473","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,24]],"date-time":"2025-11-24T19:00:29Z","timestamp":1764010829000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10993473\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12]]},"references-count":39,"journal-issue":{"issue":"12"},"URL":"https:\/\/doi.org\/10.1109\/tcad.2025.3568347","relation":{},"ISSN":["0278-0070","1937-4151"],"issn-type":[{"value":"0278-0070","type":"print"},{"value":"1937-4151","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,12]]}}}