{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,28]],"date-time":"2025-06-28T07:10:10Z","timestamp":1751094610346,"version":"3.41.0"},"reference-count":21,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,5,25]],"date-time":"2025-05-25T00:00:00Z","timestamp":1748131200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,5,25]],"date-time":"2025-05-25T00:00:00Z","timestamp":1748131200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001321","name":"National Research Foundation","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001321","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001321","name":"National Research Foundation","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001321","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,5,25]]},"DOI":"10.1109\/iscas56072.2025.11043335","type":"proceedings-article","created":{"date-parts":[[2025,6,27]],"date-time":"2025-06-27T17:42:19Z","timestamp":1751046139000},"page":"1-5","source":"Crossref","is-referenced-by-count":0,"title":["AMC: Adaptive Mixed Compression for ML Models based on Block-Wise Sensitivity"],"prefix":"10.1109","author":[{"given":"Yeji","family":"Lee","sequence":"first","affiliation":[{"name":"Department of Electrical Engineering"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Eunbin","family":"Park","sequence":"additional","affiliation":[{"name":"POSTECH,Semiconductor Engineering,Pohang,Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Youngjoo","family":"Lee","sequence":"additional","affiliation":[{"name":"KAIST,School of Electrical Engineering,Daejeon,Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1016\/j.heliyon.2018.e00938"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00748"},{"article-title":"Efficient memory management for deep neural net inference","year":"2020","author":"Pisarchyk","key":"ref3"},{"article-title":"Efficient memory management for gpu-based deep learning systems","year":"2019","author":"Zhang","key":"ref4"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/DAC56929.2023.10247799"},{"article-title":"Neural network quantization for efficient inference: A survey","year":"2023","author":"Weng","key":"ref6"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2022.3182659"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/JETCAS.2019.2952137"},{"article-title":"A white paper on neural network quantization","year":"2021","author":"Nagel","key":"ref9"},{"article-title":"To prune, or not to prune: exploring the efficacy of pruning for model compression","year":"2017","author":"Zhu","key":"ref10"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00038"},{"key":"ref12","first-page":"18 518","article-title":"Hawq-v2: Hessian aware trace-weighted quantization of neural networks","volume-title":"Advances in Neural Information Processing Systems","volume":"33","author":"Dong","year":"2020"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20083-0_16"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10445737"},{"article-title":"Deep compression: Compressing deep neural networks with pruning, trained quantization and huffman coding","year":"2016","author":"Han","key":"ref15"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.91"},{"article-title":"Very deep convolutional networks for large-scale image recognition","year":"2015","author":"Simonyan","key":"ref17"},{"article-title":"Bert: Pre-training of deep bidirectional transformers for language understanding","year":"2019","author":"Devlin","key":"ref18"},{"year":"2024","key":"ref19","article-title":"Gaudi-v2: Ai training and inference processor"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2021.07.045"},{"article-title":"A comprehensive evaluation of quantized instruction-tuned large language models: An experimental analysis up to 405b","year":"2024","author":"Lee","key":"ref21"}],"event":{"name":"2025 IEEE International Symposium on Circuits and Systems (ISCAS)","start":{"date-parts":[[2025,5,25]]},"location":"London, United Kingdom","end":{"date-parts":[[2025,5,28]]}},"container-title":["2025 IEEE International Symposium on Circuits and Systems (ISCAS)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11043142\/11042930\/11043335.pdf?arnumber=11043335","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,28]],"date-time":"2025-06-28T06:43:30Z","timestamp":1751093010000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11043335\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,25]]},"references-count":21,"URL":"https:\/\/doi.org\/10.1109\/iscas56072.2025.11043335","relation":{},"subject":[],"published":{"date-parts":[[2025,5,25]]}}}