{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T16:02:31Z","timestamp":1782921751413,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":53,"publisher":"ACM","license":[{"start":{"date-parts":[[2019,9,30]],"date-time":"2019-09-30T00:00:00Z","timestamp":1569801600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2019,9,30]]},"DOI":"10.1145\/3357526.3357532","type":"proceedings-article","created":{"date-parts":[[2019,11,6]],"date-time":"2019-11-06T14:25:56Z","timestamp":1573050356000},"page":"506-517","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":25,"title":["Co-ML: a case for\n            <u>Co<\/u>\n            llaborative\n            <u>ML<\/u>\n            acceleration using near-data processing"],"prefix":"10.1145","author":[{"given":"Shaizeen","family":"Aga","sequence":"first","affiliation":[{"name":"Advanced Micro Devices, Inc."}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Nuwan","family":"Jayasena","sequence":"additional","affiliation":[{"name":"Advanced Micro Devices, Inc."}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mike","family":"Ignatowski","sequence":"additional","affiliation":[{"name":"Advanced Micro Devices, Inc."}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2019,9,30]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"2014. Hybrid memory cube specification 2.0.  2014. Hybrid memory cube specification 2.0."},{"key":"e_1_3_2_1_2_1","unstructured":"2019. AMD Ryzen\u2122 Threadripper 2950X Processor. \"https:\/\/www.amd.com\/en\/products\/cpu\/amd-ryzen-threadripper-2950x\".  2019. AMD Ryzen\u2122 Threadripper 2950X Processor. \"https:\/\/www.amd.com\/en\/products\/cpu\/amd-ryzen-threadripper-2950x\"."},{"key":"e_1_3_2_1_3_1","unstructured":"2019. AMD's Machine Intelligence Library. \"https:\/\/github.com\/ROCmSoftwarePlatform\/MIOpen\".  2019. AMD's Machine Intelligence Library. \"https:\/\/github.com\/ROCmSoftwarePlatform\/MIOpen\"."},{"key":"e_1_3_2_1_4_1","unstructured":"2019. The CIFAR-10 dataset. \"http:\/\/www.cs.toronto.edu\/~kriz\/cifar.html\".  2019. The CIFAR-10 dataset. \"http:\/\/www.cs.toronto.edu\/~kriz\/cifar.html\"."},{"key":"e_1_3_2_1_5_1","unstructured":"2019. High Bandwidth Memory DRAM (HBM1 HBM2). \"https:\/\/www.jedec.org\/standards-documents\/docs\/jesd235a\".  2019. High Bandwidth Memory DRAM (HBM1 HBM2). \"https:\/\/www.jedec.org\/standards-documents\/docs\/jesd235a\"."},{"key":"e_1_3_2_1_6_1","unstructured":"2019. HIP: C++ Heterogeneous-Compute Interface for Portability. \"https:\/\/gpuopen.com\/compute-product\/hip-convert-cuda-to-portable-c-code\/\".  2019. HIP: C++ Heterogeneous-Compute Interface for Portability. \"https:\/\/gpuopen.com\/compute-product\/hip-convert-cuda-to-portable-c-code\/\"."},{"key":"e_1_3_2_1_7_1","unstructured":"2019. NVIDIA TENSOR CORES. \"https:\/\/www.nvidia.com\/en-us\/data-center\/tensorcore\/\".  2019. NVIDIA TENSOR CORES. \"https:\/\/www.nvidia.com\/en-us\/data-center\/tensorcore\/\"."},{"key":"e_1_3_2_1_8_1","unstructured":"2019. Radeon Compute Profiler (RCP). \"https:\/\/gpuopen.com\/compute-product\/radeon-compute-profiler-rcp\/\".  2019. Radeon Compute Profiler (RCP). \"https:\/\/gpuopen.com\/compute-product\/radeon-compute-profiler-rcp\/\"."},{"key":"e_1_3_2_1_9_1","unstructured":"2019. Radeon\u2122 Vega Frontier Edition Graphics. \"https:\/\/www.amd.com\/en\/graphics\/workstations-radeon-pro-vega-frontier-edition\".  2019. Radeon\u2122 Vega Frontier Edition Graphics. \"https:\/\/www.amd.com\/en\/graphics\/workstations-radeon-pro-vega-frontier-edition\"."},{"key":"e_1_3_2_1_10_1","unstructured":"2019. ROCm a New Era in Open GPU Computing. \"https:\/\/rocm.github.io\/\".  2019. ROCm a New Era in Open GPU Computing. \"https:\/\/rocm.github.io\/\"."},{"key":"e_1_3_2_1_11_1","unstructured":"2019. Samsung Electronics Introduces New High Bandwidth Memory Technology Tailored to Data Centers Graphic Applications and AI. \"https:\/\/www.samsung.com\/semiconductor\/insights\/tech-leadership\/samsung-electronics-introduces-new-high-bandwidth-memory-technology-tailored-to-data-centers-graphic-applications-and-ai\/\".  2019. Samsung Electronics Introduces New High Bandwidth Memory Technology Tailored to Data Centers Graphic Applications and AI. \"https:\/\/www.samsung.com\/semiconductor\/insights\/tech-leadership\/samsung-electronics-introduces-new-high-bandwidth-memory-technology-tailored-to-data-centers-graphic-applications-and-ai\/\"."},{"key":"e_1_3_2_1_12_1","volume-title":"Proceedings of the 12th USENIX Conference on Operating Systems Design and Implementation (OSDI'16)","author":"Abadi Mart\u00edn","year":"2016"},{"key":"e_1_3_2_1_13_1","volume-title":"Proceedings of the 33rd International Conference on International Conference on Machine Learning -","volume":"48","author":"Amodei Dario","year":"2016"},{"key":"e_1_3_2_1_14_1","unstructured":"Soheil Bahrampour Naveen Ramakrishnan Lukas Schott and Mohak Shah. 2015. Comparative Study of Caffe Neon Theano and Torch for Deep Learning. CoRR (2015).  Soheil Bahrampour Naveen Ramakrishnan Lukas Schott and Mohak Shah. 2015. Comparative Study of Caffe Neon Theano and Torch for Deep Learning. CoRR (2015)."},{"key":"e_1_3_2_1_15_1","volume-title":"Proceedings of the 32Nd International Conference on Neural Information Processing Systems (NIPS'18)","author":"Bjorck Johan"},{"key":"e_1_3_2_1_16_1","volume-title":"High Performance Convolutional Neural Networks for Document Processing. In Tenth International Workshop on Frontiers in Handwriting Recognition.","author":"Chellapilla Kumar","year":"2006"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"crossref","volume-title":"Eyeriss: A Spatial Architecture for Energy-Efficient Dataflow for Convolutional Neural Networks. In 2016 ACM\/IEEE 43rd Annual International Symposium on Computer Architecture (ISCA).","author":"Chen Y.","DOI":"10.1145\/3007787.3001177"},{"key":"e_1_3_2_1_18_1","volume-title":"PRIME: A Novel Processing-in-Memory Architecture for Neural Network Computation in ReRAM-Based Main Memory. In 2016 ACM\/IEEE 43rd Annual International Symposium on Computer Architecture (ISCA).","author":"Chi P."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3195970.3196029"},{"key":"e_1_3_2_1_20_1","volume-title":"2015 ACM\/IEEE 42nd Annual International Symposium on Computer Architecture (ISCA).","author":"Du Z."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA.2018.00040"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA.2018.00012"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3037697.3037702"},{"key":"e_1_3_2_1_24_1","volume":"201","author":"Han Song","journal-title":"William J. Dally."},{"key":"e_1_3_2_1_25_1","volume-title":"Proceedings of the 28th International Conference on Neural Information Processing Systems -","volume":"1","author":"Han Song"},{"key":"e_1_3_2_1_26_1","volume":"201","author":"He K.","journal-title":"J. Sun."},{"key":"e_1_3_2_1_27_1","unstructured":"Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Identity Mappings in Deep Residual Networks. CoRR (2016). arXiv:1603.05027  Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Identity Mappings in Deep Residual Networks. CoRR (2016). arXiv:1603.05027"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"crossref","unstructured":"Sepp Hochreiter and J\u00fcrgen Schmidhuber. 1997. Long Short-Term Memory. Neural Comput. (1997).  Sepp Hochreiter and J\u00fcrgen Schmidhuber. 1997. Long Short-Term Memory. Neural Comput. (1997).","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"e_1_3_2_1_29_1","unstructured":"Andrew G. Howard Menglong Zhu Bo Chen Dmitry Kalenichenko Weijun Wang Tobias Weyand Marco Andreetto and Hartwig Adam. 2017. MobileNets: Efficient Convolutional Neural Networks for Mobile Vision Applications. CoRR (2017). arXiv:1704.04861  Andrew G. Howard Menglong Zhu Bo Chen Dmitry Kalenichenko Weijun Wang Tobias Weyand Marco Andreetto and Hartwig Adam. 2017. MobileNets: Efficient Convolutional Neural Networks for Mobile Vision Applications. CoRR (2017). arXiv:1704.04861"},{"key":"e_1_3_2_1_30_1","volume-title":"Densely Connected Convolutional Networks. In 2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","author":"Huang G."},{"key":"e_1_3_2_1_31_1","volume-title":"Proceedings of the 32Nd International Conference on International Conference on Machine Learning -","volume":"37","author":"Ioffe Sergey","year":"2015"},{"key":"e_1_3_2_1_32_1","volume-title":"2017 IEEE\/ACM International Symposium on Low Power Electronics and Design (ISLPED).","author":"Jiang L."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3079856.3080246"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3007787.3001178"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/3123939.3123977"},{"key":"e_1_3_2_1_36_1","unstructured":"Min Lin Qiang Chen and Shuicheng Yan. 2014. Network In Network. CoRR (2014).  Min Lin Qiang Chen and Shuicheng Yan. 2014. Network In Network. CoRR (2014)."},{"key":"e_1_3_2_1_37_1","volume":"201","author":"Liu J.","journal-title":"J. Zhao."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"crossref","unstructured":"R. Nair S. F. Antao C. Bertolli P. Bose J. R. Brunheroto T. Chen C.$$ Cher C. H. A. Costa J. Doi C. Evangelinos B. M. Fleischer T. W. Fox D. S. Gallo L. Grinberg J. A. Gunnels A. C. Jacob P. Jacob H. M. Jacobson T. Karkhanis C. Kim J. H. Moreno J. K. O'Brien M. Ohmacht Y. Park D. A. Prener B. S. Rosenburg K. D. Ryu O. Sallenave M. J. Serrano P. D. M. Siegl K. Sugavanam and Z. Sura. 2015. Active Memory Cube: A processing-in-memory architecture for exascale systems. IBM Journal of Research and Development (2015).  R. Nair S. F. Antao C. Bertolli P. Bose J. R. Brunheroto T. Chen C.$$ Cher C. H. A. Costa J. Doi C. Evangelinos B. M. Fleischer T. W. Fox D. S. Gallo L. Grinberg J. A. Gunnels A. C. Jacob P. Jacob H. M. Jacobson T. Karkhanis C. Kim J. H. Moreno J. K. O'Brien M. Ohmacht Y. Park D. A. Prener B. S. Rosenburg K. D. Ryu O. Sallenave M. J. Serrano P. D. M. Siegl K. Sugavanam and Z. Sura. 2015. Active Memory Cube: A processing-in-memory architecture for exascale systems. IBM Journal of Research and Development (2015).","DOI":"10.1147\/JRD.2015.2409732"},{"key":"e_1_3_2_1_39_1","volume":"201","author":"Parashar Angshuman","journal-title":"William J. Dally."},{"key":"e_1_3_2_1_40_1","volume-title":"Int. J. Comput. Vision","author":"Russakovsky Olga","year":"2015"},{"key":"e_1_3_2_1_41_1","unstructured":"Hojjat Salehinejad Julianne Baarbe Sharan Sankar Joseph Barfett Errol Colak and Shahrokh Valaee. 2018. Recent Advances in Recurrent Neural Networks. CoRR (2018).  Hojjat Salehinejad Julianne Baarbe Sharan Sankar Joseph Barfett Errol Colak and Shahrokh Valaee. 2018. Recent Advances in Recurrent Neural Networks. CoRR (2018)."},{"key":"e_1_3_2_1_42_1","volume-title":"ISAAC: A Convolutional Neural Network Accelerator with In-Situ Analog Arithmetic in Crossbars. In 2016 ACM\/IEEE 43rd Annual International Symposium on Computer Architecture (ISCA).","author":"Shafiee A."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"crossref","unstructured":"H. Shin D. Kim E. Park S. Park Y. Park and S. Yoo. 2018. McDRAM: Low Latency and Energy-Efficient Matrix Computations in DRAM. IEEE Transactions on Computer-Aided Design of Integrated Circuits and Systems (2018).  H. Shin D. Kim E. Park S. Park Y. Park and S. Yoo. 2018. McDRAM: Low Latency and Energy-Efficient Matrix Computations in DRAM. IEEE Transactions on Computer-Aided Design of Integrated Circuits and Systems (2018).","DOI":"10.1109\/TCAD.2018.2857044"},{"key":"e_1_3_2_1_44_1","unstructured":"Laurent Sifre and St\u00e9phane Mallat. 2014. Rigid-Motion Scattering for Texture Classification. CoRR (2014).  Laurent Sifre and St\u00e9phane Mallat. 2014. Rigid-Motion Scattering for Texture Classification. CoRR (2014)."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3243176.3243184"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"crossref","volume-title":"NID: Processing Binary Convolutional Neural Network in Commodity DRAM. In 2018 IEEE\/ACM International Conference on Computer-Aided Design (ICCAD).","author":"Sim J.","DOI":"10.1145\/3240765.3240831"},{"key":"e_1_3_2_1_47_1","volume-title":"PipeLayer: A Pipelined ReRAM-Based Accelerator for Deep Learning. In 2017 IEEE International Symposium on High Performance Computer Architecture (HPCA).","author":"Song L."},{"key":"e_1_3_2_1_48_1","volume-title":"Proceedings of the 27th International Conference on Neural Information Processing Systems -","volume":"2","author":"Sutskever Ilya"},{"key":"e_1_3_2_1_49_1","volume-title":"Rethinking the Inception Architecture for Computer Vision. In 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","author":"Szegedy C."},{"key":"e_1_3_2_1_50_1","volume-title":"Proceedings of the 2018 USENIX Conference on Usenix Annual Technical Conference (USENIX ATC '18).","author":"Zhang Minjia","year":"2018"},{"key":"e_1_3_2_1_51_1","volume-title":"2016 49th Annual IEEE\/ACM International Symposium on Microarchitecture (MICRO).","author":"Zhang S."},{"key":"e_1_3_2_1_52_1","volume-title":"Towards Memory Friendly Long-Short Term Memory Networks (LSTMs) on Mobile GPUs. In 2018 51st Annual IEEE\/ACM International Symposium on Microarchitecture (MICRO).","author":"Zhang X."},{"key":"e_1_3_2_1_53_1","volume-title":"TBD: Benchmarking and Analyzing Deep Neural Network Training. CoRR","author":"Zhu Hongyu","year":"2018"}],"event":{"name":"MEMSYS '19: The International Symposium on Memory Systems","location":"Washington District of Columbia USA","acronym":"MEMSYS '19"},"container-title":["Proceedings of the International Symposium on Memory Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3357526.3357532","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3357526.3357532","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T23:23:22Z","timestamp":1750202602000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3357526.3357532"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,9,30]]},"references-count":53,"alternative-id":["10.1145\/3357526.3357532","10.1145\/3357526"],"URL":"https:\/\/doi.org\/10.1145\/3357526.3357532","relation":{},"subject":[],"published":{"date-parts":[[2019,9,30]]},"assertion":[{"value":"2019-09-30","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}