{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,3]],"date-time":"2025-11-03T10:34:17Z","timestamp":1762166057249,"version":"build-2065373602"},"publisher-location":"Singapore","reference-count":28,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819510207","type":"print"},{"value":"9789819510214","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,11,4]],"date-time":"2025-11-04T00:00:00Z","timestamp":1762214400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,11,4]],"date-time":"2025-11-04T00:00:00Z","timestamp":1762214400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-981-95-1021-4_10","type":"book-chapter","created":{"date-parts":[[2025,11,3]],"date-time":"2025-11-03T10:29:13Z","timestamp":1762165753000},"page":"129-144","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["CeDMA: Enhancing Memory Efficiency of\u00a0Heterogeneous Accelerator Systems Through Central DMA Controlling"],"prefix":"10.1007","author":[{"given":"Ruoshi","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Long","family":"Zheng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yu","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhiyuan","family":"Shao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Amelie Chi","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaofei","family":"Liao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hai","family":"Jin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jingling","family":"Xue","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,11,4]]},"reference":[{"key":"10_CR1","doi-asserted-by":"crossref","unstructured":"Chen, X., Chen, Y., Cheng, F., Tan, H., He, B., Wong, W.: ReGraph: scaling graph processing on HBM-enabled FPGAs with heterogeneous pipelines. In: Proceedings of the International Symposium on Microarchitecture (MICRO), pp. 1342\u20131358 (2022)","DOI":"10.1109\/MICRO56248.2022.00092"},{"issue":"7","key":"10_CR2","first-page":"1878","volume":"32","author":"A Li","year":"2020","unstructured":"Li, A., Su, S.M.: Accelerating binarized neural networks via bit-tensor-cores in turing GPUs. IEEE Trans. Parallel Distrib. Syst. 32(7), 1878\u20131891 (2020)","journal-title":"IEEE Trans. Parallel Distrib. Syst."},{"key":"10_CR3","doi-asserted-by":"crossref","unstructured":"Liao, H., et al.: Ascend: a scalable and unified architecture for ubiquitous deep neural network computing: industry track paper. In: Proceedings of the International Symposium on High-Performance Computer Architecture (HPCA), pp. 789\u2013801 (2021)","DOI":"10.1109\/HPCA51647.2021.00071"},{"key":"10_CR4","doi-asserted-by":"crossref","unstructured":"Yan, M., et al.: Hygcn: a GCN accelerator with hybrid architecture. In: Proceedings of the International Symposium on High Performance Computer Architecture (HPCA), pp. 15\u201329 (2020)","DOI":"10.1109\/HPCA47549.2020.00012"},{"key":"10_CR5","doi-asserted-by":"crossref","unstructured":"Kumar, S., Shriraman, A., Vedula, N.: Fusion: design tradeoffs in coherent cache hierarchies for accelerators. In: Proceedings of the Annual International Symposium on Computer Architecture (ISCA), pp. 375\u2013386 (2015)","DOI":"10.1145\/2749469.2750421"},{"issue":"6","key":"10_CR6","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3128571","volume":"50","author":"X Shi","year":"2018","unstructured":"Shi, X., et al.: Graph processing on GPUs: a survey. ACM Comput. Surv. 50(6), 1\u201335 (2018)","journal-title":"ACM Comput. Surv."},{"key":"10_CR7","doi-asserted-by":"crossref","unstructured":"Nazeri, A., et al.: Exploration of TPU architectures for the optimized transformer in drainage crossing detection. In: Proceedings of the International Conference on Big Data (BigData), pp. 4178\u20134187 (2024)","DOI":"10.1109\/BigData62323.2024.10826077"},{"key":"10_CR8","doi-asserted-by":"crossref","unstructured":"Song, L., Chen, F., Zhuo, Y., Qian, X., Li, H., Chen, Y.: AccPar: tensor partitioning for heterogeneous deep learning accelerators. In: Proceedings of the International Symposium on High Performance Computer Architecture (HPCA), pp. 342\u2013355 (2020)","DOI":"10.1109\/HPCA47549.2020.00036"},{"issue":"2","key":"10_CR9","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3470567","volume":"15","author":"T Alonso","year":"2021","unstructured":"Alonso, T., et al.: Elastic-DF: scaling performance of DNN inference in FPGA clouds through automatic partitioning. ACM Trans. Reconfig. Technol. Syst. 15(2), 1\u201334 (2021)","journal-title":"ACM Trans. Reconfig. Technol. Syst."},{"key":"10_CR10","doi-asserted-by":"crossref","unstructured":"Liang, S., et al.: EnGN: a high-throughput and energy-efficient accelerator for large graph neural networks. IEEE Trans. Comput. 70(9), 1511\u20131525 (2021)","DOI":"10.1109\/TC.2020.3014632"},{"key":"10_CR11","doi-asserted-by":"crossref","unstructured":"Huang, Y., et al.: A heterogeneous PIM hardware-software co-design for energy-efficient graph processing. In: Proceedings of the International Parallel and Distributed Processing Symposium (IPDPS), pp. 684\u2013695 (2020)","DOI":"10.1109\/IPDPS47924.2020.00076"},{"key":"10_CR12","doi-asserted-by":"crossref","unstructured":"Huang, Y., et al.: Accelerating graph convolutional networks using crossbar-based processing-in-memory architectures. In: Proceedings of the International Symposium on High-Performance Computer Architecture (HPCA), pp. 1029\u20131042 (2022)","DOI":"10.1109\/HPCA53966.2022.00079"},{"issue":"4","key":"10_CR13","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3291058","volume":"15","author":"H Zhao","year":"2018","unstructured":"Zhao, H., et al.: Bandwidth and locality aware task-stealing for manycore architectures with bandwidth-asymmetric memory. ACM Trans. Archit. Code Optim. 15(4), 1\u201326 (2018)","journal-title":"ACM Trans. Archit. Code Optim."},{"issue":"1","key":"10_CR14","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1109\/TC.2022.3227228","volume":"72","author":"S Riedel","year":"2023","unstructured":"Riedel, S., Kurth, A., Benini, L., Rossi, D.: MemPool: a scalable manycore architecture with a low-latency shared L1 memory. IEEE Trans. Comput. 72(1), 1\u201314 (2023)","journal-title":"IEEE Trans. Comput."},{"key":"10_CR15","doi-asserted-by":"crossref","unstructured":"Wang, Z., Park, S., Park, C.S.: Shuhai: benchmarking high bandwidth memory on FPGAs. In: Proceedings of the Annual International Symposium on Field-Programmable Custom Computing Machines (FCCM), pp. 69\u201377 (2020)","DOI":"10.1109\/FCCM48280.2020.00024"},{"issue":"4","key":"10_CR16","first-page":"1","volume":"21","author":"S Roh","year":"2022","unstructured":"Roh, S., et al.: Cohmeleon: learning-based orchestration of accelerator coherence in heterogeneous SoCs. ACM Trans. Embed. Comput. Syst. 21(4), 1\u201325 (2022)","journal-title":"ACM Trans. Embed. Comput. Syst."},{"issue":"2","key":"10_CR17","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3309987","volume":"36","author":"S Bergman","year":"2019","unstructured":"Bergman, S., Brokhman, T., Cohen, T., Silberstein, M.: SPIN: seamless operating system integration of peer-to-peer DMA between SSDs and GPUs. ACM Trans. Comput. Syst. 36(2), 1\u201326 (2019)","journal-title":"ACM Trans. Comput. Syst."},{"issue":"1","key":"10_CR18","doi-asserted-by":"publisher","first-page":"154","DOI":"10.1109\/TC.2022.3214117","volume":"72","author":"F Restuccia","year":"2022","unstructured":"Restuccia, F., Pagani, M., Biondi, A., Marinoni, M., Buttazzo, G.: Bounding memory access times in multi-accelerator architectures on FPGA SoCs. IEEE Trans. Comput. 72(1), 154\u2013167 (2022)","journal-title":"IEEE Trans. Comput."},{"issue":"1","key":"10_CR19","doi-asserted-by":"publisher","first-page":"263","DOI":"10.1109\/TC.2023.3329930","volume":"73","author":"T Benz","year":"2024","unstructured":"Benz, T., et al.: A high-performance, energy-efficient modular DMA engine architecture. IEEE Trans. Comput. 73(1), 263\u2013277 (2024)","journal-title":"IEEE Trans. Comput."},{"key":"10_CR20","doi-asserted-by":"crossref","unstructured":"Jun, H., et al.: HBM (high bandwidth memory) DRAM technology and architecture. In: Proceedings of the International Memory Workshop (IMW), pp.\u00a01\u20134 (2017)","DOI":"10.1109\/IMW.2017.7939084"},{"key":"10_CR21","unstructured":"Zu, Y., et al.: Resiliency at scale: managing Google\u2019s TPUv4 machine learning supercomputer. In: Proceedings of the International Symposium on Networked Systems Design and Implementation (NSDI), pp. 761\u2013774 (2024)"},{"key":"10_CR22","unstructured":"JEDEC: High Bandwidth Memory (HBM) DRAM (2020). https:\/\/www.jedec.org\/standards-documents\/docs\/jesd235b"},{"key":"10_CR23","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the International Conference on Computer Vision and Pattern Recognition (CVPR), pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"10_CR24","doi-asserted-by":"crossref","unstructured":"Saad, Y.: Iterative Methods for Sparse Linear Systems. Society for Industrial and Applied Mathematics (2003)","DOI":"10.1137\/1.9780898718003"},{"key":"10_CR25","doi-asserted-by":"crossref","unstructured":"Goto, K., Geijn, R.A.V.D.: Anatomy of high-performance matrix multiplication. ACM Trans. Math. Softw. 34(3), 1\u201325 (2008)","DOI":"10.1145\/1356052.1356053"},{"key":"10_CR26","doi-asserted-by":"crossref","unstructured":"Boman, E.G., Devine, K.D., Rajamanickam, S.: Scalable matrix computations on large scale-free graphs using 2D graph partitioning. In: Proceedings of the International Conference on High Performance Computing, Networking, Storage and Analysis (SC), pp. 495\u2013504 (2013)","DOI":"10.1145\/2503210.2503293"},{"key":"10_CR27","doi-asserted-by":"crossref","unstructured":"Dai, G., Huang, T., Chi, Y., Xu, N., Wang, Y., Yang, H.: ForeGraph: exploring large-scale graph processing on multi-FPGA architecture. In: Proceedings of the International Symposium on Field-Programmable Gate Arrays (FPGA), pp. 217\u2013226 (2017)","DOI":"10.1145\/3020078.3021739"},{"key":"10_CR28","doi-asserted-by":"crossref","unstructured":"Shao, Z., Li, R., Hu, D., Liao, X., Jin, H.: Improving performance of graph processing on FPGA-DRAM platform by two-level vertex caching. In: Proceedings of the International Symposium on Field-Programmable Gate Arrays (FPGA), pp. 320\u2013329 (2019)","DOI":"10.1145\/3289602.3293900"}],"container-title":["Lecture Notes in Computer Science","Advanced Parallel Processing Technologies"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-95-1021-4_10","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,3]],"date-time":"2025-11-03T10:29:23Z","timestamp":1762165763000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-95-1021-4_10"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,4]]},"ISBN":["9789819510207","9789819510214"],"references-count":28,"URL":"https:\/\/doi.org\/10.1007\/978-981-95-1021-4_10","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,11,4]]},"assertion":[{"value":"4 November 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"APPT","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Symposium on Advanced Parallel Processing Technologies","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Athens","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Greece","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"13 July 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"16 July 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"16","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"appt2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.appt-conference.com\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}