{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T23:36:26Z","timestamp":1783035386589,"version":"3.54.6"},"reference-count":35,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2021,3,1]],"date-time":"2021-03-01T00:00:00Z","timestamp":1614556800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,3,1]],"date-time":"2021-03-01T00:00:00Z","timestamp":1614556800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100012166","name":"National Key Research and Development of China","doi-asserted-by":"crossref","award":["2018YFB0204301"],"award-info":[{"award-number":["2018YFB0204301"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"crossref"}]},{"name":"The Science and Technology Planning Project of Hunan Province","award":["2019RS2027"],"award-info":[{"award-number":["2019RS2027"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["CCF Trans. HPC"],"published-print":{"date-parts":[[2021,3]]},"DOI":"10.1007\/s42514-020-00057-2","type":"journal-article","created":{"date-parts":[[2021,3,31]],"date-time":"2021-03-31T15:02:26Z","timestamp":1617202946000},"page":"114-125","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":40,"title":["Advancing DSP into HPC, AI, and beyond: challenges, mechanisms, and future directions"],"prefix":"10.1007","volume":"3","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9556-5535","authenticated-orcid":false,"given":"Yaohua","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chen","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chang","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sheng","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuanwu","family":"Lei","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jian","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yang","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yang","family":"Guo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2021,3,31]]},"reference":[{"key":"57_CR1","doi-asserted-by":"crossref","unstructured":"Anoushe-Jamshidi, D., Mehrzad, S., Mahlke, S.: D2ma: accelerating coarse-grained data transfer for gpus. In: PACT (2014)","DOI":"10.1145\/2628071.2628072"},{"key":"57_CR2","doi-asserted-by":"crossref","unstructured":"Bauer, M., Cook, H., Khailany, B.: Cudadma: optimizing gpu memory bandwidth via warp specialization. In: Intertantional conference on super computing (SC) (2011)","DOI":"10.1145\/2063384.2063400"},{"key":"57_CR3","first-page":"2613","volume":"16","author":"K Berkel","year":"2005","unstructured":"Berkel, K., Heinle, F.: Vector processing as an enabler for software-defined radio in handheld devices. EURASIP J. Appl. Signal Process. 16, 2613\u20132625 (2005)","journal-title":"EURASIP J. Appl. Signal Process."},{"key":"57_CR4","doi-asserted-by":"publisher","first-page":"559","DOI":"10.1147\/rd.515.0559","volume":"51","author":"T Chen","year":"2007","unstructured":"Chen, T., Raghavan, R., Dale, J.: Cell broadband engine architecture and its first implementation a performance view. IBM J. Res. Dev. 51, 559\u2013572 (2007)","journal-title":"IBM J. Res. Dev."},{"key":"57_CR5","doi-asserted-by":"publisher","first-page":"64","DOI":"10.1109\/MM.2013.129","volume":"34","author":"S Chen","year":"2014","unstructured":"Chen, S., Wang, Y., Liu, S., Wan, J., Chen, H., Liu, H., Zhang, K., Liu, X., Ning, X.: Ft-matrix: a coordination-aware architecture for signal processing. IEEE Micro 34, 64\u201373 (2014)","journal-title":"IEEE Micro"},{"key":"57_CR6","doi-asserted-by":"crossref","unstructured":"Dybaahl, H., Stenstrom, P.: An adaptive shared\/private nuca cache partition scheme for chip multiprocessors. In: HPCA2007 (2007)","DOI":"10.1109\/HPCA.2007.346180"},{"key":"57_CR7","doi-asserted-by":"crossref","unstructured":"Efland, G., Parikh, S., Sanghavi, H., Farooqui, A.: High performance dsp for vision, imaging and neural networks. In: Hot Chips 28 Symposium, pp. 1\u201330 (2016)","DOI":"10.1109\/HOTCHIPS.2016.7936210"},{"key":"57_CR8","unstructured":"Geforce gtx 280 specifications, NVIDIA Corporation (2008)"},{"key":"57_CR9","unstructured":"Green500.: In: http:\/\/www.top500.org\/green500 (2016)"},{"key":"57_CR10","doi-asserted-by":"crossref","unstructured":"Heinecke, A., Vaidyanathan, K., Smelyanskiy, M., Kobotov, A., Dubtsov, R., Henry, G., Shet, A. G., Chrysos, G., PradeepDubey, G.: Design and implementation of the linpack benchmarkfor single and multi-node systems based on intel xeon phitmcoprocessor. In: 2013 IEEE 27th international symposium on parallel & distributed processing (IPDPS) (2013)","DOI":"10.1109\/IPDPS.2013.113"},{"key":"57_CR11","doi-asserted-by":"crossref","unstructured":"Igual, F. D., Ali, M., Friedmann, A., Stotzer, E., Wentz, T., van de Geijn, R. A.: Unleashing the high-performance and low-power of multi-core dsps for general-purpose hpc. In: Proceedings of the international conference on high performance computing, networking, storage and analysis (SC12) (2012)","DOI":"10.1109\/SC.2012.109"},{"issue":"7","key":"57_CR12","doi-asserted-by":"publisher","first-page":"1814","DOI":"10.1109\/TPDS.2014.2321742","volume":"26","author":"G Jo","year":"2015","unstructured":"Jo, G., Nah, J., Lee, J., Kim, J., Lee, J.: Accelerating linpack with mpi-opencl on clusters of multi-gpu nodes. IEEE Trans. Parallel Distrib. Syst. 26(7), 1814\u20131825 (2015)","journal-title":"IEEE Trans. Parallel Distrib. Syst."},{"issue":"2","key":"57_CR13","doi-asserted-by":"publisher","first-page":"35","DOI":"10.1109\/40.918001","volume":"21","author":"B Khailany","year":"2001","unstructured":"Khailany, B., et al.: Imagine: media processing with streams. IEEE Micro 21(2), 35\u201346 (2001)","journal-title":"IEEE Micro"},{"key":"57_CR14","unstructured":"Kistler, M., Perrone, M., Fabrizio, P.: Cell multiprocessor communication network: Built for speed. IEEE MICRO (2020)"},{"issue":"6","key":"57_CR15","doi-asserted-by":"publisher","first-page":"84","DOI":"10.1109\/MM.2004.90","volume":"24","author":"B Krashinsky","year":"2004","unstructured":"Krashinsky, B., Batten, C., et al.: The vector-thread architecture. IEEE Micro 24(6), 84\u201390 (2004)","journal-title":"IEEE Micro"},{"issue":"08","key":"57_CR16","doi-asserted-by":"publisher","first-page":"988","DOI":"10.1109\/TC.2004.44","volume":"53","author":"T Lang","year":"2004","unstructured":"Lang, T., Bruguera, J.D.: Floating-point multiply-add-fused with reduced latency. IEEE Trans. Comput. 53(08), 988\u20131003 (2004)","journal-title":"IEEE Trans. Comput."},{"key":"57_CR17","doi-asserted-by":"crossref","unstructured":"Lee, Y., et al.: Exploring the tradeoffs between programmability and efficiency in data-parallel accelerators. In: International symposium on computer architecture (2011)","DOI":"10.1145\/2000064.2000080"},{"key":"57_CR18","unstructured":"Micrium.: In: http:\/\/www.micrium.com (2016)"},{"key":"57_CR19","unstructured":"Nvidia ampere architecture whitepaper, Nvidia (2020)"},{"key":"57_CR20","unstructured":"Nvidia\u2019s next generation cuda compute architecture: Fermi, NVIDIA Corporation (2009)"},{"key":"57_CR21","unstructured":"Philips, E.: Cuda accelerated linpack on clusters. In: ACM\/IEEE international Conference of supercomputing (2010)"},{"key":"57_CR22","first-page":"57","volume":"2007","author":"P Raghavan","year":"2007","unstructured":"Raghavan, P., Munaga, S.: A customized cross-bar for data shuffling in domain specific SIMD processors. Proc. Archit. Comput. Syst. 2007, 57\u201368 (2007)","journal-title":"Proc. Archit. Comput. Syst."},{"key":"57_CR23","unstructured":"Ranjith, S., Yannis, S.: Adaptive caches: effective shaping of cache behavior to workloads. In: 39th Annual IEEE\/ACM international symposium on microarchitecture (MICRO06) (2006)"},{"key":"57_CR24","unstructured":"Reddy, V. G.: Neon technology introduction, ARM Corp. (2008)"},{"key":"57_CR25","doi-asserted-by":"crossref","unstructured":"Rivoire, S., et al.: Vector lane threading. In: International conference on parallel processing, pp. 55\u201364 (2006)","DOI":"10.1109\/ICPP.2006.74"},{"issue":"11","key":"57_CR26","doi-asserted-by":"publisher","first-page":"1829","DOI":"10.1109\/4.726584","volume":"33","author":"S Santhanam","year":"1998","unstructured":"Santhanam, S., et al.: A low-cost, 300MHz, RISC CPU with attached media processor. IEEE J. Solid-State Circ. 33(11), 1829\u20131839 (1998)","journal-title":"IEEE J. Solid-State Circ."},{"key":"57_CR27","doi-asserted-by":"crossref","unstructured":"Seiler, L.: Larrabee: a many-core x86 architecture for visual computing. In: ACM SIGGRAPH, pp. 1\u201315. NY, USA, New York (2008)","DOI":"10.1145\/1360612.1360617"},{"key":"57_CR28","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/1369396.1369401","volume":"5","author":"A Shahbahrami","year":"2008","unstructured":"Shahbahrami, A., Juurlink, B., Vassiliadis, S.: Versatility of extended subwords and the matrix register file. ACM Trans. Archit. Code Optim. 5, 1 (2008)","journal-title":"ACM Trans. Archit. Code Optim."},{"issue":"4","key":"57_CR29","doi-asserted-by":"publisher","first-page":"56","DOI":"10.1109\/40.612224","volume":"17","author":"P Soderquis","year":"1997","unstructured":"Soderquis, P., Leeser, M.: Division and square root: choosing the right implementation. IEEE Micro 17(4), 56\u201366 (1997)","journal-title":"IEEE Micro"},{"key":"57_CR31","unstructured":"T. I. (TI): Tms320c6678 multicore fixed and floating-point digital signal processor (2012)"},{"key":"57_CR30","first-page":"379","volume":"38","author":"JS Walther","year":"1971","unstructured":"Walther, J.S.: A unified algorithm for elementary functions. Proc. AFIPS Conf. 38, 379\u2013385 (1971)","journal-title":"Proc. AFIPS Conf."},{"key":"57_CR32","doi-asserted-by":"publisher","first-page":"635","DOI":"10.1007\/s11390-005-0635-7","volume":"20","author":"Wen","year":"2005","unstructured":"Wen, et al.: Multiple-morphs adaptive stream architecture. J. Comput. Sci. Technol. 20, 635\u2013646 (2005)","journal-title":"J. Comput. Sci. Technol."},{"issue":"1","key":"57_CR33","doi-asserted-by":"publisher","first-page":"81","DOI":"10.1109\/MM.2010.8","volume":"30","author":"M Woh","year":"2010","unstructured":"Woh, M., et al.: Anysp: anytime anywhere anyway signal processing. IEEE Micro 30(1), 81\u201391 (2010)","journal-title":"IEEE Micro"},{"key":"57_CR34","doi-asserted-by":"crossref","unstructured":"Yang, X. et al.: A 64-bit stream processor architecture for scientific applications. In: International symposium on computer architecture (2007)","DOI":"10.1145\/1250662.1250689"},{"key":"57_CR35","doi-asserted-by":"publisher","first-page":"36413","DOI":"10.1109\/ACCESS.2019.2905302","volume":"7","author":"C Yang","year":"2019","unstructured":"Yang, C., Chen, S., Zhang, J., Lv, Z., Wang, Z.: A novel DSP architecture for scientific computing and deep learning. IEEE Access 7, 36413\u201336425 (2019)","journal-title":"IEEE Access"}],"container-title":["CCF Transactions on High Performance Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s42514-020-00057-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s42514-020-00057-2\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s42514-020-00057-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,4,16]],"date-time":"2021-04-16T07:55:17Z","timestamp":1618559717000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s42514-020-00057-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,3]]},"references-count":35,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2021,3]]}},"alternative-id":["57"],"URL":"https:\/\/doi.org\/10.1007\/s42514-020-00057-2","relation":{},"ISSN":["2524-4922","2524-4930"],"issn-type":[{"value":"2524-4922","type":"print"},{"value":"2524-4930","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021,3]]},"assertion":[{"value":"15 June 2020","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 October 2020","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"31 March 2021","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}