{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T14:30:56Z","timestamp":1787495456198,"version":"build-2736575974"},"publisher-location":"Singapore","reference-count":38,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819248049","type":"print"},{"value":"9789819248056","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,8,24]],"date-time":"2026-08-24T00:00:00Z","timestamp":1787529600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,8,24]],"date-time":"2026-08-24T00:00:00Z","timestamp":1787529600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-4805-6_25","type":"book-chapter","created":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T13:48:43Z","timestamp":1787492923000},"page":"375-389","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["DOA: Dataflow Optimization for\u00a0Attention on\u00a0Multi-core DSPs with\u00a0Three-Level Memory Hierarchy"],"prefix":"10.1007","author":[{"given":"Shengwei","family":"Li","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhiquan","family":"Lai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shuai","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shun","family":"Ouyang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Han","family":"Xiao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhaoning","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Menghan","family":"Jia","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Huayou","family":"Su","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dongsheng","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,8,24]]},"reference":[{"key":"25_CR1","doi-asserted-by":"crossref","unstructured":"Ainslie, J., Lee-Thorp, J., de\u00a0Jong, M., Zemlyanskiy, Y., Lebron, F., Sanghai, S.: GQA: training generalized multi-query transformer models from multi-head checkpoints. In: Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, pp. 4895\u20134901 (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.298"},{"key":"25_CR2","unstructured":"Chen, Q., Li, S., Gao, W., Sun, P., Wen, Y., Zhang, T.: SPPO: efficient long-sequence LLM training via adaptive sequence pipeline parallel offloading (2025). arXiv:2503.10377 arXiv preprint"},{"issue":"240","key":"25_CR3","first-page":"1","volume":"24","author":"A Chowdhery","year":"2023","unstructured":"Chowdhery, A., et al.: PaLM: scaling language modeling with pathways. J. Mach. Learn. Res. 24(240), 1\u2013113 (2023)","journal-title":"J. Mach. Learn. Res."},{"key":"25_CR4","unstructured":"Dao, T.: FlashAttention-2: faster attention with better parallelism and work partitioning (2023). arXiv:2307.08691 arXiv preprint"},{"key":"25_CR5","doi-asserted-by":"crossref","unstructured":"Dao, T., Fu, D., Ermon, S., Rudra, A., R\u00e9, C.: FlashAttention: fast and memory-efficient exact attention with io-awareness. In: Advances in Neural Information Processing Systems, vol. 35, pp. 16344\u201316359 (2022)","DOI":"10.52202\/068431-1189"},{"key":"25_CR6","doi-asserted-by":"crossref","unstructured":"Ding, W., Tang, X., Kandemir, M., Zhang, Y., Kultursay, E.: Optimizing off-chip accesses in multicores. In: Proceedings of the 36th ACM SIGPLAN Conference on Programming Language Design and Implementation, pp. 131\u2013142 (2015)","DOI":"10.1145\/2737924.2737989"},{"key":"25_CR7","unstructured":"Dong, J., Feng, B., Guessous, D., Liang, Y., He, H.: FlexAttention: a programming model for generating fused attention variants. Proc. Mach. Learn. Syst. 7 (2025)"},{"key":"25_CR8","doi-asserted-by":"crossref","unstructured":"Fang, J., et al.: MTGEMM: an efficient GEMM library for modern multi-core DSPs. IEEE Trans. Parallel Distrib. Syst. (2026)","DOI":"10.1109\/TPDS.2026.3664114"},{"key":"25_CR9","unstructured":"GitHub: Github copilot. https:\/\/copilot.github.com\/ (2022)"},{"issue":"8081","key":"25_CR10","doi-asserted-by":"publisher","first-page":"633","DOI":"10.1038\/s41586-025-09422-z","volume":"645","author":"D Guo","year":"2025","unstructured":"Guo, D., et al.: DeepSeek-R1 incentivizes reasoning in LLMs through reinforcement learning. Nature 645(8081), 633\u2013638 (2025)","journal-title":"Nature"},{"key":"25_CR11","doi-asserted-by":"crossref","unstructured":"Jacobs, S.A., et al.: System optimizations for enabling training of extreme long sequence transformer models. In: Proceedings of the 43rd ACM Symposium on Principles of Distributed Computing. pp. 121\u2013130 (2024)","DOI":"10.1145\/3662158.3662806"},{"key":"25_CR12","unstructured":"Jain, P., et al.: CheckMate: breaking the memory wall with optimal tensor rematerialization. In: Proceedings of Machine Learning and Systems, vol.\u00a02, pp. 497\u2013511 (2020)"},{"key":"25_CR13","doi-asserted-by":"crossref","unstructured":"Kao, S.C., Subramanian, S., Agrawal, G., Yazdanbakhsh, A., Krishna, T.: FLAT: an optimized dataflow for mitigating attention bottlenecks. In: Proceedings of the 28th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, vol. 2, pp. 295\u2013310 (2023)","DOI":"10.1145\/3575693.3575747"},{"issue":"5","key":"25_CR14","doi-asserted-by":"publisher","first-page":"1466","DOI":"10.1109\/TPDS.2023.3247001","volume":"34","author":"Z Lai","year":"2023","unstructured":"Lai, Z., et al.: Merak: an efficient distributed DNN training framework with automated 3D parallelism for giant foundation models. IEEE Trans. Parallel Distrib. Syst. 34(5), 1466\u20131478 (2023)","journal-title":"IEEE Trans. Parallel Distrib. Syst."},{"key":"25_CR15","doi-asserted-by":"crossref","unstructured":"Li, S., Xue, F., Baranwal, C., Li, Y., You, Y.: Sequence Parallelism: long sequence training from system perspective. In: Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics, (vol. 1: Long Papers), pp. 2391\u20132404 (2023)","DOI":"10.18653\/v1\/2023.acl-long.134"},{"issue":"11","key":"25_CR16","doi-asserted-by":"publisher","first-page":"2747","DOI":"10.14778\/3551793.3551828","volume":"15","author":"Y Li","year":"2022","unstructured":"Li, Y., Phanishayee, A., Murray, D., Tarnawski, J., Kim, N.S.: Harmony: overcoming the hurdles of GPU memory capacity to train massive DNN models on commodity servers. Proc. VLDB Endow. 15(11), 2747\u20132760 (2022)","journal-title":"Proc. VLDB Endow."},{"key":"25_CR17","unstructured":"Liu, A., et al.: DeepSeek-V3 technical report (2024). arXiv:2412.19437 arXiv preprint"},{"issue":"4","key":"25_CR18","doi-asserted-by":"publisher","first-page":"1196","DOI":"10.1109\/TC.2024.3517748","volume":"74","author":"W Liu","year":"2024","unstructured":"Liu, W., et al.: AutoPipe-H: a heterogeneity-aware data-paralleled pipeline approach on commodity GPU servers. IEEE Trans. Comput. 74(4), 1196\u20131209 (2024)","journal-title":"IEEE Trans. Comput."},{"key":"25_CR19","doi-asserted-by":"crossref","unstructured":"Narayanan, D., et\u00a0al.: Efficient large-scale language model training on GPU clusters using Megatron-LM. In: Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis, pp. 1\u201315 (2021)","DOI":"10.1145\/3458817.3476209"},{"key":"25_CR20","unstructured":"OpenAI: GPT-4 technical report. https:\/\/openai.com\/research\/gpt-4 (2024)"},{"key":"25_CR21","unstructured":"Paszke, A., et al.: PYTorch: an imperative style, high-performance deep learning library. In: Advances in Neural Information Processing Systems, vol. 32 (2019)"},{"key":"25_CR22","doi-asserted-by":"crossref","unstructured":"Peng, X., et al.: Capuchin: tensor-based GPU memory management for deep learning. In: Proceedings of the Twenty-Fifth International Conference on Architectural Support for Programming Languages and Operating Systems, pp. 891\u2013905 (2020)","DOI":"10.1145\/3373376.3378505"},{"key":"25_CR23","doi-asserted-by":"crossref","unstructured":"Qu, Z., Liu, L., Tu, F., Chen, Z., Ding, Y., Xie, Y.: DOTA: detect and omit weak attentions for scalable transformer acceleration. In: Proceedings of the 27th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, pp. 14\u201326 (2022)","DOI":"10.1145\/3503222.3507738"},{"issue":"140","key":"25_CR24","first-page":"1","volume":"21","author":"C Raffel","year":"2020","unstructured":"Raffel, C., et al.: Exploring the limits of transfer learning with a unified text-to-text transformer. J. Mach. Learn. Res. 21(140), 1\u201367 (2020)","journal-title":"J. Mach. Learn. Res."},{"key":"25_CR25","doi-asserted-by":"crossref","unstructured":"Rajbhandari, S., Ruwase, O., Rasley, J., Smith, S., He, Y.: Zero-infinity: breaking the GPU memory wall for extreme scale deep learning. In: Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis, pp. 1\u201314 (2021)","DOI":"10.1145\/3458817.3476205"},{"key":"25_CR26","unstructured":"Sanovar, R., Bharadwaj, S., St\u00a0Amant, R., R\u00fchle, V., Rajmohan, S.: LeanAttention: hardware-aware scalable attention mechanism for the decode-phase of transformers. Proc. Mach. Learn. Syst. 7 (2025)"},{"key":"25_CR27","doi-asserted-by":"crossref","unstructured":"Shah, J., Bikshandi, G., Zhang, Y., Thakkar, V., Ramani, P., Dao, T.: FlashAttention-3: fast and accurate attention with asynchrony and low-precision. In: Advances in Neural Information Processing Systems, vol. 37, pp. 68658\u201368685 (2024)","DOI":"10.52202\/079017-2193"},{"issue":"4","key":"25_CR28","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3689338","volume":"21","author":"Y Tang","year":"2024","unstructured":"Tang, Y., et al.: Delta: Memory-efficient training via dynamic fine-grained recomputation and swapping. ACM Trans. Archit. Code Optimiz. 21(4), 1\u201325 (2024)","journal-title":"ACM Trans. Archit. Code Optimiz."},{"key":"25_CR29","unstructured":"Team, G., et al.: Gemini: a family of highly capable multimodal models (2023). arXiv:2312.11805 arXiv preprint"},{"key":"25_CR30","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Advances in Neural Information Processing Systems, vol.\u00a030 (2017)"},{"key":"25_CR31","unstructured":"Wolf, T., et al.: Transformers: state-of-the-art natural language processing. In: Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing: System Demonstrations, pp. 38\u201345 (2020)"},{"key":"25_CR32","unstructured":"Yang, A., et al.: Context parallelism for scalable million-token inference. In: Eighth Conference on Machine Learning and Systems (2025)"},{"key":"25_CR33","unstructured":"Yang, A., et\u00a0al.: Qwen3 technical report. arXiv preprint arXiv:2505.09388 (2025)"},{"key":"25_CR34","unstructured":"Yao, J., Jacobs, S.A., Tanaka, M., Ruwase, O., Subramoni, H., Panda, D.K.: Training ultra long context language model with fully pipelined distributed transformer. Proc. Mach. Learn. Syst. 7 (2025)"},{"key":"25_CR35","unstructured":"Ye, Z., et al.: FlashInfer: efficient and customizable attention engine for LLM inference serving. Proc. Mach. Learn. Syst. 7 (2025)"},{"key":"25_CR36","doi-asserted-by":"crossref","unstructured":"Yu, K., et al.: Optimizing general matrix multiplications on modern multi-core DSPs. In: 2024 IEEE International Parallel and Distributed Processing Symposium (IPDPS), pp. 964\u2013975. IEEE (2024)","DOI":"10.1109\/IPDPS57955.2024.00090"},{"issue":"1","key":"25_CR37","doi-asserted-by":"publisher","first-page":"136","DOI":"10.1109\/TCAD.2022.3170848","volume":"42","author":"Z Zhou","year":"2022","unstructured":"Zhou, Z., Liu, J., Gu, Z., Sun, G.: EnerGON: toward efficient acceleration of transformers using dynamic sparse attention. IEEE Trans. Comput. Aided Des. Integr. Circuits Syst. 42(1), 136\u2013149 (2022)","journal-title":"IEEE Trans. Comput. Aided Des. Integr. Circuits Syst."},{"key":"25_CR38","doi-asserted-by":"crossref","unstructured":"Zuo, X., Xiao, C., Wang, Q., Shi, C.: Optimizing batched small matrix multiplication on multi-core DSP architecture. In: 2024 IEEE International Symposium on Parallel and Distributed Processing with Applications (ISPA), pp. 426\u2013435. IEEE (2024)","DOI":"10.1109\/ISPA63168.2024.00061"}],"container-title":["Lecture Notes in Computer Science","Advanced Parallel Processing Technologies"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-4805-6_25","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T13:48:46Z","timestamp":1787492926000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-4805-6_25"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8,24]]},"ISBN":["9789819248049","9789819248056"],"references-count":38,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-4805-6_25","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,8,24]]},"assertion":[{"value":"24 August 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"APPT","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Symposium on Advanced Parallel Processing Technologies","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Brussels","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Belgium","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"appt2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.appt-conference.com\/2026","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}