{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,8]],"date-time":"2026-03-08T03:41:58Z","timestamp":1772941318137,"version":"3.50.1"},"reference-count":39,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2026,3,7]],"date-time":"2026-03-07T00:00:00Z","timestamp":1772841600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,3,7]],"date-time":"2026-03-07T00:00:00Z","timestamp":1772841600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Supercomput"],"DOI":"10.1007\/s11227-026-08404-w","type":"journal-article","created":{"date-parts":[[2026,3,7]],"date-time":"2026-03-07T10:47:39Z","timestamp":1772880459000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Compiler-based loop unrolling optimization for dual-SIMD extensions"],"prefix":"10.1007","volume":"82","author":[{"given":"Jinyang","family":"Yao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lili","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xuanyu","family":"Fu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wenbo","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chaowei","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zheng","family":"Shan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,3,7]]},"reference":[{"key":"8404_CR1","doi-asserted-by":"publisher","unstructured":"Gao W, Zhao R, Han L, Pang J, Ding R (2015) Research on simd auto-vectorization compiling optimization. J Softw 26(6):1265\u20131284. https:\/\/doi.org\/10.13328\/j.cnki.jos.004811","DOI":"10.13328\/j.cnki.jos.004811"},{"key":"8404_CR2","doi-asserted-by":"publisher","unstructured":"Zheng R, Pai S (2021) Efficient execution of graph algorithms on cpu with simd extensions. In: 2021 IEEE\/ACM International Symposium on Code Generation and Optimization (CGO), pp. 262\u2013276 . https:\/\/doi.org\/10.1109\/CGO51591.2021.9370326.IEEE","DOI":"10.1109\/CGO51591.2021.9370326."},{"key":"8404_CR3","doi-asserted-by":"publisher","first-page":"371","DOI":"10.1016\/j.future.2020.10.036","volume":"116","author":"H Bian","year":"2021","unstructured":"Bian H, Huang J, Liu L, Huang D, Wang X (2021) Albus: a method for efficiently processing spmv using simd and load balancing. Futur Gener Comput Syst 116:371\u2013392. https:\/\/doi.org\/10.1016\/j.future.2020.10.036","journal-title":"Futur Gener Comput Syst"},{"key":"8404_CR4","doi-asserted-by":"publisher","unstructured":"Cheng Y, An H, Chen Z, Li F, Wang Z, Jiang X, Peng Y (2014) Understanding the simd efficiency of graph traversal on gpu. In: Sun, X.-H., Qu, W., Stojmenovic, I., Zhou, W., Li, Z., Guo, H., Min, G., Yang, T., Wu, Y., Liu, L. (eds.) Algorithms and Architectures for Parallel Processing, pp. 42\u201356. Springer, Cham. https:\/\/doi.org\/10.1007\/978-3-319-11197-1_4","DOI":"10.1007\/978-3-319-11197-1_4"},{"key":"8404_CR5","unstructured":"Yamazaki S (2021) Future possibilities and effectiveness of jit from elixir code of image processing and machine learning into native code with simd instructions. In: IPSJ Special Interest Group on Programming 136th Meeting. http:\/\/id.nii.ac.jp\/1001\/00218031\/"},{"key":"8404_CR6","unstructured":"Intel Corporation: Intel\u00ae Intrinsics Guide. https:\/\/www.intel.com\/content\/www\/us\/en\/docs\/intrinsics-guide\/index.html#ig_expand=6859. Accessed: 2023\u201304-23"},{"key":"8404_CR7","unstructured":"Intel Corporation: Intrinsics for Short Vector Math Library Operations (SVML). https:\/\/www.intel.com\/content\/www\/us\/en\/docs\/cpp-compiler\/developer-guide-reference\/2021-8\/intrinsics-for-short-vector-math-library-ops.html. Accessed: 2023\u201304-23"},{"issue":"2","key":"8404_CR8","doi-asserted-by":"publisher","first-page":"26","DOI":"10.1109\/MM.2017.35","volume":"37","author":"N Stephens","year":"2017","unstructured":"Stephens N, Biles S, Boettcher M, Eapen J, Eyole M, Gabrielli G, Horsnell M, Magklis G, Martinez A, Premillieu N et al (2017) The arm scalable vector extension. IEEE Micro 37(2):26\u201339. https:\/\/doi.org\/10.1109\/MM.2017.35","journal-title":"IEEE Micro"},{"issue":"5","key":"8404_CR9","doi-asserted-by":"publisher","first-page":"41","DOI":"10.1109\/MM.2022.3184867","volume":"42","author":"N Adit","year":"2022","unstructured":"Adit N, Sampson A (2022) Performance left on the table: an evaluation of compiler autovectorization for risc-v. IEEE Micro 42(5):41\u201348. https:\/\/doi.org\/10.1109\/MM.2022.3184867","journal-title":"IEEE Micro"},{"key":"8404_CR10","doi-asserted-by":"publisher","unstructured":"Patsidis K, Nicopoulos C, Sirakoulis GC, Dimitrakopoulos G (2020) Risc-v 2: a scalable risc-v vector processor. In: 2020 IEEE International Symposium on Circuits and Systems (ISCAS), pp. 1\u20135. https:\/\/doi.org\/10.1109\/ISCAS45731.2020.9181071.IEEE","DOI":"10.1109\/ISCAS45731.2020.9181071."},{"key":"8404_CR11","unstructured":"Chengdu Sunway Technology Corporation Limited: SW421 Processor Software Interface Manual. http:\/\/www.swcpu.cn\/uploadfile\/2018\/0709\/20180709030610409.pdf. Accessed: 2023\u201304-23"},{"key":"8404_CR12","doi-asserted-by":"publisher","unstructured":"Allen JR, Kennedy K, Porterfield C, Warren J (1983) Conversion of control dependence to data dependence. In: Proceedings of the 10th ACM SIGACT-SIGPLAN Symposium on Principles of Programming Languages. POPL \u201983, pp. 177\u2013189. Association for Computing Machinery, New York, NY, USA. https:\/\/doi.org\/10.1145\/567067.567085","DOI":"10.1145\/567067.567085"},{"key":"8404_CR13","doi-asserted-by":"publisher","unstructured":"Rocha RCO, Porpodas V, Petoumenos P, G\u00f3es LFW, Wang Z, Cole M, Leather H (2020) Vectorization-aware loop unrolling with seed forwarding. In: Proceedings of the 29th International Conference on Compiler Construction. CC 2020, pp. 1\u201313. Association for Computing Machinery, New York, NY, USA. https:\/\/doi.org\/10.1145\/3377555.3377890","DOI":"10.1145\/3377555.3377890"},{"key":"8404_CR14","doi-asserted-by":"publisher","unstructured":"Haj-Ali A, Ahmed NK, Willke T, Shao YS, Asanovic K, Stoica I (2020) Neurovectorizer: end-to-end vectorization with deep reinforcement learning. In: Proceedings of the 18th ACM\/IEEE International Symposium on Code Generation and Optimization. CGO \u201920, pp. 242\u2013255. Association for Computing Machinery, New York, NY, USA. https:\/\/doi.org\/10.1145\/3368826.3377928","DOI":"10.1145\/3368826.3377928"},{"key":"8404_CR15","doi-asserted-by":"publisher","unstructured":"Chatarasi P, Neuendorffer S, Bayliss S, Vissers K, Sarkar V (2020) Vyasa: A high-performance vectorizing compiler for tensor convolutions on the xilinx ai engine. In: 2020 IEEE High Performance Extreme Computing Conference (HPEC), pp. 1\u201310. https:\/\/doi.org\/10.1109\/HPEC43674.2020.9286183","DOI":"10.1109\/HPEC43674.2020.9286183"},{"key":"8404_CR16","unstructured":"Mendis C, Yang C, Pu Y, Amarasinghe SP, Carbin M (2019) Compiler auto-vectorization with imitation learning. In: Neural Information Processing Systems. https:\/\/api.semanticscholar.org\/CorpusID:209412939"},{"key":"8404_CR17","unstructured":"Taneja J, Laird A, Yan C, Musuvathi M, Lahiri SK (2024) LLM-Vectorizer: LLM-based Verified Loop Vectorizer. arXiv e-prints, 2406\u201304693. 10.48550\/arXiv. 2406.04693 arXiv:2406.04693 [cs.SE]"},{"key":"8404_CR18","doi-asserted-by":"publisher","unstructured":"Liu B, Laird A, Tsang WH, Mahjour B, Dehnavi MM (2023) Combining run-time checks and compile-time analysis to improve control flow auto-vectorization. In: Proceedings of the International Conference on Parallel Architectures and Compilation Techniques. PACT \u201922, pp. 439\u2013450. Association for Computing Machinery, New York, NY, USA. https:\/\/doi.org\/10.1145\/3559009.3569663","DOI":"10.1145\/3559009.3569663"},{"key":"8404_CR19","doi-asserted-by":"publisher","unstructured":"Chen Y, Mendis C, Amarasinghe S (2020) All you need is superword-level parallelism: systematic control-flow vectorization with slp. In: Proceedings of the 43rd ACM SIGPLAN International Conference on Programming Language Design and Implementation. PLDI 2022, pp. 301\u2013315. Association for Computing Machinery, New York, NY, USA. https:\/\/doi.org\/10.1145\/3519939.3523701","DOI":"10.1145\/3519939.3523701"},{"key":"8404_CR20","doi-asserted-by":"publisher","unstructured":"Porpodas V, Rocha RCO, Brevnov E, G\u00f3es LFW, Mattson T (2019) Super-node slp: Optimized vectorization for code sequences containing operators and their inverse elements. In: 2019 IEEE\/ACM International Symposium on Code Generation and Optimization (CGO), pp. 206\u2013216. https:\/\/doi.org\/10.1109\/CGO.2019.8661192","DOI":"10.1109\/CGO.2019.8661192"},{"key":"8404_CR21","doi-asserted-by":"publisher","unstructured":"Chen Y, Mendis C, Carbin M, Amarasinghe S (2021) Vegen: a vectorizer generator for simd and beyond. In: Proceedings of the 26th ACM International Conference on Architectural Support for Programming Languages and Operating Systems. ASPLOS \u201921, pp. 902\u2013914. Association for Computing Machinery, New York, NY, USA. https:\/\/doi.org\/10.1145\/3445814.3446692","DOI":"10.1145\/3445814.3446692"},{"issue":"4","key":"8404_CR22","doi-asserted-by":"publisher","first-page":"543","DOI":"10.1145\/3296979.3192413","volume":"53","author":"S Moll","year":"2018","unstructured":"Moll S, Hack S (2018) Partial control-flow linearization. SIGPLAN Not 53(4):543\u2013556. https:\/\/doi.org\/10.1145\/3296979.3192413","journal-title":"SIGPLAN Not"},{"key":"8404_CR23","doi-asserted-by":"publisher","unstructured":"Kandiah V, Lustig D, Villa O, Nellans D, Hardavellas N (2023) Parsimony: Enabling simd\/vector programming in standard compiler flows. In: Proceedings of the 21st ACM\/IEEE International Symposium on Code Generation and Optimization. CGO \u201923, pp. 186\u2013198. Association for Computing Machinery, New York, NY, USA. https:\/\/doi.org\/10.1145\/3579990.3580019","DOI":"10.1145\/3579990.3580019"},{"key":"8404_CR24","doi-asserted-by":"publisher","unstructured":"Oo NZ, Chaikan P (2021) The effect of loop unrolling in energy efficient strassen\u2019s algorithm on shared memory architecture. In: 2021 36th International Technical Conference on Circuits\/Systems, Computers and Communications (ITC-CSCC), pp. 1\u20134. https:\/\/doi.org\/10.1109\/ITC-CSCC52171.2021.9501472","DOI":"10.1109\/ITC-CSCC52171.2021.9501472"},{"key":"8404_CR25","doi-asserted-by":"publisher","unstructured":"Singh I, Singh SK, Singh R, Kumar S (2022) Efficient loop unrolling factor prediction algorithm using machine learning models. In: 2022 3rd International Conference for Emerging Technology (INCET), pp. 1\u20138. https:\/\/doi.org\/10.1109\/INCET54531.2022.9825092","DOI":"10.1109\/INCET54531.2022.9825092"},{"key":"8404_CR26","doi-asserted-by":"publisher","unstructured":"Rocha RCO, Petoumenos P, Franke B, Bhatotia P, O\u2019Boyle M (2022) Loop rolling for code size reduction. In: 2022 IEEE\/ACM International Symposium on Code Generation and Optimization (CGO), pp. 217\u2013229. https:\/\/doi.org\/10.1109\/CGO53902.2022.9741256","DOI":"10.1109\/CGO53902.2022.9741256"},{"key":"8404_CR27","doi-asserted-by":"publisher","unstructured":"Dai Y, Li Q, Zhang Q, Jay Kuo C-C (2004) Simd - efficient loop unrolling design for embedded multimedia applications. In: 2004 IEEE International Conference on Multimedia and Expo (ICME) (IEEE Cat. No.04TH8763), vol. 3, pp. 1851\u201318543. https:\/\/doi.org\/10.1109\/ICME.2004.1394618","DOI":"10.1109\/ICME.2004.1394618"},{"key":"8404_CR28","doi-asserted-by":"publisher","unstructured":"Souravlas S, Anastasiadou S (2020) Pipelined dynamic scheduling of big data streams. Applied Sciences 10(14) (2020) https:\/\/doi.org\/10.3390\/app10144796","DOI":"10.3390\/app10144796"},{"key":"8404_CR29","doi-asserted-by":"publisher","unstructured":"Mao H, Schwarzkopf M, Venkatakrishnan SB, Meng Z, Alizadeh M (2019) Learning scheduling algorithms for data processing clusters. In: Proceedings of the ACM Special Interest Group on Data Communication. SIGCOMM \u201919, pp. 270\u2013288. Association for Computing Machinery, New York, NY, USA. https:\/\/doi.org\/10.1145\/3341302.3342080","DOI":"10.1145\/3341302.3342080"},{"key":"8404_CR30","doi-asserted-by":"publisher","unstructured":"Cheng J, Josipovic L, Constantinides GA, Ienne P, Wickerson J (2020) Combining dynamic & static scheduling in high-level synthesis. In: Proceedings of the 2020 ACM\/SIGDA International Symposium on Field-Programmable Gate Arrays. FPGA \u201920, pp. 288\u2013298. Association for Computing Machinery, New York, NY, USA. https:\/\/doi.org\/10.1145\/3373087.3375297","DOI":"10.1145\/3373087.3375297"},{"key":"8404_CR31","doi-asserted-by":"publisher","first-page":"98","DOI":"10.1016\/j.future.2019.04.029","volume":"100","author":"V Arabnejad","year":"2020","unstructured":"Arabnejad V, Bubendorfer K, Ng B (2020) Dynamic multi-workflow scheduling: a deadline and cost-aware approach for commercial clouds. Futur Gener Comput Syst 100:98\u2013108. https:\/\/doi.org\/10.1016\/j.future.2019.04.029","journal-title":"Futur Gener Comput Syst"},{"key":"8404_CR32","unstructured":"Yang B, Zhang J, Li J, R\u00e9 C, Aberger CR, Sa CD (2020) Pipemare: Asynchronous pipeline parallel dnn training arXiv:1910.05124 [cs.DC]"},{"issue":"3","key":"8404_CR33","doi-asserted-by":"publisher","first-page":"635","DOI":"10.1145\/3355399","volume":"53","author":"X Liu","year":"2020","unstructured":"Liu X, Buyya R (2020) Resource management and scheduling in distributed stream processing systems: a taxonomy, review, and future directions. ACM Comput Surv 53(3):635. https:\/\/doi.org\/10.1145\/3355399","journal-title":"ACM Comput Surv"},{"issue":"6","key":"8404_CR34","doi-asserted-by":"publisher","first-page":"853","DOI":"10.1145\/267959.269966","volume":"19","author":"S-M Moon","year":"1997","unstructured":"Moon S-M, Ebcio\u011flu K (1997) Parallelizing nonnumerical code with selective scheduling and software pipelining. ACM Trans Program Lang Syst 19(6):853\u2013898. https:\/\/doi.org\/10.1145\/267959.269966","journal-title":"ACM Trans Program Lang Syst"},{"key":"8404_CR35","doi-asserted-by":"publisher","unstructured":"Gallagher DM, Chen WY, Mahlke SA, Gyllenhaal JC, Hwu W-mW (1994) Dynamic memory disambiguation using the memory conflict buffer. In: Proceedings of the Sixth International Conference on Architectural Support for Programming Languages and Operating Systems. ASPLOS VI, pp. 183\u2013193. Association for Computing Machinery, New York, NY, USA. https:\/\/doi.org\/10.1145\/195473.195534","DOI":"10.1145\/195473.195534"},{"issue":"3","key":"8404_CR36","doi-asserted-by":"publisher","first-page":"319","DOI":"10.1145\/24039.24041","volume":"9","author":"J Ferrante","year":"1987","unstructured":"Ferrante J, Ottenstein KJ, Warren JD (1987) The program dependence graph and its use in optimization. ACM Trans Program Lang Syst 9(3):319\u2013349. https:\/\/doi.org\/10.1145\/24039.24041","journal-title":"ACM Trans Program Lang Syst"},{"issue":"3","key":"8404_CR37","doi-asserted-by":"publisher","first-page":"63","DOI":"10.1177\/109434209100500306","volume":"5","author":"DH Bailey","year":"1991","unstructured":"Bailey DH, Barszcz E, Barton JT, Browning DS, Carter RL, Dagum L, Fatoohi RA, Frederickson PO, Lasinski TA, Schreiber RS, Simon HD, Venkatakrishnan V, Weeratunga SK (1991) The nas parallel benchmarks. Int J High Perform Comput Appl 5(3):63\u201373. https:\/\/doi.org\/10.1177\/109434209100500306","journal-title":"Int J High Perform Comput Appl"},{"key":"8404_CR38","unstructured":"Standard Performance Evaluation Corporation (SPEC): Spec cpu2006 benchmark descriptions. Technical report, Standard Performance Evaluation Corporation (SPEC) (2006). Available at https:\/\/www.spec.org\/cpu2006\/Docs"},{"key":"8404_CR39","unstructured":"Standard Performance Evaluation Corporation (SPEC): Spec cpu2017 benchmark descriptions. Technical report, Standard Performance Evaluation Corporation (SPEC) (2017). Available at https:\/\/www.spec.org\/cpu2017\/Docs"}],"container-title":["The Journal of Supercomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-026-08404-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11227-026-08404-w","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11227-026-08404-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,7]],"date-time":"2026-03-07T10:47:41Z","timestamp":1772880461000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11227-026-08404-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,7]]},"references-count":39,"journal-issue":{"issue":"4","published-online":{"date-parts":[[2026,3]]}},"alternative-id":["8404"],"URL":"https:\/\/doi.org\/10.1007\/s11227-026-08404-w","relation":{},"ISSN":["1573-0484"],"issn-type":[{"value":"1573-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,3,7]]},"assertion":[{"value":"14 May 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 February 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 March 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"228"}}