{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,11]],"date-time":"2025-09-11T19:06:12Z","timestamp":1757617572845,"version":"3.44.0"},"publisher-location":"Singapore","reference-count":23,"publisher":"Springer Nature Singapore","isbn-type":[{"type":"print","value":"9789819628292"},{"type":"electronic","value":"9789819628308"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-96-2830-8_14","type":"book-chapter","created":{"date-parts":[[2025,3,30]],"date-time":"2025-03-30T19:19:39Z","timestamp":1743362379000},"page":"172-190","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["VConv: Autotiling Convolution Algorithm Based on\u00a0MLIR for\u00a0Multi-core Vector accelerators"],"prefix":"10.1007","author":[{"given":"Xiaorong","family":"Chen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Cheng","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhong","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,3,29]]},"reference":[{"key":"14_CR1","unstructured":"Chen, T., et\u00a0al.: $$\\{$$TVM$$\\}$$: an automated $$\\{$$End-to-End$$\\}$$ optimizing compiler for deep learning. In: 13th USENIX Symposium on Operating Systems Design and Implementation (OSDI 2018), pp. 578\u2013594 (2018)"},{"key":"14_CR2","unstructured":"Chetlur, S., et al.: cudnn: efficient primitives for deep learning. arXiv preprint arXiv:1410.0759 (2014)"},{"issue":"4","key":"14_CR3","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3625004","volume":"20","author":"V Ferrari","year":"2023","unstructured":"Ferrari, V., et al.: Advancing direct convolution using convolution slicing optimization and isa extensions. ACM Trans. Arch. Code Optim. 20(4), 1\u201326 (2023)","journal-title":"ACM Trans. Arch. Code Optim."},{"key":"14_CR4","unstructured":"Hao, R., et al.: Towards effective depthwise convolutions on armv8 architecture. arXiv preprint arXiv:2206.12124 (2022)"},{"key":"14_CR5","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"14_CR6","doi-asserted-by":"crossref","unstructured":"Heide, F., Heidrich, W., Wetzstein, G.: Fast and flexible convolutional sparse coding. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5135\u20135143 (2015)","DOI":"10.1109\/CVPR.2015.7299149"},{"issue":"3","key":"14_CR7","doi-asserted-by":"publisher","first-page":"28","DOI":"10.1145\/3529113.3529122","volume":"49","author":"X Huang","year":"2022","unstructured":"Huang, X., Wang, Q., Lu, S., Hao, R., Mei, S., Liu, J.: Evaluating fft-based algorithms for strided convolutions on armv8 architectures? ACM SIGMETRICS Perf. Eval. Rev. 49(3), 28\u201329 (2022)","journal-title":"ACM SIGMETRICS Perf. Eval. Rev."},{"key":"14_CR8","unstructured":"Jin, T., et\u00a0al.: Compiling onnx neural network models using mlir. arXiv preprint arXiv:2008.08272 (2020)"},{"key":"14_CR9","unstructured":"Jocher, G., et\u00a0al.: ultralytics\/yolov5: v6. 2-yolov5 classification models, apple m1, reproducibility, clearml and deci.ai integrations. Zenodo (2022)"},{"key":"14_CR10","unstructured":"Krizhevsky, A.: One weird trick for parallelizing convolutional neural networks. arXiv preprint arXiv:1404.5997 (2014)"},{"key":"14_CR11","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: Imagenet classification with deep convolutional neural networks. Adv. Neural Inf. Process. Syst. 25 (2012)"},{"key":"14_CR12","doi-asserted-by":"crossref","unstructured":"Lattner, C., et al.: Mlir: Scaling compiler infrastructure for domain specific computation. In: 2021 IEEE\/ACM International Symposium on Code Generation and Optimization (CGO), pp. 2\u201314. IEEE (2021)","DOI":"10.1109\/CGO51591.2021.9370308"},{"key":"14_CR13","doi-asserted-by":"publisher","DOI":"10.1016\/j.parco.2022.102945","volume":"112","author":"Z Liu","year":"2022","unstructured":"Liu, Z., Xiao, X., Li, C., Ma, S., Rangyu, D.: Optimizing convolutional neural networks on multi-core vector accelerator. Parallel Comput. 112, 102945 (2022)","journal-title":"Parallel Comput."},{"issue":"2","key":"14_CR14","doi-asserted-by":"publisher","first-page":"150","DOI":"10.1007\/s42514-022-00095-y","volume":"4","author":"K Lu","year":"2022","unstructured":"Lu, K., et al.: Mt-3000: a heterogeneous multi-zone processor for hpc. CCF Trans. High Perfor. Comput. 4(2), 150\u2013164 (2022)","journal-title":"CCF Trans. High Perfor. Comput."},{"key":"14_CR15","unstructured":"Microsoft: Onnx runtime (2019). https:\/\/github.com\/microsoft\/onnxruntime"},{"key":"14_CR16","doi-asserted-by":"crossref","unstructured":"Qiu, C., Wu, J., Ren, H., Zhang, Z.: Optimization of tensor operation in compiler. In: International Conference on Communications and Networking in China, pp. 207\u2013219. Springer, Heidelberg (2022)","DOI":"10.1007\/978-3-031-34790-0_16"},{"key":"14_CR17","doi-asserted-by":"publisher","unstructured":"da\u00a0Silva, M.C., Sousa, L., Paulino, N., Bispo, J.: A dsl and mlir dialect for streaming and vectorisation. In: International Symposium on Applied Reconfigurable Computing, pp. 181\u2013190. Springer, Heidelberg (2024). https:\/\/doi.org\/10.1007\/978-3-031-55673-9_13","DOI":"10.1007\/978-3-031-55673-9_13"},{"key":"14_CR18","unstructured":"Simonyan, K., Zisserman, A.: Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556 (2014)"},{"key":"14_CR19","doi-asserted-by":"crossref","unstructured":"Szegedy, C., et al.: Going deeper with convolutions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp.\u00a01\u20139 (2015)","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"14_CR20","doi-asserted-by":"crossref","unstructured":"Wang, Q., Mei, S., Liu, J., Gong, C.: Parallel convolution algorithm using implicit matrix multiplication on multi-core cpus. In: 2019 International Joint Conference on Neural Networks (ijcnn), pp.\u00a01\u20137. IEEE (2019)","DOI":"10.1109\/IJCNN.2019.8852012"},{"key":"14_CR21","unstructured":"Xu, J., et al.: Parallel optimization of convolution algorithm for multi-core digital signal processing. J. Natl. Univ. Defense Technol.\/Guofang Keji Daxue Xuebao 46(1) (2024)"},{"key":"14_CR22","doi-asserted-by":"publisher","first-page":"0040","DOI":"10.34133\/icomputing.0040","volume":"2","author":"H Zhang","year":"2023","unstructured":"Zhang, H., Xing, M., Wu, Y., Zhao, C.: Compiler technologies in deep learning co-design: a survey. Intell. Comput. 2, 0040 (2023)","journal-title":"Intell. Comput."},{"key":"14_CR23","doi-asserted-by":"crossref","unstructured":"Zhong, H., Liu, Z.: Long-life sensitive modulo scheduling with adaptive loop expansion. In: 2022 IEEE 28th International Conference on Parallel and Distributed Systems (ICPADS), pp. 530\u2013537. IEEE (2023)","DOI":"10.1109\/ICPADS56603.2022.00075"}],"container-title":["Lecture Notes in Computer Science","Network and Parallel Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-96-2830-8_14","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,6]],"date-time":"2025-09-06T08:33:11Z","timestamp":1757147591000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-96-2830-8_14"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9789819628292","9789819628308"],"references-count":23,"URL":"https:\/\/doi.org\/10.1007\/978-981-96-2830-8_14","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"29 March 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"NPC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"IFIP International Conference on Network and Parallel Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Haikou","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"7 December 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"8 December 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"npc2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}