{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T23:50:48Z","timestamp":1783036248487,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":33,"publisher":"ACM","license":[{"start":{"date-parts":[[2019,2,16]],"date-time":"2019-02-16T00:00:00Z","timestamp":1550275200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100011002","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61672048 and 61520106004"],"award-info":[{"award-number":["61672048 and 61520106004"]}],"id":[{"id":"10.13039\/501100011002","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2019,2,16]]},"DOI":"10.1145\/3293883.3295734","type":"proceedings-article","created":{"date-parts":[[2019,2,5]],"date-time":"2019-02-05T20:44:12Z","timestamp":1549399452000},"page":"229-241","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":69,"title":["A coordinated tiling and batching framework for efficient GEMM on GPUs"],"prefix":"10.1145","author":[{"given":"Xiuhong","family":"Li","sequence":"first","affiliation":[{"name":"Peking University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yun","family":"Liang","sequence":"additional","affiliation":[{"name":"Peking University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shengen","family":"Yan","sequence":"additional","affiliation":[{"name":"SenseTime Incorporation"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Liancheng","family":"Jia","sequence":"additional","affiliation":[{"name":"Peking University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yinghan","family":"Li","sequence":"additional","affiliation":[{"name":"SenseTime Incorporation"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2019,2,16]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"crossref","unstructured":"Ahmad Abdelfattah Azzam Haidar Stanimire Tomov and Jack Dongarra. 2016. Performance Design and Autotuning of Batched GEMM for GPUs. In High Performance Computing. 21--38.  Ahmad Abdelfattah Azzam Haidar Stanimire Tomov and Jack Dongarra. 2016. Performance Design and Autotuning of Batched GEMM for GPUs. In High Performance Computing . 21--38.","DOI":"10.1007\/978-3-319-41321-1_2"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/3079079.3079103"},{"key":"e_1_3_2_1_3_1","volume-title":"cuDNN: Efficient Primitives for Deep Learning. ArXiv e-prints","author":"Chetlur Sharan","year":"2014","unstructured":"Sharan Chetlur , Cliff Woolley , Philippe Vandermersch , Jonathan Cohen , John Tran , Bryan Catanzaro , and Evan Shelhamer . 2014. cuDNN: Efficient Primitives for Deep Learning. ArXiv e-prints ( 2014 ). Sharan Chetlur, Cliff Woolley, Philippe Vandermersch, Jonathan Cohen, John Tran, Bryan Catanzaro, and Evan Shelhamer. 2014. cuDNN: Efficient Primitives for Deep Learning. ArXiv e-prints (2014)."},{"key":"e_1_3_2_1_4_1","volume-title":"CUSOLVER and MAGMA by example. Version","author":"Chrzeszczyk Andrzej","year":"2017","unstructured":"Andrzej Chrzeszczyk . 2017. Matrix computations on the GPU. CUBLAS , CUSOLVER and MAGMA by example. Version 2017 . Andrzej Chrzeszczyk. 2017. Matrix computations on the GPU. CUBLAS, CUSOLVER and MAGMA by example. Version 2017."},{"key":"e_1_3_2_1_5_1","unstructured":"Scott Gray. 2017. A full walk through of the SGEMM implementation. https:\/\/github.com\/NervanaSystems\/maxas\/wiki\/SGEMM. (2017).  Scott Gray. 2017. A full walk through of the SGEMM implementation. https:\/\/github.com\/NervanaSystems\/maxas\/wiki\/SGEMM. (2017)."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/InPar.2012.6339596"},{"key":"e_1_3_2_1_7_1","volume-title":"Deep Residual Learning for Image Recognition. In 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR). 770--778","author":"He Kaiming","year":"2016","unstructured":"Kaiming He , Xiangyu Zhang , Shaoqing Ren , and Jian Sun . 2016 . Deep Residual Learning for Image Recognition. In 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR). 770--778 . Kaiming He, Xiangyu Zhang, Shaoqing Ren, and Jian Sun. 2016. Deep Residual Learning for Image Recognition. In 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR). 770--778."},{"key":"e_1_3_2_1_8_1","unstructured":"Forrest N. Iandola Song Han Matthew W. Moskewicz Khalid Ashraf William J. Dally and Kurt Keutzer. {n. d.}. SqueezeNet: AlexNet-level accuracy with 50x fewer parameters and &lt;0.5MB model size. arXiv e-prints ({n. d.}) arXiv:1602.07360.  Forrest N. Iandola Song Han Matthew W. Moskewicz Khalid Ashraf William J. Dally and Kurt Keutzer. {n. d.}. SqueezeNet: AlexNet-level accuracy with 50x fewer parameters and &lt;0.5MB model size. arXiv e-prints ({n. d.}) arXiv:1602.07360."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/PACT.2005.10"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2011.311"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CGO.2013.6494986"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3037697.3037709"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/2751205.2751232"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.5555\/2971808.2971827"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3205289.3205309"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-01970-8_89"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2014.2313342"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/3070710"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.5555\/3199700.3199702"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1177\/1094342010385729"},{"key":"e_1_3_2_1_21_1","unstructured":"NVIDIA. 2018. CUDA Documentation. http:\/\/docs.nvidia.com\/cuda\/cublas\/index.html. (2018).  NVIDIA. 2018. CUDA Documentation. http:\/\/docs.nvidia.com\/cuda\/cublas\/index.html. (2018)."},{"key":"e_1_3_2_1_22_1","volume-title":"CUTLASS: Fast Linear Algebra in CUDA C++. https:\/\/devblogs.nvidia.com\/cutlass-linear-algebra-cuda\/.","author":"NVIDIA.","year":"2018","unstructured":"NVIDIA. 2018 . CUTLASS: Fast Linear Algebra in CUDA C++. https:\/\/devblogs.nvidia.com\/cutlass-linear-algebra-cuda\/. (2018). NVIDIA. 2018. CUTLASS: Fast Linear Algebra in CUDA C++. https:\/\/devblogs.nvidia.com\/cutlass-linear-algebra-cuda\/. (2018)."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/2967938.2967967"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3178487.3178500"},{"key":"e_1_3_2_1_25_1","volume-title":"Going Deeper with Convolutions. CoRR abs\/1409.4842","author":"Szegedy Christian","year":"2014","unstructured":"Christian Szegedy , Wei Liu , Yangqing Jia , Pierre Sermanet , Scott E. Reed , Dragomir Anguelov , Dumitru Erhan , Vincent Vanhoucke , and Andrew Rabinovich . 2014. Going Deeper with Convolutions. CoRR abs\/1409.4842 ( 2014 ). Christian Szegedy, Wei Liu, Yangqing Jia, Pierre Sermanet, Scott E. Reed, Dragomir Anguelov, Dumitru Erhan, Vincent Vanhoucke, and Andrew Rabinovich. 2014. Going Deeper with Convolutions. CoRR abs\/1409.4842 (2014)."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/2063384.2063431"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/2830772.2830813"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/TC.2017.2776272"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.5555\/2561828.2561929"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA.2015.7056023"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3018743.3018755"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/3123939.3123978"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3079079.3079083"}],"event":{"name":"PPoPP '19: 24th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming","location":"Washington District of Columbia","acronym":"PPoPP '19","sponsor":["SIGPLAN ACM Special Interest Group on Programming Languages","SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing"]},"container-title":["Proceedings of the 24th Symposium on Principles and Practice of Parallel Programming"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3293883.3295734","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3293883.3295734","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T01:01:47Z","timestamp":1750208507000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3293883.3295734"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,2,16]]},"references-count":33,"alternative-id":["10.1145\/3293883.3295734","10.1145\/3293883"],"URL":"https:\/\/doi.org\/10.1145\/3293883.3295734","relation":{},"subject":[],"published":{"date-parts":[[2019,2,16]]},"assertion":[{"value":"2019-02-16","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}