{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,20]],"date-time":"2026-06-20T01:10:33Z","timestamp":1781917833671,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":28,"publisher":"ACM","license":[{"start":{"date-parts":[[2017,6,18]],"date-time":"2017-06-18T00:00:00Z","timestamp":1497744000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"DFG","award":["GSC 111"],"award-info":[{"award-number":["GSC 111"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2017,6,18]]},"DOI":"10.1145\/3091966.3091968","type":"proceedings-article","created":{"date-parts":[[2017,6,9]],"date-time":"2017-06-09T17:40:22Z","timestamp":1497030022000},"page":"56-62","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":37,"title":["HPTT: a high-performance tensor transposition C++ library"],"prefix":"10.1145","author":[{"given":"Paul","family":"Springer","sequence":"first","affiliation":[{"name":"RWTH Aachen University, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tong","family":"Su","sequence":"additional","affiliation":[{"name":"RWTH Aachen University, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Paolo","family":"Bientinesi","sequence":"additional","affiliation":[{"name":"RWTH Aachen University, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2017,6,18]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"M. Abadi A. Agarwal P. Barham E. Brevdo Z. Chen C. Citro G. S. Corrado A. Davis J. Dean M. Devin et al. Tensorflow: Large-scale machine learning on heterogeneous systems. 2015. M. Abadi A. Agarwal P. Barham E. Brevdo Z. Chen C. Citro G. S. Corrado A. Davis J. Dean M. Devin et al. Tensorflow: Large-scale machine learning on heterogeneous systems. 2015."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1103\/RevModPhys.79.291"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"crossref","unstructured":"S. Chatterjee and S. Sen. Cache-efficient matrix transposition. pages 195\u2013205 2000. S. Chatterjee and S. Sen. Cache-efficient matrix transposition. pages 195\u2013205 2000.","DOI":"10.1109\/HPCA.2000.824350"},{"key":"e_1_3_2_1_4_1","unstructured":"7 Available at www.github.com\/springer13\/hptt. 7 Available at www.github.com\/springer13\/hptt."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1016\/0167-8191(96)80001-9"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2004.840301"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/TSE.1981.234523"},{"key":"e_1_3_2_1_8_1","unstructured":"G. Guennebaud B. Jacob etal Eigen v3. http:\/\/eigen.tuxfamily.org 2010. G. Guennebaud B. Jacob et al. Eigen v3. http:\/\/eigen.tuxfamily.org 2010."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"crossref","unstructured":"R. J. Harrison G. Beylkin F. A. Bischoff J. A. Calvin G. I. Fann J. Fosso-Tande D. Galindo J. R. Hammond R. Hartman-Baker J. C. Hill J. Jia J. S. Kottmann M. Y. Ou L. E. Ratcliff M. G. Reuter A. C. Richie-Halford N. A. Romero H. Sekino W. A. Shelton B. E. Sundahl W. S. Thornton E. F. Valeev \u00c1. V\u00e1zquez-Mayagoitia N. Vence and Y. Yokoi. MADNESS: A multiresolution adaptive numerical environment for scientific simulation. CoRR abs\/1507.01888 2015. R. J. Harrison G. Beylkin F. A. Bischoff J. A. Calvin G. I. Fann J. Fosso-Tande D. Galindo J. R. Hammond R. Hartman-Baker J. C. Hill J. Jia J. S. Kottmann M. Y. Ou L. E. Ratcliff M. G. Reuter A. C. Richie-Halford N. A. Romero H. Sekino W. A. Shelton B. E. Sundahl W. S. Thornton E. F. Valeev \u00c1. V\u00e1zquez-Mayagoitia N. Vence and Y. Yokoi. MADNESS: A multiresolution adaptive numerical environment for scientific simulation. CoRR abs\/1507.01888 2015.","DOI":"10.1137\/15M1026171"},{"key":"e_1_3_2_1_10_1","unstructured":"A. Hynninen and D. I. Lyakh. cuTT: A High-Performance Tensor Transpose Library for CUDA Compatible GPUs. CoRR abs\/1705.01598 2017. 01598. A. Hynninen and D. I. Lyakh. cuTT: A High-Performance Tensor Transpose Library for CUDA Compatible GPUs. CoRR abs\/1705.01598 2017. 01598."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10766-015-0366-5"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/1152154.1152190"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.cpc.2014.12.013"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/2381056.2381073"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/1122018.1122054"},{"key":"e_1_3_2_1_16_1","first-page":"25","volume-title":"IEEE Computer Society Technical Committee on Computer Architecture (TCCA) Newsletter","author":"McCalpin J. D.","year":"1995"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1137\/11082748X"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1016\/S0009-2614(89)87395-6"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2013.112"},{"key":"e_1_3_2_1_20_1","volume-title":"CoRR","author":"Springer P.","year":"2016"},{"key":"e_1_3_2_1_21_1","volume-title":"CoRR","author":"Springer P.","year":"2016"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/2935323.2935328"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1016\/0304-3991(91)90109-J"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/2764454"},{"key":"e_1_3_2_1_25_1","unstructured":"N. Vasilache J. Johnson M. Mathieu S. Chintala S. Piantino and Y. LeCun. Fast convolutional nets with fbfft: A gpu performance evaluation 2014. N. Vasilache J. Johnson M. Mathieu S. Chintala S. Piantino and Y. LeCun. Fast convolutional nets with fbfft: A gpu performance evaluation 2014."},{"key":"e_1_3_2_1_26_1","unstructured":"A. Vladimirov. Multithreaded transposition of square matrices with common code for Intel Xeon processors and Intel Xeon Phi coprocessors 2013. A. Vladimirov. Multithreaded transposition of square matrices with common code for Intel Xeon processors and Intel Xeon Phi coprocessors 2013."},{"key":"e_1_3_2_1_27_1","unstructured":"pdf. pdf."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPSW.2014.43"}],"event":{"name":"PLDI '17: ACM SIGPLAN Conference on Programming Language Design and Implementation","location":"Barcelona Spain","acronym":"PLDI '17","sponsor":["SIGPLAN ACM Special Interest Group on Programming Languages"]},"container-title":["Proceedings of the 4th ACM SIGPLAN International Workshop on Libraries, Languages, and Compilers for Array Programming"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3091966.3091968","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3091966.3091968","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T02:54:31Z","timestamp":1750301671000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3091966.3091968"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017,6,18]]},"references-count":28,"alternative-id":["10.1145\/3091966.3091968","10.1145\/3091966"],"URL":"https:\/\/doi.org\/10.1145\/3091966.3091968","relation":{},"subject":[],"published":{"date-parts":[[2017,6,18]]},"assertion":[{"value":"2017-06-18","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}