{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,27]],"date-time":"2026-02-27T03:47:55Z","timestamp":1772164075955,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":50,"publisher":"ACM","license":[{"start":{"date-parts":[[2017,6,2]],"date-time":"2017-06-02T00:00:00Z","timestamp":1496361600000},"content-version":"vor","delay-in-days":365,"URL":"http:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/100000185","name":"Defense Advanced Research Projects Agency","doi-asserted-by":"publisher","award":["HR0011-12-2-0016"],"award-info":[{"award-number":["HR0011-12-2-0016"]}],"id":[{"id":"10.13039\/100000185","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2016,6,2]]},"DOI":"10.1145\/2908080.2908105","type":"proceedings-article","created":{"date-parts":[[2016,6,2]],"date-time":"2016-06-02T15:23:42Z","timestamp":1464881022000},"page":"209-223","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":30,"title":["Latte: a language, compiler, and runtime for elegant and efficient deep neural networks"],"prefix":"10.1145","author":[{"given":"Leonard","family":"Truong","sequence":"first","affiliation":[{"name":"Intel Labs, USA \/ University of California at Berkeley, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rajkishore","family":"Barik","sequence":"additional","affiliation":[{"name":"Intel Labs, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ehsan","family":"Totoni","sequence":"additional","affiliation":[{"name":"Intel Labs, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hai","family":"Liu","sequence":"additional","affiliation":[{"name":"Intel Labs, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chick","family":"Markley","sequence":"additional","affiliation":[{"name":"University of California at Berkeley, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Armando","family":"Fox","sequence":"additional","affiliation":[{"name":"University of California at Berkeley, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tatiana","family":"Shpeisman","sequence":"additional","affiliation":[{"name":"Intel Labs, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2016,6,2]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Effective Use of the Intel Compiler\u2019s Offload Features. URL https:\/\/software.intel.com\/en-us\/articles\/ effective-use-of-the-intel-compilers-offloadfeatures."},{"key":"e_1_3_2_1_2_1","first-page":"630813","author":"Intel Math Kernel Library. Reference Manual. Intel Corporation","year":"2009","unstructured":"Intel Math Kernel Library. Reference Manual. Intel Corporation, Santa Clara, USA, 2009. ISBN 630813-054US.","journal-title":"USA"},{"key":"e_1_3_2_1_3_1","volume-title":"https:\/\/github.com\/torch\/nn","author":"Torch","year":"2015","unstructured":"Torch NN. https:\/\/github.com\/torch\/nn, 2015."},{"key":"e_1_3_2_1_4_1","volume-title":"TensorFlow: Large-scale machine learning on heterogeneous systems","author":"Abadi M.","year":"2015","unstructured":"M. Abadi, A. Agarwal, P. Barham, E. Brevdo, Z. Chen, C. Citro, G. S. Corrado, A. Davis, J. Dean, M. Devin, S. Ghemawat, I. Goodfellow, A. Harp, G. Irving, M. Isard, Y. Jia, R. Jozefowicz, L. Kaiser, M. Kudlur, J. Levenberg, D. Man\u00e9, R. Monga, S. Moore, D. Murray, C. Olah, M. Schuster, J. Shlens, B. Steiner, I. Sutskever, K. Talwar, P. Tucker, V. Vanhoucke, V. Vasudevan, F. Vi\u00e9gas, O. Vinyals, P. Warden, M. Wattenberg, M. Wicke, Y. Yu, and X. Zheng. TensorFlow: Large-scale machine learning on heterogeneous systems, 2015. URL http:\/\/tensorflow.org\/."},{"key":"e_1_3_2_1_5_1","unstructured":"A. Agarwal E. Akchurin C. Basoglu G. Chen S. Cyphers J. Droppo A. Eversole B. Guenter M. Hillebrand R. Hoens X. Huang Z. Huang V. Ivanov A. Kamenev P. Kranen O. Kuchaiev W. Manousek A. May B. Mitra O. Nano G. Navarro A. Orlov M. Padmilac H. Parthasarathi B. Peng A. Reznichenko F. Seide M. L. Seltzer M. Slaney A. Stolcke Y. Wang H. Wang K. Yao D. Yu Y. Zhang and G. Zweig. An introduction to computational networks and the computational network toolkit. Technical Report MSR-TR-2014-112 August 2014. URL http:\/\/research. microsoft.com\/apps\/pubs\/default.aspx?id=226641."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1016\/0925-"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/2688500.2688521"},{"key":"e_1_3_2_1_8_1","first-page":"7","article-title":"Nengo: a Python tool for building large-scale functional brain models","author":"Bekolay T.","year":"2013","unstructured":"T. Bekolay, J. Bergstra, E. Hunsberger, T. DeWolf, T. C. Stewart, D. Rasmussen, X. Choo, A. R. Voelker, and C. Eliasmith. Nengo: a Python tool for building large-scale functional brain models. Frontiers in Neuroinformatics, 7, 2013.","journal-title":"Frontiers in Neuroinformatics"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/1654059.1654119"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.25080\/Majora-92bf1922-003"},{"key":"e_1_3_2_1_11_1","volume-title":"Julia: A Fast Dynamic Language for Technical Computing. CoRR, abs\/1209.5145","author":"Bezanson J.","year":"2012","unstructured":"J. Bezanson, S. Karpinski, V. B. Shah, and A. Edelman. Julia: A Fast Dynamic Language for Technical Computing. CoRR, abs\/1209.5145, 2012. URL http:\/\/arxiv.org\/abs\/1209."},{"key":"e_1_3_2_1_12_1","unstructured":"5145."},{"key":"e_1_3_2_1_13_1","unstructured":"B. Catanzaro S. Kamil Y. Lee J. Demmel K. Keutzer J. Shalf K. Yelick and A. Fox. SEJITS: Getting productivity and performance with selective embedded JIT specialization."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/2038037.1941561"},{"key":"e_1_3_2_1_15_1","volume-title":"cuDNN: Efficient Primitives for Deep Learning. CoRR, abs\/1410.0759","author":"Chetlur S.","year":"2014","unstructured":"S. Chetlur, C. Woolley, P. Vandermersch, J. Cohen, J. Tran, B. Catanzaro, and E. Shelhamer. cuDNN: Efficient Primitives for Deep Learning. CoRR, abs\/1410.0759, 2014. URL http: \/\/arxiv.org\/abs\/1410.0759."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.5555\/2685048.2685094"},{"key":"e_1_3_2_1_17_1","volume-title":"https:\/\/github.com\/ soumith\/convnet-benchmarks","author":"Chintala S.","year":"2015","unstructured":"S. Chintala. Convnet Benchmarks. https:\/\/github.com\/ soumith\/convnet-benchmarks, 2015."},{"key":"e_1_3_2_1_18_1","volume-title":"NIPS Workshop, number EPFL-CONF-192376","author":"Collobert R.","year":"2011","unstructured":"R. Collobert, K. Kavukcuoglu, and C. Farabet. Torch7: A MATLAB-like environment for machine learning. In BigLearn, NIPS Workshop, number EPFL-CONF-192376, 2011."},{"key":"e_1_3_2_1_19_1","volume-title":"Distributed deep learning using synchronous stochastic gradient descent. CoRR, abs\/1602.06709","author":"Das D.","year":"2016","unstructured":"D. Das, S. Avancha, D. Mudigere, K. Vaidyanathan, S. Sridharan, D. D. Kalamkar, B. Kaul, and P. Dubey. Distributed deep learning using synchronous stochastic gradient descent. CoRR, abs\/1602.06709, 2016. URL http:\/\/arxiv.org\/ abs\/1602.06709."},{"key":"e_1_3_2_1_20_1","first-page":"1231","volume-title":"Advances in Neural Information Processing Systems","author":"Dean J.","year":"2012","unstructured":"J. Dean, G. Corrado, R. Monga, K. Chen, M. Devin, M. Mao, A. Senior, P. Tucker, K. Yang, Q. V. Le, et al. Large scale distributed deep networks. In Advances in Neural Information Processing Systems, pages 1223\u20131231, 2012."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.5555\/1953048.2021068"},{"key":"e_1_3_2_1_22_1","volume-title":"https:\/\/github.com\/ Maratyszcza\/NNPACK","author":"Dukhan M.","year":"2016","unstructured":"M. Dukhan. NNPACK. https:\/\/github.com\/ Maratyszcza\/NNPACK, 2016."},{"key":"e_1_3_2_1_23_1","unstructured":"F. Gers. Long short-term memory in recurrent neural networks."},{"key":"e_1_3_2_1_24_1","first-page":"256","volume-title":"International Conference on Artificial Intelligence and Statistics","author":"Glorot X.","year":"2010","unstructured":"X. Glorot and Y. Bengio. Understanding the difficulty of training deep feedforward neural networks. In International Conference on Artificial Intelligence and Statistics, pages 249\u2013 256, 2010."},{"key":"e_1_3_2_1_25_1","volume-title":"Maxout networks. arXiv preprint arXiv:1302.4389","author":"Goodfellow I. J.","year":"2013","unstructured":"I. J. Goodfellow, D. Warde-Farley, M. Mirza, A. Courville, and Y. Bengio. Maxout networks. arXiv preprint arXiv:1302.4389, 2013."},{"key":"e_1_3_2_1_26_1","first-page":"6649","volume-title":"Speech and Signal Processing (ICASSP), 2013 IEEE International Conference on","author":"Graves A.","unstructured":"A. Graves, A.-r. Mohamed, and G. Hinton. Speech recognition with deep recurrent neural networks. In Acoustics, Speech and Signal Processing (ICASSP), 2013 IEEE International Conference on, pages 6645\u20136649. IEEE, 2013."},{"key":"e_1_3_2_1_27_1","volume-title":"Delving deep into rectifiers: Surpassing human-level performance on imagenet classification. CoRR, abs\/1502.01852","author":"He K.","year":"2015","unstructured":"K. He, X. Zhang, S. Ren, and J. Sun. Delving deep into rectifiers: Surpassing human-level performance on imagenet classification. CoRR, abs\/1502.01852, 2015. URL http: \/\/arxiv.org\/abs\/1502.01852."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1016\/0893-6080(89)90020-8"},{"key":"e_1_3_2_1_30_1","volume-title":"Intel Data Analytics Acceleration Library (DAAL). https:\/\/software.intel.com\/en-us\/intel-daal","year":"2015","unstructured":"Intel. Intel Data Analytics Acceleration Library (DAAL). https:\/\/software.intel.com\/en-us\/intel-daal, 2015."},{"key":"e_1_3_2_1_31_1","volume-title":"https:\/\/github.com\/ IntelLabs\/ParallelAccelerator.jl","author":"Labs Intel","year":"2015","unstructured":"Intel Labs. ParallelAccelerator.jl. https:\/\/github.com\/ IntelLabs\/ParallelAccelerator.jl, 2015."},{"key":"e_1_3_2_1_32_1","volume-title":"Batch Normalization: Accelerating Deep Network Training by Reducing Internal Covariate Shift. CoRR, abs\/1502.03167","author":"Ioffe S.","year":"2015","unstructured":"S. Ioffe and C. Szegedy. Batch Normalization: Accelerating Deep Network Training by Reducing Internal Covariate Shift. CoRR, abs\/1502.03167, 2015. URL http:\/\/arxiv. org\/abs\/1502.03167."},{"key":"e_1_3_2_1_33_1","volume-title":"Caffe: Convolutional Architecture for Fast Feature Embedding. arXiv preprint arXiv:1408.5093","author":"Jia Y.","year":"2014","unstructured":"Y. Jia, E. Shelhamer, J. Donahue, S. Karayev, J. Long, R. Girshick, S. Guadarrama, and T. Darrell. Caffe: Convolutional Architecture for Fast Feature Embedding. arXiv preprint arXiv:1408.5093, 2014."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.5555\/502981"},{"key":"e_1_3_2_1_35_1","unstructured":"ISBN 1-55860-286-0."},{"key":"e_1_3_2_1_36_1","volume-title":"One weird trick for parallelizing convolutional neural networks. CoRR, abs\/1404.5997","author":"Krizhevsky A.","year":"2014","unstructured":"A. Krizhevsky. One weird trick for parallelizing convolutional neural networks. CoRR, abs\/1404.5997, 2014. URL http: \/\/arxiv.org\/abs\/1404.5997."},{"key":"e_1_3_2_1_37_1","volume-title":"https:\/\/github.com\/ akrizhevsky\/cuda-convnet2","author":"Krizhevsky A.","year":"2015","unstructured":"A. Krizhevsky. cuda-convnet2. https:\/\/github.com\/ akrizhevsky\/cuda-convnet2, 2015."},{"key":"e_1_3_2_1_38_1","first-page":"1105","volume-title":"Advances in Neural Information Processing Systems","author":"Krizhevsky A.","year":"2012","unstructured":"A. Krizhevsky, I. Sutskever, and G. E. Hinton. Imagenet classification with deep convolutional neural networks. In Advances in Neural Information Processing Systems, pages 1097\u20131105, 2012."},{"key":"e_1_3_2_1_39_1","volume-title":"https:\/\/ github.com\/Microsoft\/CNTK\/tree\/ 7d3e84e7733c1c965d995e28ff4bac60f166a03b","author":"CNTK.","year":"2015","unstructured":"Microsoft. CNTK. https:\/\/ github.com\/Microsoft\/CNTK\/tree\/ 7d3e84e7733c1c965d995e28ff4bac60f166a03b, 2015."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/2499370.2462176"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0816-y"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0816-y"},{"key":"e_1_3_2_1_43_1","volume-title":"OverFeat: Integrated Recognition, Localization and Detection using Convolutional Networks. CoRR, abs\/1312.6229","author":"Sermanet P.","year":"2013","unstructured":"P. Sermanet, D. Eigen, X. Zhang, M. Mathieu, R. Fergus, and Y. LeCun. OverFeat: Integrated Recognition, Localization and Detection using Convolutional Networks. CoRR, abs\/1312.6229, 2013. URL http:\/\/arxiv.org\/abs\/1312."},{"key":"e_1_3_2_1_44_1","unstructured":"6229."},{"key":"e_1_3_2_1_45_1","volume-title":"Very Deep Convolutional Networks for Large-Scale Image Recognition. CoRR, abs\/1409.1556","author":"Simonyan K.","year":"2014","unstructured":"K. Simonyan and A. Zisserman. Very Deep Convolutional Networks for Large-Scale Image Recognition. CoRR, abs\/1409.1556, 2014."},{"key":"e_1_3_2_1_46_1","unstructured":"Skymind. Deep Learning for Java (DL4J). http:\/\/ deeplearning4j.org\/architecture.html 2015."},{"key":"e_1_3_2_1_47_1","volume-title":"Going deeper with convolutions. CoRR, abs\/1409.4842","author":"Szegedy C.","year":"2014","unstructured":"C. Szegedy, W. Liu, Y. Jia, P. Sermanet, S. Reed, D. Anguelov, D. Erhan, V. Vanhoucke, and A. Rabinovich. Going deeper with convolutions. CoRR, abs\/1409.4842, 2014. URL http: \/\/arxiv.org\/abs\/1409.4842."},{"key":"e_1_3_2_1_48_1","volume-title":"Lecture 6.5-rmsprop","author":"Tieleman T.","year":"2012","unstructured":"T. Tieleman and G. Hinton. Lecture 6.5-rmsprop. COURSERA: Neural Networks for Machine Learning, 2012."},{"key":"e_1_3_2_1_49_1","volume-title":"Deep Image: Scaling up Image Recognition. CoRR, abs\/1501.02876","author":"Wu R.","year":"2015","unstructured":"R. Wu, S. Yan, Y. Shan, Q. Dang, and G. Sun. Deep Image: Scaling up Image Recognition. CoRR, abs\/1501.02876, 2015. URL http:\/\/arxiv.org\/abs\/1501.02876."},{"key":"e_1_3_2_1_50_1","volume-title":"https:\/\/github.com\/pluskid\/ Mocha.jl","author":"Zhang C.","year":"2015","unstructured":"C. Zhang. Mocha.jl. https:\/\/github.com\/pluskid\/ Mocha.jl, 2015."}],"event":{"name":"PLDI '16: ACM SIGPLAN Conference on Programming Language Design and Implementation","location":"Santa Barbara CA USA","acronym":"PLDI '16","sponsor":["SIGPLAN ACM Special Interest Group on Programming Languages"]},"container-title":["Proceedings of the 37th ACM SIGPLAN Conference on Programming Language Design and Implementation"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/2908080.2908105","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/2908080.2908105","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/2908080.2908105","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,18]],"date-time":"2025-11-18T09:19:04Z","timestamp":1763457544000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/2908080.2908105"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2016,6,2]]},"references-count":50,"alternative-id":["10.1145\/2908080.2908105","10.1145\/2908080"],"URL":"https:\/\/doi.org\/10.1145\/2908080.2908105","relation":{"is-identical-to":[{"id-type":"doi","id":"10.1145\/2980983.2908105","asserted-by":"object"}]},"subject":[],"published":{"date-parts":[[2016,6,2]]},"assertion":[{"value":"2016-06-02","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}