{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,17]],"date-time":"2025-10-17T14:07:42Z","timestamp":1760710062668,"version":"3.37.3"},"reference-count":44,"publisher":"Springer Science and Business Media LLC","issue":"3-4","license":[{"start":{"date-parts":[[2019,11,29]],"date-time":"2019-11-29T00:00:00Z","timestamp":1574985600000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2019,11,29]],"date-time":"2019-11-29T00:00:00Z","timestamp":1574985600000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["CCF#1822987"],"award-info":[{"award-number":["CCF#1822987"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000015","name":"U.S. Department of Energy","doi-asserted-by":"publisher","award":["DE-AC02-05CH11231"],"award-info":[{"award-number":["DE-AC02-05CH11231"]}],"id":[{"id":"10.13039\/100000015","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["CCF Trans. HPC"],"published-print":{"date-parts":[[2019,12]]},"DOI":"10.1007\/s42514-019-00018-4","type":"journal-article","created":{"date-parts":[[2019,11,29]],"date-time":"2019-11-29T10:02:34Z","timestamp":1575021754000},"page":"224-239","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":7,"title":["Performance analysis of deep learning workloads using roofline trajectories"],"prefix":"10.1007","volume":"1","author":[{"given":"M. Haseeb","family":"Javed","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Khaled Z.","family":"Ibrahim","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaoyi","family":"Lu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2019,11,29]]},"reference":[{"key":"18_CR6","unstructured":"Abadi, M., Barham, P., Chen, J., Chen, Z., Davis, A., Dean, J., Devin, M., Ghemawat, S., Irving, G., Isard, M. et al.: Tensorflow: a system for large-scale machine learning. In: 12th $$\\{$$USENIX$$\\}$$ Symposium on Operating Systems Design and Implementation ($$\\{$$OSDI$$\\}$$ 16), pp. 265\u2013283 (2016)"},{"key":"18_CR7","unstructured":"Bahrampour, S., Ramakrishnan, N., Schott, L., Shah, M.: Comparative study of caffe, neon, theano, and torch for deep learning (2016)"},{"key":"18_CR8","doi-asserted-by":"crossref","unstructured":"Biswas, R., Lu, X., Panda, D.K.: Accelerating TensorFlow with adaptive RDMA-based gRPC. In: 2018 IEEE 25th International Conference on High Performance Computing (HiPC). IEEE, pp. 2\u201311 (2018)","DOI":"10.1109\/HiPC.2018.00010"},{"key":"18_CR10","unstructured":"Chen, J., Pan, X., Monga, R., Bengio, S., Jozefowicz, R.: Revisiting distributed synchronous SGD (2016). arXiv preprint arXiv:1604.00981"},{"issue":"5","key":"18_CR9","doi-asserted-by":"publisher","first-page":"965","DOI":"10.1109\/TPDS.2018.2877359","volume":"30","author":"J Chen","year":"2018","unstructured":"Chen, J., Li, K., Bilal, K., Li, K., Philip, S.Y., et al.: A bi-layered parallel training architecture for large-scale convolutional neural networks. IEEE Trans. Parallel Distrib. Syst. 30(5), 965\u2013976 (2018)","journal-title":"IEEE Trans. Parallel Distrib. Syst."},{"key":"18_CR11","unstructured":"Chetlur, S., Woolley, C., Vandermersch, P., Cohen, J., Tran, J., Catanzaro, B., Shelhamer, E.: cuDNN: Efficient primitives for deep learning, CoRR, vol. abs\/1410.0759 (2014). [Online]. http:\/\/arxiv.org\/abs\/1410.0759"},{"key":"18_CR12","unstructured":"De Melo, A.C.: The new Linux perf tool. In: Slides from Linux Kongress, vol. 18 (2010)"},{"key":"18_CR13","unstructured":"Dean, J., Corrado, G., Monga, R., Chen, K., Devin, M., Mao, M., Senior, A., Tucker, P., Yang, K., Le, Q.V. et al.: Large scale distributed deep networks. In: Advances in Neural Information Processing Systems, pp. 1223\u20131231 (2012)"},{"key":"18_CR14","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.-J., Li, K., Fei-Fei, L.: ImageNet: a large-scale hierarchical image database. In: CVPR09 (2009)","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"18_CR15","doi-asserted-by":"crossref","unstructured":"Doerfler, D., Deslippe, J., Williams, S., Oliker, L., Cook, B., Kurth, T., Lobet, M., Malas, T., Vay, J.-L., Vincenti, H.: Applying the roofline performance model to the intel xeon phi knights landing processor. In: International Conference on High Performance Computing. Springer, pp. 339\u2013353 (2016)","DOI":"10.1007\/978-3-319-46079-6_24"},{"issue":"7639","key":"18_CR16","doi-asserted-by":"publisher","first-page":"115","DOI":"10.1038\/nature21056","volume":"542","author":"A Esteva","year":"2017","unstructured":"Esteva, A., Kuprel, B., Novoa, R.A., Ko, J., Swetter, S.M., Blau, H.M., Thrun, S.: Dermatologist-level Classification of skin cancer with deep neural networks. Nature 542(7639), 115 (2017)","journal-title":"Nature"},{"key":"18_CR1","unstructured":"gRPC (2019). http:\/\/grpc.io\/"},{"key":"18_CR17","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"18_CR18","unstructured":"Huval, B., Wang, T., Tandon, S., Kiske, J., Song, W., Pazhayampallil, J., Andriluka, M., Rajpurkar, P., Migimatsu, T., Cheng-Yue, R. et al.: An empirical evaluation of deep learning on highway driving. arXiv preprint arXiv:1504.01716, (2015)"},{"key":"18_CR19","unstructured":"Ibrahim, K., Williams, S., Oliker, L.: Performance analysis of GPU programming models using the roofline scaling trajectories. In: 2018 International Symposium on Benchmarking, Measuring and Optimizing (Bench\u201919) (2018a)"},{"key":"18_CR20","doi-asserted-by":"crossref","unstructured":"Ibrahim, K., Williams, S., Oliker, L.: Roofline scaling trajectories: a method for parallel application and architectural performance analysis. In: 2018 International Conference on High Performance Computing and Simulation (HPCS). IEEE, pp. 350\u2013358 (2018b)","DOI":"10.1109\/HPCS.2018.00065"},{"key":"18_CR4","unstructured":"InfiniBand Trade Association (2017). [Online]. http:\/\/www.infinibandta.org"},{"key":"18_CR2","unstructured":"Intel-Tensorflow (2019). [Online]. https:\/\/github.com\/Intel-tensorflow\/tensorflow"},{"key":"18_CR21","unstructured":"Jouppi, N.P., Young, C., Patil, N., Patterson, D., Agrawal, et al.: In-datacenter performance analysis of a tensor processing unit. In: 2017 ACM\/IEEE 44th Annual International Symposium on Computer Architecture (ISCA). IEEE (2017a)"},{"key":"18_CR22","unstructured":"Jouppi, N.P., Young, C., Patil, N., Patterson, D., Agrawal, G., Bajwa, R., Bates, S., Bhatia, S., Boden, N., Borchers, A., et al.: In-datacenter performance analysis of a tensor processing unit. In: 2017 ACM\/IEEE 44th Annual International Symposium on Computer Architecture (ISCA). IEEE, pp. 1\u201312 (2017b)"},{"key":"18_CR23","doi-asserted-by":"publisher","first-page":"123","DOI":"10.1201\/9781351036863-5","volume-title":"Contemporary High Performance Computing","author":"Kate Keahey","year":"2019","unstructured":"Keahey, K., Riteau, P., Stanzione, D., Cockerill, T., Mambretti, J., Rad, P., Ruth, P.: Chameleon: a scalable production testbed for computer science research. From Petascale toward Exascale, Contemporary High Performance Computing (2019)"},{"issue":"6","key":"18_CR25","doi-asserted-by":"publisher","first-page":"1201","DOI":"10.1016\/j.cpc.2011.01.025","volume":"182","author":"K-H Kim","year":"2011","unstructured":"Kim, K.-H., Kim, K., Park, Q.-H.: Performance analysis and optimization of three-dimensional FDTD on GPU using roofline model. Comput. Phys. Commun. 182(6), 1201\u20131207 (2011)","journal-title":"Comput. Phys. Commun."},{"key":"18_CR24","doi-asserted-by":"crossref","unstructured":"Kim, H., Nam, H., Jung, W., Lee, J.: Performance analysis of CNN frameworks for GPUs. In: 2017 IEEE International Symposium on Performance Analysis of Systems and Software (ISPASS). IEEE, pp. 55\u201364 (2017)","DOI":"10.1109\/ISPASS.2017.7975270"},{"key":"18_CR26","doi-asserted-by":"crossref","unstructured":"Kong, M., Pouchet, L.-N., Sadayappan, P.: A roofline-based performance estimator for distributed matrix-multiply on Intel CnC. 2015 IEEE International Parallel and Distributed Processing Symposium Workshop, pp. 1241\u20131250 (2015)","DOI":"10.1109\/IPDPSW.2015.134"},{"key":"18_CR27","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: ImageNet classification with deep convolutional neural networks. In: Advances in Neural Information Processing Systems, pp. 1097\u20131105 (2012)"},{"key":"18_CR28","unstructured":"Li, M., Andersen, D.G., Park, J.W., Smola, A.J., Ahmed, A., Josifovski, V., Long, J., Shekita, E.J., Su, B.-Y.: Scaling distributed machine learning with the parameter server. In: 11th $$\\{$$USENIX$$\\}$$ Symposium on Operating Systems Design and Implementation ($$\\{$$OSDI$$\\}$$ 14), (2014), pp. 583\u2013598"},{"key":"18_CR29","doi-asserted-by":"crossref","unstructured":"Lopes, A., Pratas, F., Sousa, L., Ilic, A.: Exploring GPU performance, power and energy-efficiency bounds with cache-aware roofline modeling. In: 2017 IEEE International Symposium on Performance Analysis of Systems and Software (ISPASS). IEEE, pp. 1\u201312 (2017)","DOI":"10.1109\/ISPASS.2017.7975297"},{"key":"18_CR30","doi-asserted-by":"crossref","unstructured":"Lu, X., Shi, H., Shankar, D., Panda, D.K.: Performance characterization and acceleration of big data workloads on OpenPOWER system. In: 2017 IEEE International Conference on Big Data (Big Data), pp. 213\u2013222 (2017)","DOI":"10.1109\/BigData.2017.8257929"},{"issue":"4","key":"18_CR31","doi-asserted-by":"publisher","first-page":"635","DOI":"10.1109\/TMSCS.2018.2845886","volume":"4","author":"X Lu","year":"2018","unstructured":"Lu, X., Shi, H., Biswas, R., Javed, M.H., Panda, D.K.: DLoBD: a comprehensive study of deep learning over big data stacks on HPC clusters. IEEE Trans. MultiScale Comput. Syst. 4(4), 635\u2013648 (2018)","journal-title":"IEEE Trans. MultiScale Comput. Syst."},{"key":"18_CR32","doi-asserted-by":"crossref","unstructured":"Meloni, P., Deriu, G., Conti, F., Loi, I., Raffo, L., Benini, L.: Curbing the roofline: a scalable and flexible architecture for CNNs on FPGA. In: Proceedings of the ACM International Conference on Computing Frontiers. ACM, pp. 376\u2013383 (2016)","DOI":"10.1145\/2903150.2911715"},{"key":"18_CR33","unstructured":"Message Passing Interface Forum (2019). [Online]. http:\/\/www.mpi-forum.org\/"},{"key":"18_CR34","doi-asserted-by":"publisher","first-page":"153","DOI":"10.1016\/j.jpdc.2016.12.015","volume":"106","author":"K Nagasu","year":"2017","unstructured":"Nagasu, K., Sano, K., Kono, F., Nakasato, N.: FPGA-based tsunami simulation: performance comparison with GPUs, and roofline model for scalability analysis. J. Parallel Distrib. Comput. 106, 153\u2013169 (2017)","journal-title":"J. Parallel Distrib. Comput."},{"key":"18_CR5","unstructured":"NVIDIA NCCL (2017). [Online]. https:\/\/github.com\/NVIDIA\/nccl"},{"issue":"2","key":"18_CR35","doi-asserted-by":"publisher","first-page":"117","DOI":"10.1016\/j.jpdc.2008.09.002","volume":"69","author":"P Patarasuk","year":"2009","unstructured":"Patarasuk, P., Yuan, X.: Bandwidth optimal all-reduce algorithms for clusters of workstations. J. Parallel Distrib. Comput. 69(2), 117\u2013124 (2009)","journal-title":"J. Parallel Distrib. Comput."},{"key":"18_CR3","unstructured":"RDMA over Converged Ethernet (2019). http:\/\/www.roceinitiative.org\/"},{"issue":"18","key":"18_CR36","doi-asserted-by":"publisher","first-page":"1719","DOI":"10.1016\/S0140-3664(02)00091-9","volume":"25","author":"JH Sarker","year":"2002","unstructured":"Sarker, J.H., Hassan, M., Halme, S.J.: Power level selection schemes to improve throughput and stability of slotted ALOHA under heavy load. Comput. Commun. 25(18), 1719\u20131726 (2002)","journal-title":"Comput. Commun."},{"key":"18_CR37","unstructured":"Sergeev, A., Del Balso, M.: Horovod: fast and easy distributed deep learning in tensorflow (2018). arXiv preprint arXiv:1802.05799"},{"key":"18_CR38","doi-asserted-by":"crossref","unstructured":"Shi, S., Wang, Q., Xu, P., Chu, X.: Benchmarking state-of-the-art deep learning software tools. In: 2016 7th International Conference on Cloud Computing and Big Data (CCBD). IEEE, pp. 99\u2013104 (2016)","DOI":"10.1109\/CCBD.2016.029"},{"key":"18_CR39","unstructured":"Simonyan, K., Zisserman, A.: Very deep convolutional networks for large-scale image recognition. (2014). arXiv preprint arXiv:1409.1556"},{"key":"18_CR40","doi-asserted-by":"crossref","unstructured":"Szegedy, C., Vanhoucke, V., Ioffe, S., Shlens, J., Wojna, Z.: Rethinking the inception architecture for computer vision. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 2818\u20132826 (2016)","DOI":"10.1109\/CVPR.2016.308"},{"key":"18_CR41","unstructured":"Wang, L., Zhan, J., Gao, W., Ren, R., He, X., Luo, C., Lu, G., Li, J.: BOPS, not FLOPS! A new metric, measuring tool, and roofline performance model for datacenter computing, CoRR, vol. abs\/1801.09212. [Online]. (2018). http:\/\/arxiv.org\/abs\/1801.09212"},{"key":"18_CR42","doi-asserted-by":"crossref","unstructured":"Williams, S., Waterman, A., Patterson, D.: Roofline: an insightful visual performance model for floating-point programs and multicore architectures, Lawrence Berkeley National Lab.(LBNL), Berkeley, CA (United States), Tech. Rep. (2009)","DOI":"10.2172\/1407078"},{"key":"18_CR43","doi-asserted-by":"crossref","unstructured":"Yan, R., Song, Y., Wu, H.: Learning to respond with deep neural networks for retrieval-based human-computer conversation system. In: Proceedings of the 39th International ACM SIGIR Conference on Research and Development in Information Retrieval. ACM, pp. 55\u201364 (2016)","DOI":"10.1145\/2911451.2911542"},{"key":"18_CR44","doi-asserted-by":"crossref","unstructured":"Zhang, C., Li, P., Sun, G., Guan, Y., Xiao, B., Cong, J.: Optimizing FPGA-based accelerator design for deep convolutional neural networks. In: Proceedings of the 2015 ACM\/SIGDA International Symposium on Field-Programmable Gate Arrays. ACM, pp. 161\u2013170 (2015)","DOI":"10.1145\/2684746.2689060"}],"container-title":["CCF Transactions on High Performance Computing"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s42514-019-00018-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s42514-019-00018-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s42514-019-00018-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2020,11,28]],"date-time":"2020-11-28T00:25:36Z","timestamp":1606523136000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s42514-019-00018-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,11,29]]},"references-count":44,"journal-issue":{"issue":"3-4","published-print":{"date-parts":[[2019,12]]}},"alternative-id":["18"],"URL":"https:\/\/doi.org\/10.1007\/s42514-019-00018-4","relation":{},"ISSN":["2524-4922","2524-4930"],"issn-type":[{"type":"print","value":"2524-4922"},{"type":"electronic","value":"2524-4930"}],"subject":[],"published":{"date-parts":[[2019,11,29]]},"assertion":[{"value":"31 July 2019","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 November 2019","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"29 November 2019","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}