{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T23:37:20Z","timestamp":1783035440007,"version":"3.54.6"},"reference-count":55,"publisher":"Springer Science and Business Media LLC","issue":"5-6","license":[{"start":{"date-parts":[[2019,2,11]],"date-time":"2019-02-11T00:00:00Z","timestamp":1549843200000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100000038","name":"Natural Sciences and Engineering Research Council of Canada","doi-asserted-by":"crossref","id":[{"id":"10.13039\/501100000038","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Parallel Prog"],"published-print":{"date-parts":[[2019,12]]},"DOI":"10.1007\/s10766-019-00630-5","type":"journal-article","created":{"date-parts":[[2019,2,11]],"date-time":"2019-02-11T12:04:00Z","timestamp":1549886640000},"page":"973-1013","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Tracing and Profiling Machine Learning Dataflow Applications on GPU"],"prefix":"10.1007","volume":"47","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3890-7690","authenticated-orcid":false,"given":"Pierre","family":"Zins","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Michel","family":"Dagenais","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2019,2,11]]},"reference":[{"issue":"8","key":"630_CR1","first-page":"114","volume":"38","author":"GE Moore","year":"1965","unstructured":"Moore, G.E.: Cramming more components onto integrated circuits. Electronics 38(8), 114 (1965)","journal-title":"Electronics"},{"issue":"5","key":"630_CR2","doi-asserted-by":"publisher","first-page":"879","DOI":"10.1109\/JPROC.2008.917757","volume":"96","author":"JD Owens","year":"2008","unstructured":"Owens, J.D., Houston, M., Luebke, D., Green, S., Stone, J.E., Phillips, J.C.: Gpu computing. Proc. IEEE 96(5), 879\u2013899 (2008)","journal-title":"Proc. IEEE"},{"issue":"3","key":"630_CR3","doi-asserted-by":"publisher","first-page":"654","DOI":"10.1109\/TSP.2017.2773424","volume":"66","author":"Jani Boutellier","year":"2018","unstructured":"Boutellier, J., Wu, J., Huttunen, H., Bhattacharyya, S.S.: PRUNE: dynamic and decidable dataflow for signal processing on heterogeneous platforms (2018). CoRR arXiv:1802.06625","journal-title":"IEEE Transactions on Signal Processing"},{"issue":"3","key":"630_CR4","doi-asserted-by":"publisher","first-page":"469","DOI":"10.1007\/s11265-017-1260-8","volume":"89","author":"J Boutellier","year":"2017","unstructured":"Boutellier, J., Nyl\u00e4nden, T.: Design flow for GPU and multicore execution of dynamic dataflow programs. J. Signal Process. Syst. 89(3), 469\u2013478 (2017)","journal-title":"J. Signal Process. Syst."},{"key":"630_CR5","doi-asserted-by":"crossref","unstructured":"Bezati, E., Mattavelli, M., Raulet, M.: Rvc-cal dataflow implementations of mpeg avc\/h.264 cabac decoding. In: 2010 Conference on Design and Architectures for Signal and Image Processing (DASIP), pp. 207\u2013213 (2010)","DOI":"10.1109\/DASIP.2010.5706266"},{"key":"630_CR6","unstructured":"Hentati, M., Aoudni, Y., Nezan, J.F., Abid, M.: A hierarchical implementation of hadamard transform using rvc-cal dataflow programming and dynamic partial reconfiguration. In: Proceedings of the 2012 Conference on Design and Architectures for Signal and Image Processing, pp. 1\u20137 (2012)"},{"key":"630_CR7","doi-asserted-by":"crossref","unstructured":"Blattner, T., Keyrouz, W., Halem, M., Brady, M., Bhattacharyya, S.S.: A hybrid task graph scheduler for high performance image processing workflows. In: 2015 IEEE Global Conference on Signal and Information Processing (GlobalSIP), pp. 634\u2013637 (2015)","DOI":"10.1109\/GlobalSIP.2015.7418273"},{"issue":"4","key":"630_CR8","doi-asserted-by":"publisher","first-page":"280","DOI":"10.1049\/iet-cds.2015.0071","volume":"10","author":"C Bourrasset","year":"2016","unstructured":"Bourrasset, C., Maggiani, L., Srot, J., Berry, F.: Dataflow object detection system for fpga-based smart camera. IET Circuits Devices Syst. 10(4), 280\u2013291 (2016)","journal-title":"IET Circuits Devices Syst."},{"issue":"9","key":"630_CR9","doi-asserted-by":"publisher","first-page":"1305","DOI":"10.1109\/5.97300","volume":"79","author":"N Halbwachs","year":"1991","unstructured":"Halbwachs, N., Caspi, P., Raymond, P., Pilaud, D.: The synchronous data flow programming language lustre. Proc. IEEE 79(9), 1305\u20131320 (1991)","journal-title":"Proc. IEEE"},{"key":"630_CR10","doi-asserted-by":"crossref","unstructured":"Caspi, P., Pilaud, D., Halbwachs, N., Plaice, J.A.: LUSTRE: a declarative language for real-time programming. In: Proceedings of the 14th ACM SIGACT-SIGPLAN Symposium on Principles of Programming Languages, POPL \u201987, pp. 178\u2013188. ACM, New York (1987)","DOI":"10.1145\/41625.41641"},{"key":"630_CR11","volume-title":"LUCID, the Dataflow Programming Language","author":"WW Wadge","year":"1985","unstructured":"Wadge, W.W., Ashcroft, E.A.: LUCID, the Dataflow Programming Language. Academic Press Professional, Inc., San Diego (1985)"},{"key":"630_CR12","unstructured":"Eker, J., Janneck, J.W.: CAL language report: specification of the CAL actor language (2003)"},{"key":"630_CR13","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.E.: Imagenet classification with deep convolutional neural networks. In: Advances in Neural Information Processing Systems 25: 26th Annual Conference on Neural Information Processing Systems 2012. Proceedings of a meeting held December 3\u20136, 2012, Lake Tahoe, Nevada, USA, pp. 1106\u20131114 (2012)"},{"key":"630_CR14","unstructured":"Theano Development Team.: Theano: a python framework for fast computation of mathematical expressions (2016). CoRR arXiv:1605.02688"},{"key":"630_CR15","doi-asserted-by":"crossref","unstructured":"Bergstra, J., Breuleux, O., Bastien, F., Lamblin, P., Pascanu, R., Desjardins, G., Turian, J., Warde-Farley, D., Bengio, Y.: Theano: a CPU and GPU math expression compiler. In: Proceedings of the Python for Scientific Computing Conference (SciPy), June 2010. Oral Presentation (2010)","DOI":"10.25080\/Majora-92bf1922-003"},{"key":"630_CR16","doi-asserted-by":"crossref","unstructured":"Abadi, M., Isard, M., Murray, D.G.: A computational model for tensorflow (an introduction) (2017)","DOI":"10.1145\/3088525.3088527"},{"key":"630_CR17","unstructured":"Abadi, M., et al.: Tensorflow: large-scale machine learning on heterogeneous distributed systems (2016). CoRR arXiv:1603.04467"},{"key":"630_CR18","unstructured":"Abadi, M., et al.: Tensorflow: a system for large-scale machine learning (2016). CoRR arXiv:1605.08695"},{"key":"630_CR19","unstructured":"David, G.: Unified kernel\/user-space efficient linux tracing architecture. Master\u2019s thesis, cole Polytechnique de Montral. Retrieved from https:\/\/publications.polymtl.ca\/842\/ (2012)"},{"key":"630_CR20","unstructured":"Fournier, P.-M., Desnoyers, M., Dagenais, M.R.: Combined tracing of the kernel and applications with LTTng. In: Proceedings of the 2009 Linux Symposium (2009)"},{"key":"630_CR21","unstructured":"Hesik, C.: CodeXL 2.6 is released!. https:\/\/gpuopen.com\/codexl-2-6-released\/ (2018)"},{"key":"630_CR22","unstructured":"NVIDIA Developer Tools Overview. https:\/\/developer.nvidia.com\/tools-overview (2018)"},{"key":"630_CR23","unstructured":"Get Started with Intel Graphics Performance Analyzers (Intel GPA). https:\/\/software.intel.com\/en-us\/gpa_getting_started (2018)"},{"key":"630_CR24","unstructured":"Event Tracing. https:\/\/docs.microsoft.com\/en-us\/windows\/desktop\/etw\/event-tracing-portal (2018)"},{"key":"630_CR25","unstructured":"Gregg, B.: Strace wow much syscall (2014)"},{"key":"630_CR26","unstructured":"Gregg, B.: Perf Examples (2014)"},{"key":"630_CR27","unstructured":"Rostedt, S.: Finding Origins of Latencies Using Ftrace (2009)"},{"key":"630_CR28","unstructured":"Gregg, B.: Flame Graphs (2011)"},{"key":"630_CR29","unstructured":"Desnoyers, M., Dagenais, M.R.: The LTTng tracer: a low impact performance and behavior monitor for GNU\/Linux. OLS (Ottawa Linux Symposium) (2006)"},{"key":"630_CR30","unstructured":"NVIDIA Nsight Systems User Guide. https:\/\/docs.nvidia.com\/nsight-systems\/index.html (2018)"},{"key":"630_CR31","unstructured":"Nsight Compute. https:\/\/docs.nvidia.com\/nsight-compute\/index.html (2018)"},{"key":"630_CR32","unstructured":"Nsight Graphics. https:\/\/docs.nvidia.com\/nsight-graphics\/UserGuide\/index.html (2018)"},{"key":"630_CR33","unstructured":"Ponweiser, T.: Profiling and tracing tools for performance analysis of large scale applications (2017)"},{"key":"630_CR34","unstructured":"Pillet, V., Labarta, J., Cortes, T., Girona, S., and Departament D\u2019arquitectura\u00a0De Computadors.: Paraver: a tool to visualize and analyze parallel code. Technical report, In WoTUG-18 (1995)"},{"key":"630_CR35","doi-asserted-by":"crossref","unstructured":"Canale, M., Casale-Brunet, S., Bezati, E., Mattavelli, M., Janneck, J., Casale-Brunet, S., Bezati, E., Mattavelli, M., Marco Mattavelli@epfl Ch., Janneck, J.: Dataflow programs analysis and optimization using model predictive control techniques two examples of bounded buffer scheduling: deadlock avoidance and deadlock recovery strategies. J. Signal Process. Syst. 84, 371\u2013381 (2016)","DOI":"10.1007\/s11265-015-1083-4"},{"key":"630_CR36","doi-asserted-by":"crossref","unstructured":"Janneck, J.W., Miller, I.D., Parlour, D.B.: Profiling dataflow programs. In: 2008 IEEE International Conference on Multimedia and Expo, ICME 2008-Proceedings, pp. 1065\u20131068 (2008)","DOI":"10.1109\/ICME.2008.4607622"},{"key":"630_CR37","doi-asserted-by":"crossref","unstructured":"Brunet, S.C., Mattavelli, M., Janneck, J.W.: Profiling of dataflow programs using post mortem causation traces. In: IEEE Workshop on Signal Processing Systems, SiPS: Design and Implementation, pp. 220\u2013225 (2012)","DOI":"10.1109\/SiPS.2012.54"},{"issue":"2","key":"630_CR38","doi-asserted-by":"publisher","first-page":"211","DOI":"10.1145\/1353535.1346308","volume":"42","author":"Shashidhar Mysore","year":"2008","unstructured":"Mysore, S., Mazloom, B., Agrawal, B., Sherwood, T.: Understanding and visualizing full systems with data flow tomography (2008)","journal-title":"ACM SIGOPS Operating Systems Review"},{"key":"630_CR39","doi-asserted-by":"crossref","unstructured":"Osmari, D.K., Vo, H.T., Silva, C.T., Comba, J.L.D., Lins, L.: Visualization and analysis of parallel dataflow execution with smart traces. In: Brazilian Symposium of Computer Graphic and Image Processing, pp. 165\u2013172 (2014)","DOI":"10.1109\/SIBGRAPI.2014.2"},{"key":"630_CR40","unstructured":"Stoner, G.: ROCm: platform for a new era of heterogeneous in HPC and ultrascale computing (2016)"},{"key":"630_CR41","doi-asserted-by":"publisher","first-page":"7","DOI":"10.1016\/B978-0-12-800386-2.00001-8","volume-title":"Heterogeneous System Architecture","author":"P. Rogers","year":"2016","unstructured":"Rogers, P.: HSA Overview, pp. 7\u201318 (2015)"},{"key":"630_CR42","doi-asserted-by":"crossref","unstructured":"Goli, M., Iwanski, L., Richards, A.: Accelerated machine learning using TensorFlow and SYCL on OpenCL Devices. In: Proceedings of the 5th International Workshop on OpenCL, IWOCL 2017, pp. 8:1\u20138:4. ACM, New York (2017)","DOI":"10.1145\/3078155.3078160"},{"key":"630_CR43","doi-asserted-by":"crossref","unstructured":"Keryell, R., Reyes, R., Howes, L.: Khronos sycl for opencl: a tutorial. In: Proceedings of the 3rd International Workshop on OpenCL, IWOCL \u201915, pp. 24:1\u201324:1. ACM, New York (2015)","DOI":"10.1145\/2791321.2791345"},{"key":"630_CR44","unstructured":"Lea, D.: A Memory Allocator (1996)"},{"key":"630_CR45","unstructured":"Paul, M.: Traage logiciel d\u2019applications utilisant un processeur graphique. Master\u2019s thesis, cole Polytechnique de Montral. Retrieved from https:\/\/publications.polymtl.ca\/2838\/ (2017)"},{"key":"630_CR46","doi-asserted-by":"publisher","first-page":"2:2","DOI":"10.1155\/2015\/940628","volume":"2015","author":"D Couturier","year":"2015","unstructured":"Couturier, D., Dagenais, M.R.: LTTng CLUST: a system-wide unified CPU and GPU tracing tool for OpenCL applications. Adv. Softw. Eng. 2015, 2:2\u20132:2 (2015)","journal-title":"Adv. Softw. Eng."},{"issue":"3","key":"630_CR47","doi-asserted-by":"publisher","first-page":"75","DOI":"10.1145\/1842733.1842747","volume":"44","author":"B Poirier","year":"2010","unstructured":"Poirier, B., Roy, R., Dagenais, M.: Accurate offline synchronization of distributed traces using kernel-level events. SIGOPS Oper. Syst. Rev. 44(3), 75\u201387 (2010)","journal-title":"SIGOPS Oper. Syst. Rev."},{"key":"630_CR48","unstructured":"Jabbarifar, M.: On line trace synchronization for large scale distributed systems. PhD thesis, \u00c9cole Polytechnique de Montr\u00e9al (2013)"},{"key":"630_CR49","doi-asserted-by":"publisher","first-page":"201","DOI":"10.1007\/s11219-016-9311-0","volume":"25","author":"F Wininger","year":"2016","unstructured":"Wininger, F., Ezzati-Jivan, N., Dagenais, M.R.: A declarative framework for stateful analysis of execution traces. Softw. Qual. J. 25, 201\u2013229 (2016)","journal-title":"Softw. Qual. J."},{"key":"630_CR50","doi-asserted-by":"crossref","unstructured":"Kouame, K., Ezzati-Jivan, N., Dagenais, M.R.: A flexible data-driven approach for execution trace filtering. In: 2015 IEEE International Congress on Big Data, pp. 698\u2013703 (2015)","DOI":"10.1109\/BigDataCongress.2015.112"},{"key":"630_CR51","unstructured":"Moindrot, O.: Triplet Loss and Online Triplet Mining in TensorFlow (2018)"},{"key":"630_CR52","unstructured":"Springenberg, J.T., Dosovitskiy, A., Brox, T., Riedmiller, M.A.: Striving for simplicity: the all convolutional net (2014). CoRR arXiv:1412.6806"},{"key":"630_CR53","doi-asserted-by":"crossref","unstructured":"Mayer, R., Mayer, C., Laich, L.: The TensorFlow Partitioning and Scheduling Problem: It\u2019s the Critical Path! pp. 1\u20136 (2017)","DOI":"10.1145\/3154842.3154843"},{"key":"630_CR54","unstructured":"Mirhoseini, A., Pham, H., Le, Q.V., Steiner, B., Larsen, R., Zhou, Y., Kumar, N., Norouzi, M., Bengio, S., Dean, J.: Device placement optimization with reinforcement learning. In: Icml (2017)"},{"key":"630_CR55","unstructured":"Optimizing for mobile. https:\/\/www.tensorflow.org\/lite\/tfmobile\/optimizing (2018)"}],"container-title":["International Journal of Parallel Programming"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10766-019-00630-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10766-019-00630-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10766-019-00630-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2020,11,27]],"date-time":"2020-11-27T09:53:10Z","timestamp":1606470790000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10766-019-00630-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,2,11]]},"references-count":55,"journal-issue":{"issue":"5-6","published-print":{"date-parts":[[2019,12]]}},"alternative-id":["630"],"URL":"https:\/\/doi.org\/10.1007\/s10766-019-00630-5","relation":{},"ISSN":["0885-7458","1573-7640"],"issn-type":[{"value":"0885-7458","type":"print"},{"value":"1573-7640","type":"electronic"}],"subject":[],"published":{"date-parts":[[2019,2,11]]},"assertion":[{"value":"25 July 2018","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 February 2019","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 February 2019","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}