{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T23:48:18Z","timestamp":1782949698287,"version":"3.54.5"},"publisher-location":"Cham","reference-count":38,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783030710576","type":"print"},{"value":"9783030710583","type":"electronic"}],"license":[{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021]]},"DOI":"10.1007\/978-3-030-71058-3_10","type":"book-chapter","created":{"date-parts":[[2021,3,1]],"date-time":"2021-03-01T17:03:25Z","timestamp":1614618205000},"page":"157-174","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["ComScribe: Identifying Intra-node GPU Communication"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0279-031X","authenticated-orcid":false,"given":"Palwisha","family":"Akhtar","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5129-4166","authenticated-orcid":false,"given":"Erhan","family":"Tezcan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3955-2836","authenticated-orcid":false,"given":"Fareed Mohammad","family":"Qararyah","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2351-0770","authenticated-orcid":false,"given":"Didem","family":"Unat","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2021,3,2]]},"reference":[{"key":"10_CR1","unstructured":"Abadi, M., et al.: TensorFlow: a system for large-scale machine learning. In: Proceedings of the 12th USENIX Conference on Operating Systems Design and Implementation, OSDI 2016, pp. 265\u2013283. USENIX Association, USA (2016)"},{"key":"10_CR2","doi-asserted-by":"crossref","unstructured":"Amari, S.: Backpropagation and stochastic gradient descent method. Neurocomputing 5(4\u20135), 185\u2013196 (1993)","DOI":"10.1016\/0925-2312(93)90006-O"},{"issue":"2","key":"10_CR3","doi-asserted-by":"publisher","first-page":"56","DOI":"10.1145\/1531793.1531803","volume":"43","author":"R Azimi","year":"2009","unstructured":"Azimi, R., Tam, D.K., Soares, L., Stumm, M.: Enhancing operating system support for multicore processors by using hardware performance monitoring. ACM SIGOPS Oper. Syst. Rev. 43(2), 56\u201365 (2009)","journal-title":"ACM SIGOPS Oper. Syst. Rev."},{"key":"10_CR4","doi-asserted-by":"crossref","unstructured":"Barrow-Williams, N., Fensch, C., Moore, S.: A communication characterisation of Splash-2 and Parsec. In: 2009 IEEE International Symposium on Workload Characterization (IISWC), pp. 86\u201397. IEEE (2009)","DOI":"10.1109\/IISWC.2009.5306792"},{"issue":"4","key":"10_CR5","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3320060","volume":"52","author":"T Ben-Nun","year":"2019","unstructured":"Ben-Nun, T., Hoefler, T.: Demystifying parallel and distributed deep learning: an in-depth concurrency analysis. ACM Comput. Surv. (CSUR) 52(4), 1\u201343 (2019)","journal-title":"ACM Comput. Surv. (CSUR)"},{"key":"10_CR6","doi-asserted-by":"publisher","unstructured":"Ben-Nun, T., Levy, E., Barak, A., Rubin, E.: Memory access patterns: the missing piece of the multi-GPU puzzle. In: International Conference for High Performance Computing, Networking, Storage and Analysis (SC), 15\u201320 November 2015 (2015). https:\/\/doi.org\/10.1145\/2807591.2807611","DOI":"10.1145\/2807591.2807611"},{"key":"10_CR7","unstructured":"Ben-Nuun, T.: MGBench: multi-GPU computing benchmark suite (CUDA) (2017). https:\/\/github.com\/tbennun\/mgbench. Accessed 29 Jul 2020"},{"key":"10_CR8","doi-asserted-by":"publisher","unstructured":"Buono, D., Artico, F., Checconi, F., Choi, J.W., Que, X., Schneidenbach, L.: Data analytics with NVLink: an SpMV case study. In: ACM International Conference on Computing Frontiers 2017, CF 2017, pp. 89\u201396 (2017). https:\/\/doi.org\/10.1145\/3075564.3075569","DOI":"10.1145\/3075564.3075569"},{"key":"10_CR9","unstructured":"da Cruz, E.H.M., Alves, M.A.Z., Carissimi, A., Navaux, P.O.A., Ribeiro, C.P., M\u00e9haut, J.F.: Using memory access traces to map threads and data on hierarchical multi-core platforms. In: 2011 IEEE International Symposium on Parallel and Distributed Processing Workshops and Phd Forum, pp. 551\u2013558. IEEE (2011)"},{"key":"10_CR10","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"166","DOI":"10.1007\/11549468_21","volume-title":"Euro-Par 2005 Parallel Processing","author":"V Danjean","year":"2005","unstructured":"Danjean, V., Namyst, R., Wacrenier, P.-A.: An efficient multi-level trace toolkit for multi-threaded applications. In: Cunha, J.C., Medeiros, P.D. (eds.) Euro-Par 2005. LNCS, vol. 3648, pp. 166\u2013175. Springer, Heidelberg (2005). https:\/\/doi.org\/10.1007\/11549468_21"},{"key":"10_CR11","doi-asserted-by":"crossref","unstructured":"Diener, M., Cruz, E.H., Alves, M.A., Navaux, P.O.: Communication in shared memory: concepts, definitions, and efficient detection. In: 2016 24th Euromicro International Conference on Parallel, Distributed, and Network-Based Processing (PDP), pp. 151\u2013158. IEEE (2016)","DOI":"10.1109\/PDP.2016.16"},{"key":"10_CR12","doi-asserted-by":"publisher","first-page":"18","DOI":"10.1016\/j.peva.2015.03.001","volume":"88","author":"M Diener","year":"2015","unstructured":"Diener, M., Cruz, E.H., Pilla, L.L., Dupros, F., Navaux, P.O.: Characterizing communication and page usage of parallel applications for thread and data mapping. Perform. Eval. 88, 18\u201336 (2015)","journal-title":"Perform. Eval."},{"issue":"2","key":"10_CR13","doi-asserted-by":"publisher","first-page":"7","DOI":"10.1109\/MM.2017.37","volume":"37","author":"D Foley","year":"2017","unstructured":"Foley, D., Danskin, J.: Ultra-performance pascal GPU and NVLink interconnect. IEEE Micro 37(2), 7\u201317 (2017)","journal-title":"IEEE Micro"},{"key":"10_CR14","unstructured":"Harris, M.: Unified memory for cuda beginners \u2014 nvidia developer blog (June 2017). https:\/\/devblogs.nvidia.com\/unified-memory-cuda-beginners\/"},{"issue":"1","key":"10_CR15","doi-asserted-by":"publisher","first-page":"94","DOI":"10.1109\/TPDS.2019.2928289","volume":"31","author":"A Li","year":"2020","unstructured":"Li, A., et al.: Evaluating modern GPU interconnect: PCIe, NVLink, NV-SLI, NVSwitch and GPUDirect. IEEE Trans. Parallel Distrib. Syst. 31(1), 94\u2013110 (2020)","journal-title":"IEEE Trans. Parallel Distrib. Syst."},{"key":"10_CR16","doi-asserted-by":"crossref","unstructured":"Li, A., Song, S.L., Chen, J., Liu, X., Tallent, N., Barker, K.: Tartan: evaluating modern GPU interconnect via a multi-GPU benchmark suite. In: 2018 IEEE International Symposium on Workload Characterization (IISWC), pp. 191\u2013202 (2018)","DOI":"10.1109\/IISWC.2018.8573483"},{"key":"10_CR17","unstructured":"Micikevicius, P.: Multi-GPU programming (2012). https:\/\/on-demand.gputechconf.com\/gtc\/2012\/presentations\/S0515-GTC2012-Multi-GPU-Programming.pdf"},{"key":"10_CR18","unstructured":"NVIDIA: NVIDIA GPUDirect technology (2012). http:\/\/developer.download.nvidia.com\/devzone\/devcenter\/cuda\/docs\/GPUDirect_Technology_Overview.pdf"},{"key":"10_CR19","unstructured":"NVIDIA: NVIDIA DGX-1 with the Tesla V100 system architecture. White Paper (2017). https:\/\/images.nvidia.com\/content\/pdf\/dgx1-v100-system-architecture-whitepaper.pdf. Accessed 28 Jul 2020"},{"key":"10_CR20","unstructured":"NVIDIA: DGX-2 : AI servers for solving complex AI challenges \u2014 NVIDIA (2018). https:\/\/www.nvidia.com\/en-us\/data-center\/dgx-2\/. Accessed 28 Jul 2020"},{"key":"10_CR21","unstructured":"NVIDIA: Multi-gpu-programming-models: examples demonstrating available options to program multiple GPUs in a single node or a cluster (2018). https:\/\/github.com\/NVIDIA\/multi-gpu-programming-models. Accessed 29 Jul 2020"},{"key":"10_CR22","unstructured":"NVIDIA: CUDA runtime API (July 2019). https:\/\/docs.nvidia.com\/cuda\/pdf\/CUDA_Runtime_API.pdf. Accessed 29 Jul 2020"},{"key":"10_CR23","unstructured":"NVIDIA: Ising-gpu: GPU-accelerated Monte Carlo simulations of 2d Ising model (2019). https:\/\/github.com\/NVIDIA\/ising-gpu. Accessed 29 Jul 2020"},{"key":"10_CR24","unstructured":"NVIDIA: Best practices guide: CUDA toolkit documentation (2020). https:\/\/docs.nvidia.com\/cuda\/cuda-c-best-practices-guide\/index.html#zero-copy. Accessed 12 May 2020"},{"key":"10_CR25","unstructured":"NVIDIA: CUDA profiler user\u2019s guide (July 2020). https:\/\/docs.nvidia.com\/cuda\/pdf\/CUDA_Profiler_Users_Guide.pdf. Accessed 28 Jul 2020"},{"key":"10_CR26","unstructured":"NVIDIA DGX A100: universal system for AI infrastructure \u2014 NVIDIA (2020). https:\/\/www.nvidia.com\/en-us\/data-center\/dgx-a100\/. Accessed 28 Jul 2020"},{"key":"10_CR27","unstructured":"NVIDIA: NVIDIA Nsigh systems documentation (2020). https:\/\/docs.nvidia.com\/nsight-systems\/index.html. Accessed 28 Jul 2020"},{"key":"10_CR28","doi-asserted-by":"publisher","unstructured":"Pearson, C., et al.: Evaluating characteristics of CUDA communication primitives on high-bandwidth interconnects. In: Proceedings of the 2019 ACM\/SPEC International Conference on Performance Engineering, ICPE 2019, pp. 209\u2013218. Association for Computing Machinery, New York (2019). https:\/\/doi.org\/10.1145\/3297663.3310299","DOI":"10.1145\/3297663.3310299"},{"key":"10_CR29","doi-asserted-by":"crossref","unstructured":"Sasongko, M.A., Chabbi, M., Akhtar, P., Unat, D.: ComDetective: a lightweight communication detection tool for threads. In: Proceedings of the International Conference for High Performance Computing, Networking, Storage and Analysis, pp. 1\u201321 (2019)","DOI":"10.1145\/3295500.3356214"},{"key":"10_CR30","unstructured":"Shazeer, N., et al.: Mesh-TensorFlow: deep learning for supercomputers. In: Advances in Neural Information Processing Systems, pp. 10414\u201310423 (2018)"},{"key":"10_CR31","doi-asserted-by":"publisher","unstructured":"Sourouri, M., Gillberg, T., Baden, S.B., Cai, X.: Effective multi-GPU communication using multiple CUDA streams and threads. In: Proceedings of the International Conference on Parallel and Distributed Systems, ICPADS 2015, 10 April 2015, pp. 981\u2013986 (2014). https:\/\/doi.org\/10.1109\/PADSW.2014.7097919","DOI":"10.1109\/PADSW.2014.7097919"},{"issue":"3","key":"10_CR32","doi-asserted-by":"publisher","first-page":"47","DOI":"10.1145\/1272998.1273004","volume":"41","author":"D Tam","year":"2007","unstructured":"Tam, D., Azimi, R., Stumm, M.: Thread clustering: sharing-aware scheduling on SMP-CMP-SMT multiprocessors. ACM SIGOPS Oper. Syst. Rev. 41(3), 47\u201358 (2007)","journal-title":"ACM SIGOPS Oper. Syst. Rev."},{"key":"10_CR33","doi-asserted-by":"crossref","unstructured":"Trahay, F., Rue, F., Faverge, M., Ishikawa, Y., Namyst, R., Dongarra, J.: EZTrace: a generic framework for performance analysis. In: 2011 11th IEEE\/ACM International Symposium on Cluster, Cloud and Grid Computing, pp. 618\u2013619. IEEE (2011)","DOI":"10.1109\/CCGrid.2011.83"},{"issue":"8","key":"10_CR34","doi-asserted-by":"publisher","first-page":"103","DOI":"10.1145\/79173.79181","volume":"33","author":"LG Valiant","year":"1990","unstructured":"Valiant, L.G.: A bridging model for parallel computation. Commun. ACM 33(8), 103\u2013111 (1990)","journal-title":"Commun. ACM"},{"key":"10_CR35","unstructured":"Vaswani, A., et al.: Attention is all you need. CoRR abs\/1706.03762 (2017). http:\/\/arxiv.org\/abs\/1706.03762"},{"key":"10_CR36","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Advances in Neural Information Processing Systems, pp. 5998\u20136008 (2017)"},{"key":"10_CR37","unstructured":"Wang, Y., Jiang, L., Yang, M.H., Li, L.J., Long, M., Fei-Fei, L.: Eidetic 3d LSTM: a model for video prediction and beyond. In: International Conference on Learning Representations (2018)"},{"key":"10_CR38","unstructured":"Wang, Y., Jiang, L., Yang, M.H., Li, L.J., Long, M., Fei-Fei, L.: Eidetic 3d LSTM: a model for video prediction and beyond. In: International Conference on Learning Representations (2019). https:\/\/openreview.net\/forum?id=B1lKS2AqtX"}],"container-title":["Lecture Notes in Computer Science","Benchmarking, Measuring, and Optimizing"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-71058-3_10","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,3,1]],"date-time":"2021-03-01T17:10:37Z","timestamp":1614618637000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-030-71058-3_10"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021]]},"ISBN":["9783030710576","9783030710583"],"references-count":38,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-71058-3_10","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021]]},"assertion":[{"value":"2 March 2021","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"Bench","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Symposium on Benchmarking, Measuring and Optimization","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2020","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"15 November 2020","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"16 November 2020","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"3","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"bench2020","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.benchcouncil.org\/bench20\/index.html","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Single-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"EasyChair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"28","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"12","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"1","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"43% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"4.7","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.5","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"No","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}