{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,12]],"date-time":"2026-05-12T01:35:31Z","timestamp":1778549731150,"version":"3.51.4"},"publisher-location":"Cham","reference-count":43,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031856969","type":"print"},{"value":"9783031856976","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-85697-6_17","type":"book-chapter","created":{"date-parts":[[2025,4,2]],"date-time":"2025-04-02T05:27:10Z","timestamp":1743571630000},"page":"256-270","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Tracing of\u00a0GPU-Aware MPI Applications: First Benchmarks for\u00a0the\u00a0Angara Interconnect"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7381-5759","authenticated-orcid":false,"given":"Timur","family":"Ismagilov","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1391-1407","authenticated-orcid":false,"given":"Anatoly","family":"Mukosey","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-2478-2484","authenticated-orcid":false,"given":"Felix","family":"Smirnov","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-6659-1541","authenticated-orcid":false,"given":"Vladislav","family":"Galigerov","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-6426-5049","authenticated-orcid":false,"given":"Yuri","family":"Grishichkin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5349-3991","authenticated-orcid":false,"given":"Vladimir","family":"Stegailov","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1156-893X","authenticated-orcid":false,"given":"Alexey","family":"Timofeev","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,4,1]]},"reference":[{"key":"17_CR1","doi-asserted-by":"publisher","unstructured":"Petrini, F., Feng, W.c., Hoisie, A., Coll, S., Frachtenberg, E.: The quadrics network: high-performance clustering technology. IEEE Micro 22(1), 46\u201357 (2002). https:\/\/doi.org\/10.1109\/40.988689","DOI":"10.1109\/40.988689"},{"issue":"1","key":"17_CR2","doi-asserted-by":"publisher","first-page":"29","DOI":"10.1109\/40.342015","volume":"15","author":"NJ Boden","year":"1995","unstructured":"Boden, N.J., et al.: Myrinet: a gigabit-per-second local area network. IEEE Micro 15(1), 29\u201336 (1995). https:\/\/doi.org\/10.1109\/40.342015","journal-title":"IEEE Micro"},{"key":"17_CR3","doi-asserted-by":"publisher","unstructured":"Birrittella, M.S., et al.: Intel\u00ae omni-path architecture: Enabling scalable, high performance fabrics. In: 2015 IEEE 23rd Annual Symposium on High-Performance Interconnects, pp. 1\u20139. IEEE (2015). https:\/\/doi.org\/10.1109\/hoti.2015.22","DOI":"10.1109\/hoti.2015.22"},{"key":"17_CR4","unstructured":"InfiniBand trade association: InfiniBand architecture specification. Release 1.0 (2000)"},{"issue":"2","key":"17_CR5","doi-asserted-by":"publisher","first-page":"241","DOI":"10.1145\/264107.264206","volume":"25","author":"J Laudon","year":"1997","unstructured":"Laudon, J., Lenoski, D.: The SGI origin: a ccNUMA highly scalable server. ACM SIGARCH Comput. Archit. News 25(2), 241\u2013251 (1997). https:\/\/doi.org\/10.1145\/264107.264206","journal-title":"ACM SIGARCH Comput. Archit. News"},{"key":"17_CR6","unstructured":"Semiconductor, M.: Rapidio: An Embedded System Component Network Architecture. White Paper (2000)"},{"key":"17_CR7","unstructured":"Introducing 200G HDR InfiniBand Solutions. Mellanox Technologies (2019). http:\/\/mvapich.cse.ohio-state.edu\/benchmarks\/"},{"key":"17_CR8","doi-asserted-by":"publisher","unstructured":"Ruhela, A., Xu, S., Manian, K.V., Subramoni, H., Panda, D.K.: Analyzing and understanding the impact of interconnect performance on HPC, big data, and deep learning applications: A case study with InfiniBand EDR and HDR. In: 2020 IEEE International Parallel and Distributed Processing Symposium Workshops (IPDPSW), pp. 869\u2013878. IEEE (2020). https:\/\/doi.org\/10.1109\/ipdpsw50202.2020.00147","DOI":"10.1109\/ipdpsw50202.2020.00147"},{"key":"17_CR9","doi-asserted-by":"publisher","unstructured":"De\u00a0Sensi, D., Di\u00a0Girolamo, S., McMahon, K.H., Roweth, D., Hoefler, T.: An in-depth analysis of the slingshot interconnect. In: SC20: International Conference for High Performance Computing, Networking, Storage and Analysis, pp. 1\u201314. IEEE (2020). https:\/\/doi.org\/10.1109\/sc41405.2020.00039","DOI":"10.1109\/sc41405.2020.00039"},{"key":"17_CR10","doi-asserted-by":"publisher","unstructured":"Kim, J., Dally, W.J., Scott, S., Abts, D.: Technology-driven, highly-scalable dragonfly topology. In: 2008 International Symposium on Computer Architecture, pp. 77\u201388. IEEE (2008). https:\/\/doi.org\/10.1109\/isca.2008.19","DOI":"10.1109\/isca.2008.19"},{"key":"17_CR11","doi-asserted-by":"publisher","unstructured":"N\u00fcssle, M., Geib, B., Fr\u00f6ning, H., Br\u00fcning, U.: An fpga-based custom high performance interconnection network. In: 2009 International Conference on Reconfigurable Computing and FPGAs, pp. 113\u2013118. IEEE (2009). https:\/\/doi.org\/10.1109\/reconfig.2009.23","DOI":"10.1109\/reconfig.2009.23"},{"key":"17_CR12","doi-asserted-by":"publisher","unstructured":"Neuwirth, S.: Assessment of the I\/O and storage subsystem in modular supercomputing architectures. In: 2022 IEEE International Conference on Cluster Computing (CLUSTER), pp. 589\u2013596. IEEE (2022). https:\/\/doi.org\/10.1109\/CLUSTER51413.2022.00077","DOI":"10.1109\/CLUSTER51413.2022.00077"},{"key":"17_CR13","doi-asserted-by":"publisher","unstructured":"Ajima, Y., et\u00a0al.: The tofu interconnect D. In: 2018 IEEE International Conference on Cluster Computing (CLUSTER), pp. 646\u2013654. IEEE (2018). https:\/\/doi.org\/10.1109\/CLUSTER.2018.00090","DOI":"10.1109\/CLUSTER.2018.00090"},{"key":"17_CR14","doi-asserted-by":"publisher","unstructured":"Emmanuel, J., Moy, M., Henrio, L., Pichon, G.: S4BXI: the MPI-ready Portals 4 simulator. In: 2021 29th International Symposium on Modeling, Analysis, and Simulation of Computer and Telecommunication Systems (MASCOTS), pp. 1\u20138. IEEE (2021). https:\/\/doi.org\/10.1109\/MASCOTS53633.2021.9614285","DOI":"10.1109\/MASCOTS53633.2021.9614285"},{"issue":"9","key":"17_CR15","doi-asserted-by":"publisher","first-page":"1369","DOI":"10.3390\/electronics11091369","volume":"11","author":"PJ Lu","year":"2022","unstructured":"Lu, P.J., Lai, M.C., Chang, J.S.: A survey of high-performance interconnection networks in high-performance computer systems. Electronics 11(9), 1369 (2022). https:\/\/doi.org\/10.3390\/electronics11091369","journal-title":"Electronics"},{"key":"17_CR16","doi-asserted-by":"publisher","unstructured":"Ammendola, R., et\u00a0al.: Outlines in hardware and software for new generations of exascale interconnects. In: EPJ Web of Conferences. vol. 295, p. 10006. EDP Sciences (2024). https:\/\/doi.org\/10.1051\/epjconf\/202429510006","DOI":"10.1051\/epjconf\/202429510006"},{"key":"17_CR17","doi-asserted-by":"publisher","unstructured":"Mukosey, A.V., Semenov, A.S., Simonov, A.S.: Simulation of collective operations hardware support for \u00abangara\u00bb interconnect. Vestnik Yuzhno-Ural\u2019skogo Gosudarstvennogo Universiteta. Seriya Vychislitelnaya Matematika i Informatika 4(3), 40\u201355 (2015). https:\/\/doi.org\/10.14529\/cmse150304","DOI":"10.14529\/cmse150304"},{"key":"17_CR18","doi-asserted-by":"publisher","unstructured":"Simonov, A., Brekhov, O.: Architecture and functionality of the collective operations subnet of the angara interconnect. In: Vishnevskiy, V.M., Samouylov, K.E., Kozyrev, D.V. (eds.) Distributed Computer and Communication Networks, pp. 209\u2013219. Springer International Publishing, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-66471-8_17","DOI":"10.1007\/978-3-030-66471-8_17"},{"key":"17_CR19","unstructured":"Basalov, V.G., Vyalukhin, V.M.: Adaptive routing system for the domestic interconnect SMPO-10G. VANT. Ser. Mat. Mod. Fiz. Proc. 3, 64\u201370 (2012)"},{"issue":"9","key":"17_CR20","doi-asserted-by":"publisher","first-page":"1159","DOI":"10.1134\/s1995080218090081","volume":"39","author":"V Akimov","year":"2018","unstructured":"Akimov, V., Silaev, D., Aksenov, A., Zhluktov, S., Savitskiy, D., Simonov, A.: Flowvision scalability on supercomputers with angara interconnect. Lobachevskii J. Math. 39(9), 1159\u20131169 (2018). https:\/\/doi.org\/10.1134\/s1995080218090081","journal-title":"Lobachevskii J. Math."},{"issue":"3","key":"17_CR21","doi-asserted-by":"publisher","first-page":"507","DOI":"10.1177\/1094342019826667","volume":"33","author":"V Stegailov","year":"2019","unstructured":"Stegailov, V., et al.: Angara interconnect makes GPU-based desmos supercomputer an efficient tool for molecular dynamics calculations. Int. J. High Perform. Comput. Appl. 33(3), 507\u2013521 (2019). https:\/\/doi.org\/10.1177\/1094342019826667","journal-title":"Int. J. High Perform. Comput. Appl."},{"key":"17_CR22","doi-asserted-by":"publisher","unstructured":"Goncharuk, Y., Grishichkin, Y., Semenov, A., Stegailov, V., Umrihin, V.: Evaluation of the Angara interconnect prototype TCP\/IP software stack: Implementation, basic tests and BeeGFS benchmarks. In: Voevodin, V., Sobolev, S., Yakobovskiy, M., Shagaliev, R. (eds.) Supercomputing, pp. 423\u2013435. Springer International Publishing, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-22941-1_31","DOI":"10.1007\/978-3-031-22941-1_31"},{"key":"17_CR23","doi-asserted-by":"publisher","DOI":"10.1016\/j.jpdc.2023.104765","volume":"183","author":"A Mukosey","year":"2024","unstructured":"Mukosey, A., Semenov, A., Tretiakov, A.: Graph based routing algorithm for torus topology and its evaluation for the Angara interconnect. J. Parallel Distrib. Comput. 183, 104765 (2024). https:\/\/doi.org\/10.1016\/j.jpdc.2023.104765","journal-title":"J. Parallel Distrib. Comput."},{"issue":"1","key":"17_CR24","doi-asserted-by":"publisher","first-page":"94","DOI":"10.1109\/TPDS.2019.2928289","volume":"31","author":"A Li","year":"2019","unstructured":"Li, A., et al.: Evaluating modern GPU interconnect: PCIe, NVlink, NV-SLI, NVSwitch and GPUdirect. IEEE Trans. Parallel Distrib. Syst. 31(1), 94\u2013110 (2019). https:\/\/doi.org\/10.1109\/TPDS.2019.2928289","journal-title":"IEEE Trans. Parallel Distrib. Syst."},{"key":"17_CR25","doi-asserted-by":"publisher","unstructured":"Zhou, K., Krentel, M.W., Mellor-Crummey, J.: Tools for top-down performance analysis of GPU-accelerated applications. In: Proceedings of the 34th ACM International Conference on Supercomputing, pp. 1\u201312 (2020). https:\/\/doi.org\/10.1145\/3392717.3392752","DOI":"10.1145\/3392717.3392752"},{"key":"17_CR26","doi-asserted-by":"publisher","DOI":"10.1016\/j.parco.2021.102837","volume":"108","author":"K Zhou","year":"2021","unstructured":"Zhou, K., et al.: Measurement and analysis of GPU-accelerated applications with HPCToolkit. Parallel Comput. 108, 102837 (2021). https:\/\/doi.org\/10.1016\/j.parco.2021.102837","journal-title":"Parallel Comput."},{"key":"17_CR27","doi-asserted-by":"publisher","unstructured":"Zhou, K., Anderson, J., Meng, X., Mellor-Crummey, J.: Low overhead and context sensitive profiling of GPU-accelerated applications. In: Proceedings of the 36th ACM International Conference on Supercomputing, pp. 1\u201313 (2022). https:\/\/doi.org\/10.1145\/3524059.3532388","DOI":"10.1145\/3524059.3532388"},{"key":"17_CR28","doi-asserted-by":"publisher","DOI":"10.1145\/3649510","author":"S Darche","year":"2023","unstructured":"Darche, S., Dagenais, M.R.: Low-overhead trace collection and profiling on GPU compute kernels. ACM Trans. Parallel Comput. (2023). https:\/\/doi.org\/10.1145\/3649510","journal-title":"ACM Trans. Parallel Comput."},{"key":"17_CR29","doi-asserted-by":"publisher","unstructured":"Mey, D.A., et\u00a0al.: Score-p: a unified performance measurement system for petascale applications. In: Competence in High Performance Computing 2010: Proceedings of an International Conference on Competence in High Performance Computing, June 2010, Schloss Schwetzingen, Germany, pp. 85\u201397. Springer (2012). https:\/\/doi.org\/10.1007\/978-3-642-24025-6_8","DOI":"10.1007\/978-3-642-24025-6_8"},{"key":"17_CR30","doi-asserted-by":"publisher","unstructured":"Dietrich, R., Winkler, F., Tsch\u00fcter, R., Weber, M.: Enabling performance analysis of Kokkos applications with Score-P. In: Tools for High Performance Computing 2018\/2019: Proceedings of the 12th and of the 13th International Workshop on Parallel Tools for High Performance Computing, Stuttgart, Germany, September 2018, and Dresden, Germany, September 2019, pp. 169\u2013182. Springer (2021). https:\/\/doi.org\/10.1007\/978-3-030-66057-4_9","DOI":"10.1007\/978-3-030-66057-4_9"},{"issue":"23","key":"17_CR31","doi-asserted-by":"publisher","DOI":"10.1002\/cpe.7188","volume":"34","author":"A Fiorini","year":"2022","unstructured":"Fiorini, A., Dagenais, M.R.: Visualization of profiling and tracing in CPU-GPU programs. Concurr. Comput. Pract. Exp. 34(23), e7188 (2022). https:\/\/doi.org\/10.1002\/cpe.7188","journal-title":"Concurr. Comput. Pract. Exp."},{"key":"17_CR32","doi-asserted-by":"publisher","unstructured":"Potluri, S., Hamidouche, K., Venkatesh, A., Bureddy, D., Panda, D.K.: Efficient inter-node MPI communication using GPUDirect RDMA for InfiniBand clusters with NVIDIA GPUs. In: 2013 42nd International Conference on Parallel Processing, pp. 80\u201389. IEEE (2013). https:\/\/doi.org\/10.1109\/ICPP.2013.17","DOI":"10.1109\/ICPP.2013.17"},{"key":"17_CR33","doi-asserted-by":"publisher","unstructured":"Li, R., et al.: Performance implications of Async Memcpy and UVM: a tale of two data transfer modes. In: 2023 IEEE International Symposium on Workload Characterization (IISWC), pp. 115\u2013127. IEEE (2023). https:\/\/doi.org\/10.1109\/iiswc59245.2023.00024","DOI":"10.1109\/iiswc59245.2023.00024"},{"key":"17_CR34","doi-asserted-by":"publisher","DOI":"10.1016\/j.parco.2021.102831","volume":"108","author":"RT Mills","year":"2021","unstructured":"Mills, R.T., et al.: Toward performance-portable PETSc for GPU-based exascale systems. Parallel Comput. 108, 102831 (2021). https:\/\/doi.org\/10.1016\/j.parco.2021.102831","journal-title":"Parallel Comput."},{"key":"17_CR35","doi-asserted-by":"publisher","unstructured":"Azad, M.A.K., Iqbal, N., Hassan, F., Roy, P.: An empirical study of high performance computing (HPC) performance bugs. In: 2023 IEEE\/ACM 20th International Conference on Mining Software Repositories (MSR), pp. 194\u2013206. IEEE (2023). https:\/\/doi.org\/10.1109\/msr59073.2023.00037","DOI":"10.1109\/msr59073.2023.00037"},{"key":"17_CR36","doi-asserted-by":"publisher","DOI":"10.1109\/tsc.2023.3337662","author":"J Tronge","year":"2023","unstructured":"Tronge, J., et al.: An HPC-container based continuous integration tool for detecting scaling and performance issues in HPC applications. IEEE Trans. Serv. Comput. (2023). https:\/\/doi.org\/10.1109\/tsc.2023.3337662","journal-title":"IEEE Trans. Serv. Comput."},{"key":"17_CR37","doi-asserted-by":"publisher","unstructured":"Rohr, D., Neskovic, G., Lindenstruth, V.: The L-CSC cluster: optimizing power efficiency to become the greenest supercomputer in the world in the green500 list of november 2014. Supercomput. Front. Innov.: Int. J. 2(3), 41\u201348 (2015). https:\/\/doi.org\/10.14529\/jsfi150304","DOI":"10.14529\/jsfi150304"},{"key":"17_CR38","unstructured":"rocHPL. https:\/\/github.com\/ROCm\/rocHPL. Accessed 24 May 2024"},{"key":"17_CR39","doi-asserted-by":"publisher","unstructured":"Khalilov, M., Timofeev, A., Polyakov, D.: Towards OpenUCX and GPUDirect technology support for the Angara interconnect. In: Voevodin, V., Sobolev, S., Yakobovskiy, M., Shagaliev, R. (eds.) Supercomputing, pp. 591\u2013603. Springer International Publishing, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-22941-1_43","DOI":"10.1007\/978-3-031-22941-1_43"},{"key":"17_CR40","unstructured":"GPU-enabled message passing interface. https:\/\/rocm.docs.amd.com\/en\/develop\/how-to\/gpu-enabled-mpi.html. Accessed 24 May 2024"},{"key":"17_CR41","unstructured":"Unified communication X. https:\/\/openucx.org. Accessed 24 May 2024"},{"key":"17_CR42","doi-asserted-by":"publisher","unstructured":"Williams, W., Brunst, H.: Parallel performance engineering using Score-P and Vampir. In: Companion of the 2023 ACM\/SPEC International Conference on Performance Engineering, pp. 121\u2013125 (2023). https:\/\/doi.org\/10.1145\/3578245.3583715","DOI":"10.1145\/3578245.3583715"},{"key":"17_CR43","unstructured":"Score-P. https:\/\/gitlab.com\/score-p\/scorep\/-\/issues\/1005. Accessed 24 May 2024"}],"container-title":["Lecture Notes in Computer Science","Parallel Processing and Applied Mathematics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-85697-6_17","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,4,2]],"date-time":"2025-04-02T05:27:25Z","timestamp":1743571645000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-85697-6_17"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9783031856969","9783031856976"],"references-count":43,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-85697-6_17","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"1 April 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"PPAM","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Parallel Processing and Applied Mathematics","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Ostrava","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Czech Republic","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"9 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"12 September 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"15","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ppam2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ppam.edu.pl\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}