{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T01:10:23Z","timestamp":1755825023480,"version":"3.44.0"},"publisher-location":"New York, NY, USA","reference-count":52,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,11,12]],"date-time":"2023-11-12T00:00:00Z","timestamp":1699747200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/501100004052","name":"King Abdullah University of Science and Technology","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100004052","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,11,12]]},"DOI":"10.1145\/3624062.3624602","type":"proceedings-article","created":{"date-parts":[[2023,11,10]],"date-time":"2023-11-10T13:53:39Z","timestamp":1699624419000},"page":"1185-1193","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["shmem4py: High-Performance One-Sided Communication for Python Applications"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5662-2082","authenticated-orcid":false,"given":"Marcin","family":"Rogowski","sequence":"first","affiliation":[{"name":"King Abdullah University of Science and Technology, Saudi Arabia and NVIDIA, Poland"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3181-8190","authenticated-orcid":false,"given":"Jeff R.","family":"Hammond","sequence":"additional","affiliation":[{"name":"NVIDIA Helsinki Oy, Finland"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4052-7224","authenticated-orcid":false,"given":"David E.","family":"Keyes","sequence":"additional","affiliation":[{"name":"King Abdullah University of Science and Technology, Saudi Arabia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8086-0155","authenticated-orcid":false,"given":"Lisandro","family":"Dalcin","sequence":"additional","affiliation":[{"name":"King Abdullah University of Science and Technology, Saudi Arabia"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,11,12]]},"reference":[{"key":"e_1_3_2_2_1_1","unstructured":"Collin Abidi. 2020. shmem4py. https:\/\/github.com\/collinabidi\/shmem4py"},{"key":"e_1_3_2_2_2_1","volume-title":"Cray XC series network","author":"Alverson Bob","year":"2012","unstructured":"Bob Alverson, Edwin Froese, Larry Kaplan, and Duncan Roweth. 2012. Cray XC series network. Cray Inc., White Paper WP-Aries01-1112 (2012). https:\/\/www.alcf.anl.gov\/files\/CrayXCNetwork.pdf"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3295500.3356175"},{"key":"e_1_3_2_2_4_1","volume-title":"Understanding the Python GIL. In PyCON Python Conference","author":"Beazley David","year":"2010","unstructured":"David Beazley. 2010. Understanding the Python GIL. In PyCON Python Conference. Atlanta, Georgia."},{"key":"e_1_3_2_2_5_1","unstructured":"James Bradbury Roy Frostig Peter Hawkins Matthew\u00a0James Johnson Chris Leary Dougal Maclaurin George Necula Adam Paszke Jake VanderPlas Skye Wanderman-Milne and Qiao Zhang. 2018. JAX: composable transformations of Python+NumPy programs. http:\/\/github.com\/google\/jax"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/2020373.2020375"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/PGAS.2015.14"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/3126908.3126926"},{"key":"e_1_3_2_2_9_1","volume-title":"Dask: Scale the Python tools you love. https:\/\/www.dask.org\/","author":"Dask","year":"2023","unstructured":"Dask core developers. 2023. Dask: Scale the Python tools you love. https:\/\/www.dask.org\/"},{"key":"e_1_3_2_2_10_1","unstructured":"Cray Research Inc.1993. Cray T3D System Architecture Overview. http:\/\/www.bitsavers.org\/pdf\/cray\/HR-04033_CRAY_T3D_System_Architecture_Overview_Sep93.pdf"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.25080\/Majora-ebaa42b7-004"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/MCSE.2021.3083216"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.advwatres.2011.04.013"},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jpdc.2007.09.005"},{"key":"e_1_3_2_2_15_1","unstructured":"Jeff R.\u00a0Hammond et al.2019. Parallel Research Kernels. https:\/\/github.com\/ParRes\/Kernels"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2012.39"},{"key":"e_1_3_2_2_17_1","volume-title":"Shared memory access (SHMEM) routines. Cray Research 53","author":"Feind Karl","year":"1995","unstructured":"Karl Feind. 1995. Shared memory access (SHMEM) routines. Cray Research 53 (1995). https:\/\/cug.org\/5-publications\/proceedings_attendee_lists\/1997CD\/S95PROC\/303_308.PDF"},{"volume-title":"Recent Advances in Parallel Virtual Machine and Message Passing Interface, Dieter Kranzlm\u00fcller, P\u00e9ter Kacsuk","author":"Gabriel Edgar","key":"e_1_3_2_2_18_1","unstructured":"Edgar Gabriel, Graham\u00a0E. Fagg, George Bosilca, Thara Angskun, Jack\u00a0J. Dongarra, Jeffrey\u00a0M. Squyres, Vishal Sahay, Prabhanjan Kambadur, Brian Barrett, Andrew Lumsdaine, Ralph\u00a0H. Castain, David\u00a0J. Daniel, Richard\u00a0L. Graham, and Timothy\u00a0S. Woodall. 2004. Open MPI: Goals, Concept, and Design of a Next Generation MPI Implementation. In Recent Advances in Parallel Virtual Machine and Message Passing Interface, Dieter Kranzlm\u00fcller, P\u00e9ter Kacsuk, and Jack Dongarra (Eds.). Springer Berlin Heidelberg, Berlin, Heidelberg, 97\u2013104."},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/HOTI.2015.19"},{"key":"e_1_3_2_2_20_1","volume-title":"Shaheen II. In Proceedings of the Cray User Group Meeting","author":"Hadri Bilel","year":"2015","unstructured":"Bilel Hadri, Samuel Kortas, Saber Feki, Rooh Khurram, and Greg Newby. 2015. Overview of the KAUST\u2019s Cray X40 System \u2013 Shaheen II. In Proceedings of the Cray User Group Meeting. Chicago, USA."},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-05215-1"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1038\/s41586-020-2649-2"},{"volume-title":"ZeroMQ: messaging for many applications. O\u2019Reilly Media","author":"Hintjens Pieter","key":"e_1_3_2_2_23_1","unstructured":"Pieter Hintjens. 2013. ZeroMQ: messaging for many applications. O\u2019Reilly Media, Inc."},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPSW50202.2020.00104"},{"key":"e_1_3_2_2_25_1","volume-title":"2015 USENIX Annual Technical Conference (USENIX ATC 15)","author":"Huang Chien-Chin","year":"2015","unstructured":"Chien-Chin Huang, Qi Chen, Zhaoguo Wang, Russell Power, Jorge Ortiz, Jinyang Li, and Zhen Xiao. 2015. Spartan: A distributed array framework with smart tiling. In 2015 USENIX Annual Technical Conference (USENIX ATC 15). 1\u201315."},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/MCSE.2007.55"},{"key":"e_1_3_2_2_27_1","unstructured":"IEC 60027-2. 2000. Letter symbols to be used in electrical technology - Part 2: Telecommunications and electronics. Standard. International Electrotechnical Commission."},{"key":"e_1_3_2_2_28_1","unstructured":"Intel.com. 2019. Intel\u00ae Xeon\u00ae Processor E5-2698v3. https:\/\/ark.intel.com\/content\/www\/us\/en\/ark\/products\/81060\/intel-xeon-processor-e52698-v3-40m-cache-2-30-ghz.html"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICPP.2012.55"},{"volume-title":"Positioning and Power in Academic Publishing: Players, Agents and Agendas, F.\u00a0Loizides and B.\u00a0Schmidt (Eds.)","author":"Kluyver Thomas","key":"e_1_3_2_2_30_1","unstructured":"Thomas Kluyver, Benjamin Ragan-Kelley, Fernando P\u00e9rez, Brian Granger, Matthias Bussonnier, Jonathan Frederic, Kyle Kelley, Jessica Hamrick, Jason Grout, Sylvain Corlay, Paul Ivanov, Dami\u00e1n Avila, Safia Abdalla, and Carol Willing. 2016. Jupyter Notebooks \u2013 a publishing format for reproducible computational workflows. In Positioning and Power in Academic Publishing: Players, Agents and Agendas, F.\u00a0Loizides and B.\u00a0Schmidt (Eds.). IOS Press, 87 \u2013 90."},{"key":"e_1_3_2_2_31_1","unstructured":"Arvind Krishnamurthy David\u00a0E. Culler and Katherine Yelick. 1998. Empirical Evaluation of Global Memory Support on the Cray-T3D and Cray-T3E. Technical Report. University of California at Berkeley USA. https:\/\/apps.dtic.mil\/sti\/pdfs\/ADA538557.pdf"},{"key":"e_1_3_2_2_32_1","volume-title":"Richland, WA","author":"Krishnan Manojkumar","year":"2012","unstructured":"Manojkumar Krishnan, Bruce Palmer, Abhinav Vishnu, Sriram Krishnamoorthy, Jeff Daily, and Daniel Chavarria. 2012. The Global Arrays user manual. Pacific Northwest National Laboratory, Richland, WA (2012). https:\/\/hpc.pnnl.gov\/globalarrays\/papers\/GA-UserManual-Main.pdf"},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/2833157.2833162"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.25080\/Majora-92bf1922-00a"},{"key":"e_1_3_2_2_35_1","volume-title":"13th USENIX symposium on operating systems design and implementation (OSDI 18)","author":"Moritz Philipp","year":"2018","unstructured":"Philipp Moritz, Robert Nishihara, Stephanie Wang, Alexey Tumanov, Richard Liaw, Eric Liang, Melih Elibol, Zongheng Yang, William Paul, Michael\u00a0I Jordan, 2018. Ray: A distributed framework for emerging AI applications. In 13th USENIX symposium on operating systems design and implementation (OSDI 18). 561\u2013577."},{"key":"e_1_3_2_2_36_1","unstructured":"Jesse Noller and Richard Oudkerk. 2008. Addition of the multiprocessing package to the standard library. PEP 371. https:\/\/www.python.org\/dev\/peps\/pep-0371\/"},{"key":"e_1_3_2_2_37_1","unstructured":"NVIDIA. 2023. An Aspiring Drop-In Replacement for NumPy at Scale. https:\/\/github.com\/nv-legate\/cunumeric"},{"key":"e_1_3_2_2_38_1","unstructured":"NVIDIA. 2023. NVIDIA cuNumeric. https:\/\/developer.nvidia.com\/cunumeric"},{"key":"e_1_3_2_2_39_1","unstructured":"Ryosuke Okuta Yuya Unno Daisuke Nishino Shohei Hido and Crissman Loomis. 2017. CuPy: A NumPy-Compatible Library for NVIDIA GPU Calculations. In Proceedings of Workshop on Machine Learning Systems (LearningSys) in The Thirty-first Annual Conference on Neural Information Processing Systems (NIPS). http:\/\/learningsys.org\/nips17\/assets\/papers\/paper_16.pdf"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/HiPC.2017.00037"},{"key":"e_1_3_2_2_41_1","unstructured":"Brian Quinlan. 2009. futures - execute computations asynchronously. PEP 3148. https:\/\/www.python.org\/dev\/peps\/pep-3148\/"},{"key":"e_1_3_2_2_42_1","volume-title":"CFFI: C Foreign Function Interface for Python. https:\/\/cffi.readthedocs.io\/","author":"Rigo Armin","year":"2022","unstructured":"Armin Rigo and Maciej Fijalkowski. 2022. CFFI: C Foreign Function Interface for Python. https:\/\/cffi.readthedocs.io\/"},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.25080\/Majora-7b98e3ed-013"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2022.3225481"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.21105\/joss.05444"},{"volume-title":"OpenSHMEM and Related Technologies. OpenSHMEM in the Era of Exascale and Smart Networks","author":"Si Min","key":"e_1_3_2_2_46_1","unstructured":"Min Si, Huansong Fu, Jeff\u00a0R. Hammond, and Pavan Balaji. 2022. OpenSHMEM over MPI as a Performance Contender: Thorough Analysis and Optimizations. In OpenSHMEM and Related Technologies. OpenSHMEM in the Era of Exascale and Smart Networks, Stephen Poole, Oscar Hernandez, Matthew Baker, and Tony Curtis (Eds.). Springer International Publishing, Cham, 39\u201360."},{"key":"e_1_3_2_2_47_1","volume-title":"Cray User Group Conference. https:\/\/cug.org\/5-publications\/proceedings_attendee_lists\/CUG10CD\/pages\/1-program\/final_program\/CUG10_Proceedings\/pages\/authors\/01-5Monday\/03B-tenBruggencate-Paper-2.pdf","author":"Bruggencate Monika","year":"2010","unstructured":"Monika ten Bruggencate and Duncan Roweth. 2010. DMAPP - an API for one-sided program models on Baker systems. In Cray User Group Conference. https:\/\/cug.org\/5-publications\/proceedings_attendee_lists\/CUG10CD\/pages\/1-program\/final_program\/CUG10_Proceedings\/pages\/authors\/01-5Monday\/03B-tenBruggencate-Paper-2.pdf"},{"key":"e_1_3_2_2_48_1","volume-title":"Theano: A Python framework for fast computation of mathematical expressions. arXiv e-prints abs\/1605.02688 (May","author":"Team Theano Development","year":"2016","unstructured":"Theano Development Team. 2016. Theano: A Python framework for fast computation of mathematical expressions. arXiv e-prints abs\/1605.02688 (May 2016). http:\/\/arxiv.org\/abs\/1605.02688"},{"volume-title":"Comparing Runtime Systems with Exascale Ambitions Using the Parallel Research Kernels","author":"der Wijngaart F. Van","key":"e_1_3_2_2_49_1","unstructured":"Rob\u00a0F. Van der Wijngaart, Abdullah Kayi, Jeff\u00a0R. Hammond, Gabriele Jost, Tom St.\u00a0John, Srinivas Sridharan, Timothy\u00a0G. Mattson, John Abercrombie, and Jacob Nelson. 2016. Comparing Runtime Systems with Exascale Ambitions Using the Parallel Research Kernels. In High Performance Computing, Julian\u00a0M. Kunkel, Pavan Balaji, and Jack Dongarra (Eds.). Springer International Publishing, Cham, 321\u2013339."},{"key":"e_1_3_2_2_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPEC.2014.7040972"},{"key":"e_1_3_2_2_51_1","doi-asserted-by":"publisher","DOI":"10.1038\/s41592-019-0686-2"},{"key":"e_1_3_2_2_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/PGAS.2015.20"}],"event":{"name":"SC-W 2023: Workshops of The International Conference on High Performance Computing, Network, Storage, and Analysis","acronym":"SC-W 2023","location":"Denver CO USA"},"container-title":["Proceedings of the SC '23 Workshops of the International Conference on High Performance Computing, Network, Storage, and Analysis"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3624062.3624602","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3624062.3624602","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T03:02:53Z","timestamp":1755745373000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3624062.3624602"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,11,12]]},"references-count":52,"alternative-id":["10.1145\/3624062.3624602","10.1145\/3624062"],"URL":"https:\/\/doi.org\/10.1145\/3624062.3624602","relation":{},"subject":[],"published":{"date-parts":[[2023,11,12]]},"assertion":[{"value":"2023-11-12","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}