{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,10]],"date-time":"2026-01-10T02:17:39Z","timestamp":1768011459554,"version":"3.49.0"},"publisher-location":"New York, NY, USA","reference-count":35,"publisher":"ACM","funder":[{"name":"National Science Foundation (NSF)","award":["2103986, 1931512"],"award-info":[{"award-number":["2103986, 1931512"]}]},{"name":"The US Department of Energy (DOE)","award":["DE-AC02-06CH11357"],"award-info":[{"award-number":["DE-AC02-06CH11357"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,11,16]]},"DOI":"10.1145\/3731599.3767587","type":"proceedings-article","created":{"date-parts":[[2025,11,7]],"date-time":"2025-11-07T16:13:44Z","timestamp":1762532024000},"page":"2245-2256","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Integrating and Characterizing HPC Task Runtime Systems for hybrid AI-HPC workloads"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7228-4327","authenticated-orcid":false,"given":"Andre","family":"Merzky","sequence":"first","affiliation":[{"name":"RADICAL-Computing Inc., Wilmington, DE, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2357-7382","authenticated-orcid":false,"given":"Mikhail","family":"Titov","sequence":"additional","affiliation":[{"name":"Brookhaven National Laboratory, Upton, NY, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0527-1435","authenticated-orcid":false,"given":"Matteo","family":"Turilli","sequence":"additional","affiliation":[{"name":"Rutgers University, New Brunswick, NJ, USA and IE University, Madrid, Spain"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5040-026X","authenticated-orcid":false,"given":"Shantenu","family":"Jha","sequence":"additional","affiliation":[{"name":"Rutgers University, New Brunswick, NJ, USA; Princeton Plasma Physics Laboratory, Princeton, NJ, USA and Princeton University, Princeton, NJ, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,11,15]]},"reference":[{"key":"e_1_3_3_1_2_2","doi-asserted-by":"crossref","unstructured":"Dong\u00a0H Ahn Ned Bass Albert Chu Jim Garlick Mark Grondona Stephen Herbein Helgi\u00a0I Ing\u00f3lfsson Joseph Koning Tapasya Patki Thomas\u00a0RW Scogland et\u00a0al. 2020. Flux: Overcoming scheduling challenges for exascale workflows. Future Generation Computer Systems 110 (2020) 202\u2013213.","DOI":"10.1016\/j.future.2020.04.006"},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"publisher","DOI":"10.1109\/WORKS54523.2021.00012"},{"key":"e_1_3_3_1_4_2","doi-asserted-by":"publisher","DOI":"10.1109\/WORKS56498.2022.00009"},{"key":"e_1_3_3_1_5_2","doi-asserted-by":"publisher","DOI":"10.1145\/3307681.3325400"},{"key":"e_1_3_3_1_6_2","doi-asserted-by":"publisher","DOI":"10.1109\/CCGRID.2005.1558641"},{"key":"e_1_3_3_1_7_2","doi-asserted-by":"crossref","unstructured":"Jumana Dakka Matteo Turilli David\u00a0W Wright Stefan\u00a0J Zasada Vivek Balasubramanian Shunzhou Wan Peter\u00a0V Coveney and Shantenu Jha. 2018. High-throughput binding affinity calculations at extreme scales. BMC bioinformatics 19 (2018) 33\u201345.","DOI":"10.1186\/s12859-018-2506-6"},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"crossref","unstructured":"Ewa Deelman Karan Vahi Mats Rynge Rajiv Mayani Rafael\u00a0Ferreira da Silva George Papadimitriou and Miron Livny. 2019. The evolution of the pegasus workflow management software. Computing in Science & Engineering 21 4 (2019) 22\u201336.","DOI":"10.1109\/MCSE.2019.2919690"},{"key":"e_1_3_3_1_9_2","unstructured":"DragonHPC Team. 2025. Dragon: A Composable Distributed Runtime for HPC and AI Workflows. https:\/\/dragonhpc.org\/."},{"key":"e_1_3_3_1_10_2","volume-title":"Supporting Many Task Workloads on Frontier using PMIx and PRRTE","author":"Elwasif Wael","year":"2023","unstructured":"Wael Elwasif and Thomas Naughton\u00a0III. 2023. Supporting Many Task Workloads on Frontier using PMIx and PRRTE. Technical Report. Oak Ridge National Laboratory (ORNL), Oak Ridge, TN (United States)."},{"key":"e_1_3_3_1_11_2","volume-title":"Proceedings of the 1st International Workshop on Power-Aware Systems and Architectures","author":"Gayen Rituparna","year":"2012","unstructured":"Rituparna Gayen, Seung-Hwan Lim, and Vincent\u00a0W. Freeh. 2012. ALPS: An Active-Passive Scheduler for Power Management in HPC. In Proceedings of the 1st International Workshop on Power-Aware Systems and Architectures. ALPS is more commonly cited as part of Cray XT\/XE system documentation."},{"key":"e_1_3_3_1_12_2","doi-asserted-by":"crossref","unstructured":"Eugen Hruska Vivekanandan Balasubramanian Hyungro Lee Shantenu Jha and Cecilia Clementi. 2020. Extensible and scalable adaptive sampling on supercomputers. Journal of Chemical Theory and Computation 16 12 (2020) 7915\u20137925.","DOI":"10.1021\/acs.jctc.0c00991"},{"key":"e_1_3_3_1_13_2","unstructured":"IBM Corporation. 2020. jsrun Command Reference \u2014 IBM Spectrum LSF. https:\/\/www.ibm.com\/docs\/en\/spectrum-lsf\/10.1.0?topic=lsf-jsrun-command."},{"key":"e_1_3_3_1_14_2","volume-title":"Beowulf Cluster Computing with Linux","author":"Jones James\u00a0Patton","year":"2001","unstructured":"James\u00a0Patton Jones. 2001. PBS: Portable Batch System. In Beowulf Cluster Computing with Linux. The MIT Press."},{"key":"e_1_3_3_1_15_2","doi-asserted-by":"publisher","unstructured":"Hyungro Lee Andre Merzky Li Tan Mikhail Titov Matteo Turilli Dario Alfe Agastya Bhati Alex Brace Austin Clyde Peter Coveney Heng Ma Arvind Ramanathan Rick Stevens Anda Trifan Hubertus\u00a0Van Dam Shunzhou Wan Sean Wilkinson and Shantenu Jha. 2021. Scalable HPC and AI Infrastructure for COVID-19 Therapeutics. Platform for Advanced Scientific Computing Conference (PASC \u201921) July 5\u20139 2021 Geneva Switzerland. ACM New York NY USA (2021). 10.1145\/3468267.3470573.https:\/\/arxiv.org\/abs\/2010.10517.","DOI":"10.1145\/3468267.3470573"},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"crossref","unstructured":"Bertram Lud\u00e4scher Ilkay Altintas Chad Berkley Dan Higgins Efrat Jaeger Matthew Jones Edward\u00a0A Lee Jing Tao and Yang Zhao. 2006. Scientific workflow management and the Kepler system. Concurrency and computation: Practice and experience 18 10 (2006) 1039\u20131065.","DOI":"10.1002\/cpe.994"},{"key":"e_1_3_3_1_17_2","doi-asserted-by":"crossref","unstructured":"Tadashi Maeno Aleksandr Alekseev Fernando\u00a0Harald Barreiro\u00a0Megino Kaushik De Wen Guan Edward Karavakis Alexei Klimentov Tatiana Korchuganova FaHui Lin Paul Nilsson et\u00a0al. 2024. Panda: Production and distributed analysis system. Computing and Software for Big Science 8 1 (2024) 4.","DOI":"10.1007\/s41781-024-00114-3"},{"key":"e_1_3_3_1_18_2","doi-asserted-by":"publisher","DOI":"10.1051\/epjconf\/201921403057"},{"key":"e_1_3_3_1_19_2","doi-asserted-by":"crossref","unstructured":"Andre Merzky Matteo Turilli Mikhail Titov Aymen Al-Saadi and Shantenu Jha. 2021. Design and performance characterization of radical-pilot on leadership-class platforms. IEEE Transactions on Parallel and Distributed Systems 33 4 (2021) 818\u2013829.","DOI":"10.1109\/TPDS.2021.3105994"},{"key":"e_1_3_3_1_20_2","first-page":"561","volume-title":"13th USENIX Symposium on Operating Systems Design and Implementation (OSDI 18)","author":"Moritz Philipp","year":"2018","unstructured":"Philipp Moritz, Robert Nishihara, Stephanie Wang, Alexey Tumanov, Richard Liaw, Eric Liang, Melih Elibol, Zongheng Yang, William Paul, Michael\u00a0I Jordan, et\u00a0al. 2018. Ray: A distributed framework for emerging { AI} applications. In 13th USENIX Symposium on Operating Systems Design and Implementation (OSDI 18). 561\u2013577."},{"key":"e_1_3_3_1_21_2","doi-asserted-by":"publisher","DOI":"10.1145\/3225058.3225128"},{"key":"e_1_3_3_1_22_2","unstructured":"RADICAL Dev Team. 2025. RADICAL AsyncFlow (RAF): Fast and Scalable Asynchronous Workflows on HPC. https:\/\/github.com\/radical-cybertools\/radical.asyncflow."},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"publisher","DOI":"10.25080\/Majora-7b98e3ed-013"},{"key":"e_1_3_3_1_24_2","doi-asserted-by":"publisher","DOI":"10.1145\/3472456.3473524"},{"key":"e_1_3_3_1_25_2","doi-asserted-by":"publisher","DOI":"10.1109\/CSIE.2009.950"},{"key":"e_1_3_3_1_26_2","doi-asserted-by":"publisher","DOI":"10.1145\/3624062.3624277"},{"key":"e_1_3_3_1_27_2","first-page":"88","volume-title":"Workshop on Job Scheduling Strategies for Parallel Processing","author":"Titov Mikhail","year":"2022","unstructured":"Mikhail Titov, Matteo Turilli, Andre Merzky, Thomas Naughton, Wael Elwasif, and Shantenu Jha. 2022. Radical-pilot and pmix\/prrte: Executing heterogeneous workloads at large scale on partitioned hpc resources. In Workshop on Job Scheduling Strategies for Parallel Processing. Springer, 88\u2013107."},{"key":"e_1_3_3_1_28_2","doi-asserted-by":"publisher","DOI":"10.1088\/1742-6596\/119\/6\/062048"},{"key":"e_1_3_3_1_29_2","doi-asserted-by":"crossref","unstructured":"Matteo Turilli Vivek Balasubramanian Andre Merzky Ioannis Paraskevakos and Shantenu Jha. 2019. Middleware building blocks for workflow systems. Computing in Science & Engineering 21 4 (2019) 62\u201375.","DOI":"10.1109\/MCSE.2019.2920048"},{"key":"e_1_3_3_1_30_2","doi-asserted-by":"crossref","unstructured":"Matteo Turilli Mihael Hategan-Marandiuc Mikhail Titov Ketan Maheshwari Aymen Alsaadi Andre Merzky Ramon Arambula Mikhail Zakharchanka Matt Cowan Justin\u00a0M Wozniak et\u00a0al. 2024. ExaWorks software development kit: a robust and scalable collection of interoperable workflows technologies. Frontiers in High Performance Computing 2 (2024) 1394615.","DOI":"10.3389\/fhpcp.2024.1394615"},{"key":"e_1_3_3_1_31_2","doi-asserted-by":"publisher","DOI":"10.1109\/IPDPS.2016.64"},{"key":"e_1_3_3_1_32_2","doi-asserted-by":"publisher","DOI":"10.1109\/IPDRM49579.2019.00007"},{"key":"e_1_3_3_1_33_2","doi-asserted-by":"crossref","unstructured":"Matteo Turilli Mark Santcroos and Shantenu Jha. 2018. A comprehensive perspective on pilot-job systems. Comput. Surveys 51 2 (2018) 1\u201332.","DOI":"10.1145\/3177851"},{"key":"e_1_3_3_1_34_2","doi-asserted-by":"publisher","DOI":"10.1109\/CCGrid.2013.99"},{"key":"e_1_3_3_1_35_2","doi-asserted-by":"publisher","DOI":"10.1007\/10968987_3"},{"key":"e_1_3_3_1_36_2","volume-title":"I Workshop on cluster computing","author":"Zhou Songnian","year":"1992","unstructured":"Songnian Zhou. 1992. LSF: Load sharing in large heterogeneous distributed systems. In I Workshop on cluster computing , Vol.\u00a0136."}],"event":{"name":"SC Workshops '25: Workshops of the International Conference for High Performance Computing, Networking, Storage and Analysis","location":"St Louis MO USA","acronym":"SC Workshops '25","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing"]},"container-title":["Proceedings of the SC '25 Workshops of the International Conference for High Performance Computing, Networking, Storage and Analysis"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3731599.3767587","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,9]],"date-time":"2026-01-09T19:28:42Z","timestamp":1767986922000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3731599.3767587"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,15]]},"references-count":35,"alternative-id":["10.1145\/3731599.3767587","10.1145\/3731599"],"URL":"https:\/\/doi.org\/10.1145\/3731599.3767587","relation":{},"subject":[],"published":{"date-parts":[[2025,11,15]]},"assertion":[{"value":"2025-11-15","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}