{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,27]],"date-time":"2025-08-27T00:15:22Z","timestamp":1756253722666,"version":"3.44.0"},"publisher-location":"New York, NY, USA","reference-count":37,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,6,16]]},"DOI":"10.1145\/3743642.3743649","type":"proceedings-article","created":{"date-parts":[[2025,7,3]],"date-time":"2025-07-03T09:05:16Z","timestamp":1751533516000},"page":"1-10","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Invited Paper: Rethinking Benchmarks for Parallel Machine Learning Techniques: Integrating Qualitative and Quantitative Evaluation Metrics"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-5517-8749","authenticated-orcid":false,"given":"Abdulfatah","family":"Bahbouh","sequence":"first","affiliation":[{"name":"Department of Computer Science, University of Texas at Arlington, Arlington, Tx"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4281-0143","authenticated-orcid":false,"given":"Ishfaq","family":"Ahmad","sequence":"additional","affiliation":[{"name":"Department of Computer Science, University of Texas at Arlington, Arlington, Tx"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,7,3]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2022.3193690"},{"key":"e_1_3_2_1_2_1","volume-title":"May.","author":"Alnaasan N.","year":"2022","unstructured":"N. Alnaasan, A. Jain, A. Shafi, H. Subramoni, and D. K. Panda, \"OMB-Py: Python Micro-Benchmarks for Evaluating Performance of MPI Libraries on HPC Systems,\" IEEE International Parallel and Distributed Processing Symposium Workshops, May. 2022."},{"key":"e_1_3_2_1_3_1","volume-title":"American Federation of Information Processing Societies Conference","volume":"30","author":"Amdahl G. M.","year":"1967","unstructured":"G. M. Amdahl, \"Validity of the Single Processor Approach to Achieving Large Scale Computing Capabilities,\" American Federation of Information Processing Societies Conference, vol. 30, Atlantic City, NJ, USA, Apr. 1967."},{"key":"e_1_3_2_1_4_1","volume-title":"Denver","author":"Atchley S.","year":"2023","unstructured":"S. Atchley, C. Zimmer, J. R. Lange, D. E. Bernholdt, V. G. Melesse Vergara, T. Beck, M. J. Brim, R. Budiardja, S. Chandrasekaran, et al., \"Frontier: Exploring Exascale,\" International Conference for High-Performance Computing, Networking, Storage and Analysis, Denver, United States, 2023."},{"key":"e_1_3_2_1_5_1","volume-title":"HyPar-Flow: Exploiting MPI and Keras for Scalable Hybrid-Parallel DNN Training with TensorFlow,\" International Conference on High Performance Computing","author":"Awan A. A.","year":"2020","unstructured":"A. A. Awan, A. Jain, Q. Anthony, H. Subramoni, and D. K. Panda, \"HyPar-Flow: Exploiting MPI and Keras for Scalable Hybrid-Parallel DNN Training with TensorFlow,\" International Conference on High Performance Computing, 2020."},{"key":"e_1_3_2_1_6_1","volume-title":"International Conference for High Performance Computing, Networking, Storage and Analysis","author":"Bauman P. T.","year":"2023","unstructured":"P. T. Bauman, R. D. Budiardja, D. Bykov, N. Chalmers, J. Chen, N. Curtis, M. Day, M. Eisenbach, L. Esclapez, A. Fanfarillo, W. Freitag, N. Frontiere, et al., \"Experiences Readying Applications for Exascale,\" International Conference for High Performance Computing, Networking, Storage and Analysis, Denver, United States, 2023."},{"key":"e_1_3_2_1_7_1","volume-title":"Maximizing Parallelism in Distributed Training for Huge Neural Networks,\" arXiv","author":"Bian Z.","year":"2021","unstructured":"Z. Bian, Q. Xu, B. Wang, and Y. You, \"Maximizing Parallelism in Distributed Training for Huge Neural Networks,\" arXiv, 2021."},{"key":"e_1_3_2_1_8_1","volume-title":"Quasi-Recurrent Neural Networks,\" International Conference on Learning Representations","author":"Bradbury J.","year":"2017","unstructured":"J. Bradbury, S. Merity, C. Xiong and R. Socher, \"Quasi-Recurrent Neural Networks,\" International Conference on Learning Representations, 2017."},{"key":"e_1_3_2_1_9_1","volume-title":"Language Models are Few-Shot Learners,\" arXiv","author":"Brown T. B.","year":"2020","unstructured":"T. B. Brown, B. Mann, N. Ryder, M. Subbiah, J. D. Kaplan, P. Dhariwal, et al., \"Language Models are Few-Shot Learners,\" arXiv, 2020."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/TSE.2009.70"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10710-017-9314-z"},{"key":"e_1_3_2_1_12_1","volume-title":"Open Graph Benchmark: Datasets for Machine Learning on Graphs,\" Advances in Neural Information Processing Systems","author":"Hu W.","year":"2020","unstructured":"W. Hu, M. Fey, M. Zitnik, Y. Dong, H. Ren, B. Liu, M. Catasta, and J. Leskovec, \"Open Graph Benchmark: Datasets for Machine Learning on Graphs,\" Advances in Neural Information Processing Systems, 2020."},{"key":"e_1_3_2_1_13_1","first-page":"10","volume-title":"HPC AI500: A Benchmark Suite for HPC AI Systems,\" Lecture notes in computer science","author":"Jiang Z.","year":"2019","unstructured":"Z. Jiang, W. Gao, L. Wang, X. Xiong, Y. Zhang, X. Wen, C. Luo, H. Ye, Y. Zhang, S. Feng, K. Li, W. Xu, and J. Zhan, \"HPC AI500: A Benchmark Suite for HPC AI Systems,\" Lecture notes in computer science, pp. 10--22, Jan. 2019."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.tbench.2022.100083"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11704-022-2096-3"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1134\/S1995080223120211"},{"key":"e_1_3_2_1_17_1","volume-title":"Measuring Algorithmic Interpretability: A Human-Learning-Based Framework and the Corresponding Cognitive Complexity Score,\" arXiv preprint","author":"Lalor J. P.","year":"2022","unstructured":"J. P. Lalor and H. Guo, \"Measuring Algorithmic Interpretability: A Human-Learning-Based Framework and the Corresponding Cognitive Complexity Score,\" arXiv preprint, 2022."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1038\/nature14539"},{"key":"e_1_3_2_1_19_1","first-page":"1","volume-title":"Storage and Analysis","author":"Liu Z.","year":"2023","unstructured":"Z. Liu, S. Cheng, H. Zhou, and Y. You, \"Hanayo: Harnessing Wave-like Pipeline Parallelism for Enhanced Large Model Training Efficiency,\" International Conference for High Performance Computing, Networking, Storage and Analysis, pp. 1--13, 2023."},{"key":"e_1_3_2_1_20_1","volume-title":"Euromicro International Conference on Parallel, Distributed and Network-Based Processing","author":"Marowka A.","year":"2021","unstructured":"A. Marowka, \"Toward a Better Performance Portability Metric,\" Euromicro International Conference on Parallel, Distributed and Network-Based Processing, Valladolid, Spain, 2021."},{"key":"e_1_3_2_1_21_1","first-page":"17","article-title":"Metrics to Measure Code Complexity Based on Software Design: Practical Evaluation","volume":"14","author":"Masmali O.","year":"2022","unstructured":"O. Masmali, O. Badreddin, and R. Khandoker, \"Metrics to Measure Code Complexity Based on Software Design: Practical Evaluation,\" International Journal of Software Engineering and Applications, vol. 14, pp. 17--30, Jan. 2022.","journal-title":"International Journal of Software Engineering and Applications"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2020.2974843"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3363554"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2021.01.125"},{"key":"e_1_3_2_1_25_1","first-page":"1","article-title":"PipeDream: Generalized Pipeline Parallelism for DNN Training","author":"Narayanan D.","year":"2019","unstructured":"D. Narayanan, A. Harlap, A. Phanishayee, V. Seshadri, N. R. Devanur, G. R. Ganger, P. B. Gibbons, and M. Zaharia, \"PipeDream: Generalized Pipeline Parallelism for DNN Training,\" ACM Symposium on Operating Systems Principles, pp. 1--15, Nov. 2019.","journal-title":"ACM Symposium on Operating Systems Principles"},{"key":"e_1_3_2_1_26_1","unstructured":"OpenAI \"OpenAI o3 Breakthrough on ARC-AGI-Pub \" ARC Prize Dec. 2024."},{"issue":"7","key":"e_1_3_2_1_27_1","first-page":"1641","article-title":"The Case for Strong Scaling in Deep Learning: Training Large 3D CNNs with Hybrid Parallelism","volume":"32","author":"Oyama Y.","year":"2021","unstructured":"Y. Oyama, N. Maruyama, N. Dryden, E. McCarthy, P. Harrington, J. Balewski, S. Matsuoka, P. Nugent, and B. V. Essen, \"The Case for Strong Scaling in Deep Learning: Training Large 3D CNNs with Hybrid Parallelism,\" IEEE Transactions on Parallel and Distributed Systems, vol. 32, no. 7, pp. 1641--1652, July. 2021.","journal-title":"IEEE Transactions on Parallel and Distributed Systems"},{"key":"e_1_3_2_1_28_1","volume-title":"Networking, Storage, and Analysis","author":"Pennycook J.","year":"2016","unstructured":"J. Pennycook, J. Sewall, and V. W. Lee, \"A Metric for Performance Portability,\" International Conference for High Performance Computing, Networking, Storage, and Analysis, 2016."},{"issue":"6","key":"e_1_3_2_1_29_1","first-page":"1203","article-title":"Implications of a Metric for Performance Portability","volume":"33","author":"Pennycook S. J.","year":"2019","unstructured":"S. J. Pennycook, C. J. Hughes, M. D. Wright, and S. A. Jarvis, \"Implications of a Metric for Performance Portability,\" International Journal of High-Performance Computing Applications, vol. 33, no. 6, pp. 1203--1220, Nov. 2019.","journal-title":"International Journal of High-Performance Computing Applications"},{"key":"e_1_3_2_1_30_1","first-page":"446","volume-title":"International Symposium on Computer Architecture","author":"Reddi V. J.","year":"2020","unstructured":"V. J. Reddi et al., \"MLPerf Inference Benchmark,\" International Symposium on Computer Architecture, pp. 446--459, 2020."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.26599\/BDMA.2021.9020004"},{"key":"e_1_3_2_1_32_1","first-page":"93","volume-title":"Assembly and Circuits Technology Conference","author":"Sadrossadat S. A.","year":"2021","unstructured":"S. A. Sadrossadat and Z. Naghibi, \"Parallelizing Time-Delay Recurrent Neural Network Modeling Technique on Multi-Core Architectures,\" International Microsystems, Packaging, Assembly and Circuits Technology Conference, pp. 93--96, 2021."},{"key":"e_1_3_2_1_33_1","volume-title":"Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism,\" arXiv","author":"Shoeybi M.","year":"2020","unstructured":"M. Shoeybi, M. Patwary, R. Puri, P. LeGresley, J. Casper, and B. Catanzaro, \"Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism,\" arXiv, 2020."},{"key":"e_1_3_2_1_34_1","first-page":"6000","article-title":"Attention is All You Need","author":"Vaswani A.","year":"2017","unstructured":"A. Vaswani, N. Shazeer, N. Parmar, J. Uszkoreit, L. Jones, A. N. Gomez, \u0141. Kaiser, and I. Polosukhin, \"Attention is All You Need,\" Advances in Neural Information Processing Systems, pp. 6000--6010, 2017.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPDS.2022.3228733"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-25158-0_12"},{"key":"e_1_3_2_1_37_1","volume-title":"Nov.","author":"Zini J.","year":"2019","unstructured":"J. Zini, Y. Rizk, and M. Awad, \"An Optimized and Energy-Efficient Parallel Implementation of Non-Iteratively Trained Recurrent Neural Networks,\" arXiv, Nov. 2019."}],"event":{"name":"PODC '25: ACM Symposium on Principles of Distributed Computing","sponsor":["SIGOPS ACM Special Interest Group on Operating Systems","SIGACT ACM Special Interest Group on Algorithms and Computation Theory"],"location":"Huatulco Mexico","acronym":"PODC '25"},"container-title":["Proceedings of the Advanced tools programming languages and PLatforms for Implementing and Evaluating algorithms for Distributed systems on ZZZ"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3743642.3743649","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,26]],"date-time":"2025-08-26T12:46:23Z","timestamp":1756212383000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3743642.3743649"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,16]]},"references-count":37,"alternative-id":["10.1145\/3743642.3743649","10.1145\/3743642"],"URL":"https:\/\/doi.org\/10.1145\/3743642.3743649","relation":{},"subject":[],"published":{"date-parts":[[2025,6,16]]},"assertion":[{"value":"2025-07-03","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}