{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,2]],"date-time":"2025-08-02T15:02:12Z","timestamp":1754146932875,"version":"3.41.2"},"publisher-location":"New York, NY, USA","reference-count":13,"publisher":"ACM","funder":[{"name":"National Science Foundation","award":["2005597"],"award-info":[{"award-number":["2005597"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,7,20]]},"DOI":"10.1145\/3708035.3736101","type":"proceedings-article","created":{"date-parts":[[2025,7,18]],"date-time":"2025-07-18T12:10:41Z","timestamp":1752840641000},"page":"1-5","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Evaluating Pretraining Efficiency of Language Models on AI Accelerators"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8226-6237","authenticated-orcid":false,"given":"Mei-Yu","family":"Wang","sequence":"first","affiliation":[{"name":"Pittsburgh Supercomputing Center, Carnegie Mellon University, Pittsburgh, PA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5023-8410","authenticated-orcid":false,"given":"Paola","family":"Buitrago","sequence":"additional","affiliation":[{"name":"Pittsburgh Supercomputing Center, Carnegie Mellon University, Pittsburgh, PA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,7,18]]},"reference":[{"key":"e_1_3_3_2_2_2","unstructured":"Walid Ahmad Elana Simon Seyone Chithrananda Gabriel Grand and Bharath Ramsundar. 2022. ChemBERTa-2: Towards Chemical Foundation Models. arXiv e-prints (Sept. 2022). arxiv:https:\/\/arXiv.org\/abs\/2209.01712"},{"key":"e_1_3_3_2_3_2","doi-asserted-by":"publisher","DOI":"10.1145\/3332186.3332253"},{"key":"e_1_3_3_2_4_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-68035-0_15"},{"key":"e_1_3_3_2_5_2","unstructured":"Seyone Chithrananda Gabriel Grand and Bharath Ramsundar. 2020. ChemBERTa: Large-Scale Self-Supervised Pretraining for Molecular Property Prediction. arXiv e-prints (Oct. 2020). arxiv:https:\/\/arXiv.org\/abs\/2010.09885"},{"key":"e_1_3_3_2_6_2","unstructured":"Tri Dao. 2023. FlashAttention-2: Faster Attention with Better Parallelism and Work Partitioning. arxiv:https:\/\/arXiv.org\/abs\/2307.08691\u00a0[cs.LG] https:\/\/arxiv.org\/abs\/2307.08691"},{"key":"e_1_3_3_2_7_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/n19-1423"},{"key":"e_1_3_3_2_8_2","unstructured":"Aaron Gokaslan and Vanya Cohen. 2019. OpenWebText Corpus. http:\/\/Skylion007.github.io\/OpenWebTextCorpus"},{"key":"e_1_3_3_2_9_2","unstructured":"Aaron Grattafiori and et al.2024. The Llama 3 Herd of Models. arxiv:https:\/\/arXiv.org\/abs\/2407.21783\u00a0[cs.AI] https:\/\/arxiv.org\/abs\/2407.21783"},{"key":"e_1_3_3_2_10_2","doi-asserted-by":"crossref","unstructured":"Sunghwan Kim Jie Chen Tiejun Cheng and et al.2018. PubChem 2019 update: improved access to chemical data. Nucleic Acids Research 47 D1 (10 2018) D1102\u2013D1109.","DOI":"10.1093\/nar\/gky1033"},{"key":"e_1_3_3_2_11_2","unstructured":"Yinhan Liu Myle Ott Naman Goyal Jingfei Du Mandar Joshi Danqi Chen Omer Levy Mike Lewis Luke Zettlemoyer and Veselin Stoyanov. 2019. RoBERTa: A Robustly Optimized BERT Pretraining Approach. CoRR abs\/1907.11692 (2019). arXiv:https:\/\/arXiv.org\/abs\/1907.11692http:\/\/arxiv.org\/abs\/1907.11692"},{"key":"e_1_3_3_2_12_2","unstructured":"Stephen Merity Caiming Xiong James Bradbury and Richard Socher. 2016. Pointer Sentinel Mixture Models. arxiv:https:\/\/arXiv.org\/abs\/1609.07843\u00a0[cs.CL]"},{"key":"e_1_3_3_2_13_2","unstructured":"Hugo Touvron Thibaut Lavril Gautier Izacard Xavier Martinet Marie-Anne Lachaux Timoth\u00e9e Lacroix Baptiste Rozi\u00e8re Naman Goyal Eric Hambro Faisal Azhar Aurelien Rodriguez Armand Joulin Edouard Grave and Guillaume Lample. 2023. LLaMA: Open and Efficient Foundation Language Models. arxiv:https:\/\/arXiv.org\/abs\/2302.13971\u00a0[cs.CL] https:\/\/arxiv.org\/abs\/2302.13971"},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"publisher","DOI":"10.1145\/3569951.3597596"}],"event":{"name":"PEARC '25: Practice and Experience in Advanced Research Computing","location":"Columbus Ohio USA","acronym":"PEARC '25","sponsor":["SIGAPP ACM Special Interest Group on Applied Computing","SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing"]},"container-title":["Practice and Experience in Advanced Research Computing 2025: The Power of Collaboration"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3708035.3736101","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,18]],"date-time":"2025-07-18T12:31:21Z","timestamp":1752841881000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3708035.3736101"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7,18]]},"references-count":13,"alternative-id":["10.1145\/3708035.3736101","10.1145\/3708035"],"URL":"https:\/\/doi.org\/10.1145\/3708035.3736101","relation":{},"subject":[],"published":{"date-parts":[[2025,7,18]]},"assertion":[{"value":"2025-07-18","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}