{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,13]],"date-time":"2026-01-13T21:24:21Z","timestamp":1768339461202,"version":"3.49.0"},"publisher-location":"New York, NY, USA","reference-count":25,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,5,5]],"date-time":"2025-05-05T00:00:00Z","timestamp":1746403200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,5,5]]},"DOI":"10.1145\/3676151.3719379","type":"proceedings-article","created":{"date-parts":[[2025,5,3]],"date-time":"2025-05-03T00:57:09Z","timestamp":1746233829000},"page":"105-112","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Optimization Strategies for Enhancing Resource Efficiency in Transformers &amp; Large Language Models"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8936-3163","authenticated-orcid":false,"given":"Tom","family":"Wallace","sequence":"first","affiliation":[{"name":"Brock University, St. Catharines, Ontario, Canada"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7296-9510","authenticated-orcid":false,"given":"Beatrice","family":"Ombuki-Berman","sequence":"additional","affiliation":[{"name":"Brock University, St. Catharines, Ontario, Canada"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1435-6297","authenticated-orcid":false,"given":"Naser","family":"Ezzati-Jivan","sequence":"additional","affiliation":[{"name":"Brock University, St. Catharines, Ontario, Canada"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,5,5]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Carbontracker: Tracking and Predicting the Carbon Footprint of Training Deep Learning Models. arXiv:2007.03051 [cs.CY] https:\/\/arxiv.org\/abs\/2007.03051","author":"Wolff Anthony Lasse F.","year":"2020","unstructured":"Lasse F. Wolff Anthony, Benjamin Kanding, and Raghavendra Selvan. 2020. Carbontracker: Tracking and Predicting the Carbon Footprint of Training Deep Learning Models. arXiv:2007.03051 [cs.CY] https:\/\/arxiv.org\/abs\/2007.03051"},{"key":"e_1_3_2_1_2_1","unstructured":"Peter Clark Isaac Cowhey Oren Etzioni Tushar Khot Ashish Sabharwal Carissa Schoenick and Oyvind Tafjord. 2018. Think you have Solved Question Answering? Try ARC the AI2 Reasoning Challenge. arXiv:1803.05457 [cs.AI] https:\/\/arxiv.org\/abs\/1803.05457"},{"key":"e_1_3_2_1_3_1","unstructured":"Tim Dettmers Mike Lewis Younes Belkada and Luke Zettlemoyer. 2022. LLM.int8(): 8-bit Matrix Multiplication for Transformers at Scale. arXiv:2208.07339 [cs.LG] https:\/\/arxiv.org\/abs\/2208.07339"},{"key":"e_1_3_2_1_4_1","volume-title":"8- bit Optimizers via Block-wise Quantization. CoRR abs\/2110.02861","author":"Dettmers Tim","year":"2021","unstructured":"Tim Dettmers, Mike Lewis, Sam Shleifer, and Luke Zettlemoyer. 2021. 8- bit Optimizers via Block-wise Quantization. CoRR abs\/2110.02861 (2021). arXiv:2110.02861 https:\/\/arxiv.org\/abs\/2110.02861"},{"key":"e_1_3_2_1_5_1","unstructured":"Tim Dettmers Artidoro Pagnoni Ari Holtzman and Luke Zettlemoyer. 2023. QLoRA: Efficient Finetuning of Quantized LLMs. arXiv:2305.14314 [cs.LG] https:\/\/arxiv.org\/abs\/2305.14314"},{"key":"e_1_3_2_1_6_1","unstructured":"Tim Dettmers Ruslan Svirschevski Vage Egiazarian Denis Kuznedelev Elias Frantar Saleh Ashkboos Alexander Borzunov Torsten Hoefler and Dan Alistarh. 2023. SpQR: A Sparse-Quantized Representation for Near-Lossless LLM Weight Compression. arXiv:2306.03078 [cs.CL] https:\/\/arxiv.org\/abs\/2306.03078"},{"key":"e_1_3_2_1_7_1","unstructured":"Tim Dettmers and Luke Zettlemoyer. 2023. The case for 4-bit precision: k-bit Inference Scaling Laws. arXiv:2212.09720 [cs.LG] https:\/\/arxiv.org\/abs\/2212.09720"},{"key":"e_1_3_2_1_8_1","unstructured":"Elias Frantar and Dan Alistarh. 2023. SparseGPT: Massive Language Models Can Be Accurately Pruned in One-Shot. arXiv:2301.00774 [cs.LG] https:\/\/arxiv.org\/abs\/2301.00774"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.5281\/zenodo.12608602"},{"key":"e_1_3_2_1_10_1","unstructured":"Yuxian Gu Li Dong Furu Wei and Minlie Huang. 2024. MiniLLM: Knowledge Distillation of Large Language Models. arXiv:2306.08543 [cs.CL] https:\/\/arxiv. org\/abs\/2306.08543"},{"key":"e_1_3_2_1_11_1","unstructured":"Dan Hendrycks Collin Burns Steven Basart Andy Zou Mantas Mazeika Dawn Song and Jacob Steinhardt. 2021. Measuring Massive Multitask Language Understanding. arXiv:2009.03300 [cs.CY] https:\/\/arxiv.org\/abs\/2009.03300"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3613904.3642109"},{"key":"e_1_3_2_1_13_1","unstructured":"Dhiraj Kalamkar Dheevatsa Mudigere Naveen Mellempudi Dipankar Das Kunal Banerjee Sasikanth Avancha Dharma Teja Vooturi Nataraj Jammalamadaka Jianyu Huang Hector Yuen Jiyan Yang Jongsoo Park Alexander Heinecke Evangelos Georganas Sudarshan Srinivasan Abhisek Kundu Misha Smelyanskiy Bharat Kaul and Pradeep Dubey. 2019. A Study of BFLOAT16 for Deep Learning Training. arXiv:1905.12322 [cs.LG] https:\/\/arxiv.org\/abs\/1905.12322"},{"key":"e_1_3_2_1_14_1","unstructured":"Stephanie Lin Jacob Hilton and Owain Evans. 2022. TruthfulQA: Measuring How Models Mimic Human Falsehoods. arXiv:2109.07958 [cs.CL] https:\/\/arxiv.org\/abs\/2109.07958"},{"key":"e_1_3_2_1_15_1","unstructured":"Xinyin Ma Gongfan Fang and Xinchao Wang. 2023. LLM-Pruner: On the Structural Pruning of Large Language Models. arXiv:2305.11627 [cs.CL] https:\/\/arxiv.org\/abs\/2305.11627"},{"key":"e_1_3_2_1_16_1","unstructured":"Stephen Merity Caiming Xiong James Bradbury and Richard Socher. 2016. Pointer Sentinel Mixture Models. arXiv:1609.07843 [cs.CL]"},{"key":"e_1_3_2_1_17_1","unstructured":"Paul Michel Omer Levy and Graham Neubig. 2019. Are Sixteen Heads Really Better than One? arXiv:1905.10650 [cs.CL] https:\/\/arxiv.org\/abs\/1905.10650"},{"key":"e_1_3_2_1_18_1","volume-title":"Raviraj Joshi, Marcin Chochowski, Mostofa Patwary, Mohammad Shoeybi, Bryan Catanzaro, Jan Kautz, and Pavlo Molchanov.","author":"Muralidharan Saurav","year":"2024","unstructured":"Saurav Muralidharan, Sharath Turuvekere Sreenivas, Raviraj Joshi, Marcin Chochowski, Mostofa Patwary, Mohammad Shoeybi, Bryan Catanzaro, Jan Kautz, and Pavlo Molchanov. 2024. Compact Language Models via Pruning and Knowledge Distillation. arXiv:2407.14679 [cs.CL] https:\/\/arxiv.org\/abs\/2407.14679"},{"key":"e_1_3_2_1_19_1","volume-title":"Chandra Bhagavatula, and Yejin Choi.","author":"Sakaguchi Keisuke","year":"2019","unstructured":"Keisuke Sakaguchi, Ronan Le Bras, Chandra Bhagavatula, and Yejin Choi. 2019. WinoGrande: An Adversarial Winograd Schema Challenge at Scale. arXiv:1907.10641 [cs.CL] https:\/\/arxiv.org\/abs\/1907.10641"},{"key":"e_1_3_2_1_20_1","unstructured":"Victor Sanh Lysandre Debut Julien Chaumond and Thomas Wolf. 2020. DistilBERT a distilled version of BERT: smaller faster cheaper and lighter. arXiv:1910.01108 [cs.CL] https:\/\/arxiv.org\/abs\/1910.01108"},{"key":"e_1_3_2_1_21_1","unstructured":"Sharath Turuvekere Sreenivas Saurav Muralidharan Raviraj Joshi Marcin Chochowski Mostofa Patwary Mohammad Shoeybi Bryan Catanzaro Jan Kautz and Pavlo Molchanov. 2024. LLM Pruning and Distillation in Practice: The Minitron Approach. arXiv:2408.11796 [cs.CL] https:\/\/arxiv.org\/abs\/2408.11796"},{"key":"e_1_3_2_1_22_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N. Gomez Lukasz Kaiser and Illia Polosukhin. 2023. Attention Is All You Need. arXiv:1706.03762 [cs.CL] https:\/\/arxiv.org\/abs\/1706.03762"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1580"},{"key":"e_1_3_2_1_24_1","unstructured":"Mengzhou Xia Tianyu Gao Zhiyuan Zeng and Danqi Chen. 2024. Sheared LLaMA: Accelerating Language Model Pre-training via Structured Pruning. arXiv:2310.06694 [cs.CL] https:\/\/arxiv.org\/abs\/2310.06694"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"crossref","unstructured":"Rowan Zellers Ari Holtzman Yonatan Bisk Ali Farhadi and Yejin Choi. 2019. HellaSwag: Can a Machine Really Finish Your Sentence? arXiv:1905.07830 [cs.CL] https:\/\/arxiv.org\/abs\/1905.07830","DOI":"10.18653\/v1\/P19-1472"}],"event":{"name":"ICPE '25: 16th ACM\/SPEC International Conference on Performance Engineering","location":"Toronto ON Canada","acronym":"ICPE '25","sponsor":["SIGSOFT ACM Special Interest Group on Software Engineering","SIGMETRICS ACM Special Interest Group on Measurement and Evaluation"]},"container-title":["Proceedings of the 16th ACM\/SPEC International Conference on Performance Engineering"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3676151.3719379","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3676151.3719379","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,25]],"date-time":"2025-09-25T16:22:12Z","timestamp":1758817332000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3676151.3719379"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,5]]},"references-count":25,"alternative-id":["10.1145\/3676151.3719379","10.1145\/3676151"],"URL":"https:\/\/doi.org\/10.1145\/3676151.3719379","relation":{},"subject":[],"published":{"date-parts":[[2025,5,5]]},"assertion":[{"value":"2025-05-05","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}