{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T05:58:09Z","timestamp":1785563889105,"version":"3.56.0"},"reference-count":40,"publisher":"IEEE","license":[{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026,5]]},"DOI":"10.1109\/ccgrid68966.2026.00013","type":"proceedings-article","created":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T18:11:43Z","timestamp":1785521503000},"page":"33-43","source":"Crossref","is-referenced-by-count":0,"title":["Characterizing LLM Inference Energy-Performance Tradeoffs Across Workloads and GPU Scaling"],"prefix":"10.1109","author":[{"given":"Paul Joe","family":"Maliakel","sequence":"first","affiliation":[{"name":"TU Wien"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shashikant","family":"Ilager","sequence":"additional","affiliation":[{"name":"University of Amsterdam"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ivona","family":"Brandic","sequence":"additional","affiliation":[{"name":"TU Wien"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"issue":"1","key":"ref1","article-title":"Palm: scaling language modeling with pathways","volume":"24","author":"Chowdhery","year":"2023","journal-title":"J. Mach. Learn. Res."},{"key":"ref2","author":"Bommasani","year":"2021","journal-title":"On the opportunities and risks of foundation models"},{"key":"ref3","volume-title":"A survey of large language models","author":"Zhao","year":"2025"},{"key":"ref4","article-title":"Language models are few-shot learners","volume-title":"Proceedings of the 34th International Conference on Neural Information Processing Systems, ser. NIPS \u201920","author":"Brown"},{"key":"ref5","volume-title":"Gpt-4 technical report","author":"OpenAI","year":"2024"},{"key":"ref6","first-page":"3645","article-title":"Energy and policy considerations for deep learning in NLP","volume-title":"Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics","author":"Strubell"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1145\/3381831"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/isca45697.2020.00045"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1145\/3661825"},{"key":"ref10","first-page":"611","article-title":"Efficient memory management for large language model serving with pagedattention","volume-title":"Proceedings of the 29th Symposium on Operating Systems Principles, ser. SOSP \u201923","author":"Li"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.52202\/068431-1189"},{"key":"ref12","first-page":"521","article-title":"Orca: A distributed serving system for Transformer-Based generative models","volume-title":"16th USENIX Symposium on Operating Systems Design and Implementation (OSDI 22)","author":"Yu"},{"key":"ref13","first-page":"4149","article-title":"CommonsenseQA: A question answering challenge targeting commonsense knowledge","volume-title":"Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies","volume":"1","author":"Talmor"},{"key":"ref14","doi-asserted-by":"crossref","first-page":"346","DOI":"10.1162\/tacl_a_00370","article-title":"Did aristotle use a laptop? a question answering benchmark with implicit reasoning strategies","volume":"9","author":"Geva","year":"2021","journal-title":"Transactions of the Association for Computational Linguistics"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.52202\/075280-1693"},{"key":"ref16","first-page":"3119","article-title":"LongBench: A bilingual, multitask benchmark for long context understanding","volume-title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics","volume":"1","author":"Bai"},{"key":"ref17","first-page":"5905","article-title":"QMSum: A new benchmark for query-based multi-domain meeting summarization","volume-title":"Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies. Online: Association for Computational Linguistics","author":"Zhong","year":"2021"},{"key":"ref18","doi-asserted-by":"crossref","first-page":"317","DOI":"10.1162\/tacl_a_00023","article-title":"The NarrativeQA reading comprehension challenge","volume":"6","author":"Ko\u010disk\u00fd","year":"2018","journal-title":"Transactions of the Association for Computational Linguistics"},{"key":"ref19","first-page":"5988","article-title":"Understanding dataset difficulty with V-usable information","volume-title":"Proceedings of the 39th International Conference on Machine Learning, ser. Proceedings of Machine Learning Research","volume":"162","author":"Ethayarajh"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/LCA.2024.3406038"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA59077.2024.00019"},{"key":"ref22","volume-title":"Megatron-lm: Training multi-billion parameter language models using model parallelism","author":"Shoeybi","year":"2020"},{"key":"ref23","first-page":"2924","article-title":"BoolQ: Exploring the surprising difficulty of natural yes\/no questions","volume-title":"Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies","volume":"1","author":"Clark"},{"key":"ref24","first-page":"4791","article-title":"HellaSwag: Can a machine really finish your sentence?","volume-title":"Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics","author":"Zellers"},{"key":"ref25","first-page":"3214","article-title":"TruthfulQA: Measuring how models mimic human falsehoods","volume-title":"Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics","volume":"1","author":"Lin"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref27","volume-title":"Language models are unsupervised multitask learners","author":"Radford","year":"2019"},{"key":"ref28","volume-title":"Opt: Open pre-trained transformer language models","author":"Zhang","year":"2022"},{"key":"ref29","volume":"abs\/2302.13971","author":"Touvron","year":"2023","journal-title":"Llama: Open and efficient foundation language models"},{"key":"ref30","volume-title":"Qwen2.5 technical report","author":"Qwen","year":"2025"},{"key":"ref31","volume-title":"Mistral 7b","author":"Jiang","year":"2023"},{"key":"ref32","volume-title":"Greenllm: Slo-aware dynamic frequency scaling for energy-efficient 1lm serving","author":"Liu","year":"2025"},{"key":"ref33","first-page":"119","article-title":"Zeus: Understanding and optimizing GPU energy consumption of DNN training","volume-title":"20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23)","author":"You"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1145\/3694715.3695970"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1145\/3394885.3431535"},{"key":"ref36","volume-title":"Accelerating llm inference with staged speculative decoding","author":"Spector","year":"2023"},{"key":"ref37","article-title":"Medusa: Simple llm inference acceleration framework with multiple decoding heads","volume-title":"Proceedings of the 41st International Conference on Machine Learning, ser. ICML\u201924. JMLR.org","author":"Cai","year":"2024"},{"key":"ref38","article-title":"Smoothquant: accurate and efficient post-training quantization for large language models","volume-title":"Proceedings of the 40th International Conference on Machine Learning, ser. ICML\u201923. JMLR.org","author":"Xiao","year":"2023"},{"key":"ref39","volume-title":"Frugalgpt: How to use large language models while reducing cost and improving performance","author":"Chen","year":"2023"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/ICPR.2016.7900006"}],"event":{"name":"2026 IEEE 26th International Symposium on Cluster, Cloud and Internet Computing (CCGrid)","location":"Sydney, Australia","start":{"date-parts":[[2026,5,18]]},"end":{"date-parts":[[2026,5,21]]}},"container-title":["2026 IEEE 26th International Symposium on Cluster, Cloud and Internet Computing (CCGrid)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11618979\/11618983\/11619149.pdf?arnumber=11619149","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T05:11:05Z","timestamp":1785561065000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11619149\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5]]},"references-count":40,"URL":"https:\/\/doi.org\/10.1109\/ccgrid68966.2026.00013","relation":{},"subject":[],"published":{"date-parts":[[2026,5]]}}}