{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T05:51:58Z","timestamp":1781589118823,"version":"3.54.5"},"reference-count":62,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"2","license":[{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"DOI":"10.13039\/100010002","name":"Federal Ministry of Education and Research","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100010002","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Germany under Project NEUROTEC-II","award":["16ME0398K"],"award-info":[{"award-number":["16ME0398K"]}]},{"name":"Germany under Project NEUROTEC-II","award":["16ME0399"],"award-info":[{"award-number":["16ME0399"]}]},{"name":"Neurosys as part of the Initiative ``Cluster4Future'' funded by the BMBF","award":["03ZU1106CB"],"award-info":[{"award-number":["03ZU1106CB"]}]},{"name":"Phase II: NeuroSys as part of the Initiative ``Clusters4Future'' funded by the Federal Ministry of Research, Technology and Space","award":["03ZU2106CB"],"award-info":[{"award-number":["03ZU2106CB"]}]},{"name":"BMBF: Verbundprojekt: WestAI - AI Service Center West through BMBF","award":["01IS22094E"],"award-info":[{"award-number":["01IS22094E"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE J. Emerg. Sel. Topics Circuits Syst."],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1109\/jetcas.2026.3661247","type":"journal-article","created":{"date-parts":[[2026,2,6]],"date-time":"2026-02-06T20:50:16Z","timestamp":1770411016000},"page":"483-499","source":"Crossref","is-referenced-by-count":0,"title":["Algorithm-Hardware Implications of Softmax Approximations for In-Memory Computing-Based LLM Accelerators"],"prefix":"10.1109","volume":"16","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-4556-3758","authenticated-orcid":false,"given":"Jan","family":"Finkbeiner","sequence":"first","affiliation":[{"name":"RWTH Aachen University, Aachen, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9922-4861","authenticated-orcid":false,"given":"Sebastian","family":"Siegel","sequence":"additional","affiliation":[{"name":"Forschungszentrum J&#x00FC;lich GmbH, J&#x00FC;lich, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1651-1935","authenticated-orcid":false,"given":"Chirag","family":"Sudarshan","sequence":"additional","affiliation":[{"name":"Forschungszentrum J&#x00FC;lich GmbH, J&#x00FC;lich, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-9490-1384","authenticated-orcid":false,"given":"Yuankang","family":"Zhao","sequence":"additional","affiliation":[{"name":"RWTH Aachen University, Aachen, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1382-3677","authenticated-orcid":false,"given":"John","family":"Paul Strachan","sequence":"additional","affiliation":[{"name":"RWTH Aachen University, Aachen, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0332-3273","authenticated-orcid":false,"given":"Emre","family":"Neftci","sequence":"additional","affiliation":[{"name":"RWTH Aachen University, Aachen, Germany"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1038\/s43588-024-00753-x"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1038\/s43588-025-00854-1"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/isscc19947.2020.9062985"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/JSSC.2023.3314433"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/isscc49657.2024.10454323"},{"key":"ref6","first-page":"1","article-title":"22 nm STT-MRAM for reflow and automotive uses with high yield, reliability, and magnetic immunity and with performance and shielding options","author":"Gallagher","year":"2019","journal-title":"IEDM Tech. Dig."},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ISCAS58744.2024.10558086"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/DAC18074.2021.9586134"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1145\/3676536.3676766"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/SOCC.2016.7905501"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.3390\/technologies8030046"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1038\/s41598-021-94691-7"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ICECCME55909.2022.9988065"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICDSP.2018.8631588"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICSICT.2018.8565706"},{"key":"ref16","volume-title":"Language Models Are Unsupervised Multitask Learners","author":"Radford","year":"2019"},{"key":"ref17","article-title":"The llama 3 herd of models","author":"Grattafiori","year":"2024","journal-title":"arXiv:2407.21783"},{"key":"ref18","article-title":"DeepSeek-v2: A strong, economical, and efficient mixture-of-experts language model","author":"Liu","year":"2024","journal-title":"arXiv:2405.04434"},{"key":"ref19","article-title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","author":"Team","year":"2024","journal-title":"arXiv:2403.05530"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.23919\/DATE51398.2021.9474146"},{"key":"ref21","article-title":"Theory, analysis, and best practices for sigmoid self-attention","author":"Ramapuram","year":"2024","journal-title":"arXiv:2409.04431"},{"key":"ref22","first-page":"148","article-title":"FlashDecoding++: Faster large language model inference on GPUs","volume":"6","author":"Hong","year":"2024","journal-title":"Proc. Mach. Learn. Syst."},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.3389\/felec.2022.847069"},{"key":"ref24","article-title":"OpenAI o1 system card","author":"Jaech","year":"2024","journal-title":"arXiv:2412.16720"},{"key":"ref25","article-title":"DeepSeek-r1: Incentivizing reasoning capability in LLMs via reinforcement learning","author":"Guo","year":"2025","journal-title":"arXiv:2501.12948"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/APCCAS.2018.8605654"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-20870-7_7"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1038\/s41928-023-01010-1"},{"key":"ref29","article-title":"Attention is all you need","author":"Vaswani","year":"2017","journal-title":"arXiv:1706.03762"},{"key":"ref30","article-title":"OPT: Open pre-trained transformer language models","author":"Zhang","year":"2022","journal-title":"arXiv:2205.01068"},{"key":"ref31","article-title":"Language models are few-shot learners","author":"Brown","year":"2020","journal-title":"arXiv:2005.14165"},{"key":"ref32","article-title":"LLaMA: Open and efficient foundation language models","author":"Touvron","year":"2023","journal-title":"arXiv:2302.13971"},{"key":"ref33","article-title":"Llama 2: Open foundation and fine-tuned chat models","author":"Touvron","year":"2023","journal-title":"arXiv:2307.09288"},{"key":"ref34","article-title":"FlashAttention-2: Faster attention with better parallelism and work partitioning","author":"Dao","year":"2023","journal-title":"arXiv:2307.08691"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1038\/s41467-023-44365-x"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1145\/3713082.3730381"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/TEC.1962.5219391"},{"key":"ref38","article-title":"SmoothQuant: Accurate and efficient post-training quantization for large language models","author":"Xiao","year":"2022","journal-title":"arXiv:2211.10438"},{"key":"ref39","article-title":"Mamba: Linear-time sequence modeling with selective state spaces","author":"Gu","year":"2023","journal-title":"arXiv:2312.00752"},{"key":"ref40","article-title":"XLSTM: Extended long short-term memory","author":"Beck","year":"2024","journal-title":"arXiv:2405.04517"},{"key":"ref41","article-title":"Pointer sentinel mixture models","author":"Merity","year":"2016","journal-title":"arXiv:1609.07843"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P16-1144"},{"key":"ref43","article-title":"Think you have solved question answering? Try ARC, the AI2 reasoning challenge","author":"Clark","year":"2018","journal-title":"arXiv:1803.05457"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1145\/3474381"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/p19-1472"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6239"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1260"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/MM.2018.112130359"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA52012.2021.00010"},{"key":"ref51","volume-title":"NVIDIA L40 GPU Datasheet","year":"2023"},{"key":"ref52","volume-title":"NVIDIA A100 Tensor Core Gpu Datasheet","year":"2020"},{"key":"ref53","article-title":"Dissecting the NVIDIA Volta GPU architecture via microbenchmarking","author":"Jia","year":"Apr. 2018","journal-title":"arXiv:1804.06826"},{"key":"ref54","article-title":"NVIDIA a100 GPU: Performance and innovation for HPC and AI","volume-title":"Proc. Hot Chips 32","author":"Dally"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/MCSE.2013.95"},{"key":"ref56","article-title":"XKV: Cross-layer SVD for KV-cache compression","author":"Chang","year":"2025","journal-title":"arXiv:2503.18893"},{"key":"ref57","first-page":"396","article-title":"A systematic study of cross-layer KV sharing for efficient LLM inference","volume-title":"Proc. Conf. Nations Americas Chapter Assoc. Comput. Linguistics: Human Lang. Technol.","author":"Wu"},{"key":"ref58","article-title":"Gemma 2: Improving open language models at a practical size","author":"Team","year":"2024","journal-title":"arXiv:2408.00118"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/ISCAS46773.2023.10181406"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.3390\/electronics10091004"},{"key":"ref61","article-title":"Analog implementation of the softmax function","author":"Sillman","year":"2023","journal-title":"arXiv:2305.13649"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/ISCAS56072.2025.11043251"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.17815\/jlsrf-7-182"}],"container-title":["IEEE Journal on Emerging and Selected Topics in Circuits and Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/5503868\/11563572\/11372754.pdf?arnumber=11372754","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T04:58:57Z","timestamp":1781585937000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11372754\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6]]},"references-count":62,"journal-issue":{"issue":"2"},"URL":"https:\/\/doi.org\/10.1109\/jetcas.2026.3661247","relation":{},"ISSN":["2156-3357","2156-3365"],"issn-type":[{"value":"2156-3357","type":"print"},{"value":"2156-3365","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,6]]}}}