{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,5]],"date-time":"2026-06-05T15:52:24Z","timestamp":1780674744044,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":21,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61934002, 62234008"],"award-info":[{"award-number":["61934002, 62234008"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,6,30]]},"DOI":"10.1145\/3716368.3735147","type":"proceedings-article","created":{"date-parts":[[2025,6,27]],"date-time":"2025-06-27T13:58:23Z","timestamp":1751032703000},"page":"450-456","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["VLSUMaP: A High-Performance Matrix Processor with Virtually Expanded LSU Boosting HBM Bandwidth Utilization"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-5168-4683","authenticated-orcid":false,"given":"Xin","family":"Yang","sequence":"first","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-4973-2199","authenticated-orcid":false,"given":"Xinjie","family":"Kong","sequence":"additional","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4523-4201","authenticated-orcid":false,"given":"Kaixuan","family":"Wang","sequence":"additional","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-4658-8192","authenticated-orcid":false,"given":"Xin","family":"Fan","sequence":"additional","affiliation":[{"name":"Hong Kong University of Science and Technology, Hong Kong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-2530-3886","authenticated-orcid":false,"given":"Zikang","family":"Zhou","sequence":"additional","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-7328-957X","authenticated-orcid":false,"given":"Zhuoyuan","family":"Yang","sequence":"additional","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-5575-8440","authenticated-orcid":false,"given":"Zengshi","family":"Wang","sequence":"additional","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5245-0754","authenticated-orcid":false,"given":"Jun","family":"Han","sequence":"additional","affiliation":[{"name":"Fudan University, Shanghai, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,6,29]]},"reference":[{"key":"e_1_3_3_1_2_2","doi-asserted-by":"crossref","unstructured":"Nathan Binkert Bradford Beckmann Gabriel Black Steven\u00a0K Reinhardt Ali Saidi Arkaprava Basu Joel Hestness Derek\u00a0R Hower Tushar Krishna Somayeh Sardashti et\u00a0al. 2011. The gem5 simulator. ACM SIGARCH computer architecture news 39 2 (2011) 1\u20137.","DOI":"10.1145\/2024716.2024718"},{"key":"e_1_3_3_1_3_2","unstructured":"Jacob Devlin. 2018. Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1810.04805 (2018)."},{"key":"e_1_3_3_1_4_2","unstructured":"Zane Durante Qiuyuan Huang Naoki Wake Ran Gong Jae\u00a0Sung Park Bidipta Sarkar Rohan Taori Yusuke Noda Demetri Terzopoulos Yejin Choi et\u00a0al. 2024. Agent ai: Surveying the horizons of multimodal interaction. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2401.03568 (2024)."},{"key":"e_1_3_3_1_5_2","unstructured":"EleutherAI. [n. d.]. GPT-J. https:\/\/huggingface.co\/docs\/transformers\/model_doc\/gptj"},{"key":"e_1_3_3_1_6_2","doi-asserted-by":"publisher","DOI":"10.1109\/DAC18074.2021.9586216"},{"key":"e_1_3_3_1_7_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"publisher","DOI":"10.1145\/1006209.1006211"},{"key":"e_1_3_3_1_9_2","unstructured":"Intel. 2022. Intel\u00ae Architecture Instruction Set Extensions Programming Reference. https:\/\/www.intel.com\/content\/www\/us\/en\/content-details\/671368\/intel-architecture-instruction-set-extensions-programming-reference.html"},{"key":"e_1_3_3_1_10_2","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA56546.2023.10071058"},{"key":"e_1_3_3_1_11_2","doi-asserted-by":"publisher","DOI":"10.1109\/DAC18074.2021.9586257"},{"key":"e_1_3_3_1_12_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISSCC.2014.6757501"},{"key":"e_1_3_3_1_13_2","doi-asserted-by":"crossref","unstructured":"Roktaek Lim Yeongha Lee Raehyun Kim Jaeyoung Choi and Myungho Lee. 2019. Auto-tuning GEMM kernels on the Intel KNL and Intel Skylake-SP processors. The Journal of Supercomputing 75 (2019) 7895\u20137908.","DOI":"10.1007\/s11227-018-2702-1"},{"key":"e_1_3_3_1_14_2","doi-asserted-by":"publisher","DOI":"10.1109\/HOTCHIPS.2015.7477461"},{"key":"e_1_3_3_1_15_2","unstructured":"Maxim Naumov Dheevatsa Mudigere Hao-Jun\u00a0Michael Shi Jianyu Huang Narayanan Sundaraman Jongsoo Park Xiaodong Wang Udit Gupta Carole-Jean Wu Alisson\u00a0G Azzolini et\u00a0al. 2019. Deep learning recommendation model for personalization and recommendation systems. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1906.00091 (2019)."},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"publisher","DOI":"10.1145\/3240302.3240306"},{"key":"e_1_3_3_1_17_2","doi-asserted-by":"publisher","DOI":"10.1109\/MICRO.2003.1253245"},{"key":"e_1_3_3_1_18_2","doi-asserted-by":"crossref","unstructured":"Weikang Qiao Licheng Guo Zhenman Fang Mau-Chung\u00a0Frank Chang and Jason Cong. 2022. TopSort: A high-performance two-phase sorting accelerator optimized on HBM-based FPGAs. IEEE Transactions on Emerging Topics in Computing 11 2 (2022) 404\u2013419.","DOI":"10.1109\/TETC.2022.3228575"},{"key":"e_1_3_3_1_19_2","doi-asserted-by":"publisher","DOI":"10.1109\/ISCA45697.2020.00045"},{"key":"e_1_3_3_1_20_2","volume-title":"NIPS","author":"Waswani A","year":"2017","unstructured":"A Waswani, N Shazeer, N Parmar, J Uszkoreit, L Jones, A Gomez, L Kaiser, and I Polosukhin. 2017. Attention is all you need. In NIPS."},{"key":"e_1_3_3_1_21_2","doi-asserted-by":"publisher","DOI":"10.1109\/FPL60245.2023.00032"},{"key":"e_1_3_3_1_22_2","doi-asserted-by":"publisher","DOI":"10.1109\/SC.2010.52"}],"event":{"name":"GLSVLSI '25: Great Lakes Symposium on VLSI 2025","location":"New Orleans LA USA","acronym":"GLSVLSI '25","sponsor":["SIGDA ACM Special Interest Group on Design Automation"]},"container-title":["Proceedings of the Great Lakes Symposium on VLSI 2025"],"original-title":[],"deposited":{"date-parts":[[2025,6,27]],"date-time":"2025-06-27T14:36:29Z","timestamp":1751034989000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3716368.3735147"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,29]]},"references-count":21,"alternative-id":["10.1145\/3716368.3735147","10.1145\/3716368"],"URL":"https:\/\/doi.org\/10.1145\/3716368.3735147","relation":{},"subject":[],"published":{"date-parts":[[2025,6,29]]},"assertion":[{"value":"2025-06-29","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}