{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,19]],"date-time":"2026-06-19T06:51:13Z","timestamp":1781851873896,"version":"3.54.5"},"reference-count":28,"publisher":"IEEE","license":[{"start":{"date-parts":[[2026,5,24]],"date-time":"2026-05-24T00:00:00Z","timestamp":1779580800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,5,24]],"date-time":"2026-05-24T00:00:00Z","timestamp":1779580800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026,5,24]]},"DOI":"10.1109\/iscas66217.2026.11562930","type":"proceedings-article","created":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T20:06:41Z","timestamp":1781813201000},"page":"4799-4803","source":"Crossref","is-referenced-by-count":0,"title":["HiMARS: High-Performance Inference for Multi-Modal\/Model AI Assistant via Runtime Scheduling"],"prefix":"10.1109","author":[{"given":"Maoliang","family":"Li","sequence":"first","affiliation":[{"name":"Peking University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiayu","family":"Chen","sequence":"additional","affiliation":[{"name":"Peking University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zihao","family":"Zheng","sequence":"additional","affiliation":[{"name":"Peking University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chenchen","family":"Liu","sequence":"additional","affiliation":[{"name":"Beihang University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Weisheng","family":"Zhao","sequence":"additional","affiliation":[{"name":"Beihang University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiang","family":"Chen","sequence":"additional","affiliation":[{"name":"Peking University"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/HCS55958.2022.9895532"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/HCS59251.2023.10254701"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-acl.1158"},{"key":"ref4","article-title":"Screen.pipe"},{"key":"ref5","article-title":"A survey of multi-tenant deep learning inference on gpu","author":"Yu","year":"2022"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1145\/3498361.3538948"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1145\/3503222.3507752"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/ICCAD51958.2021.9643501"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1145\/3583120.3586953"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1145\/3676536.3676718"},{"key":"ref11","first-page":"8343","article-title":"Nimble: Lightweight and Parallel GPU Task Scheduling for Deep Learning","volume-title":"Proceedings of the Advances in Neural Information Processing Systems (NeurIPS)","volume":"33","author":"Kwon"},{"key":"ref12","first-page":"167","article-title":"Ios: Inter-operator scheduler for cnn acceleration","volume-title":"Proceedings of Machine Learning and Systems (MLSys)","volume":"3","author":"Ding"},{"key":"ref13","first-page":"1069","article-title":"AxoNN: Energy-aware execution of neural network inference on multi-accelerator heterogeneous SoCs","volume-title":"Proceedings of the ACM\/IEEE Design Automation Conference (DAC)","author":"Dagli"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1145\/3627535.3638502"},{"key":"ref15","article-title":"Robust speech recognition via large-scale weak supervision","author":"Radford","year":"2021"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01548"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6812"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2016.2646371"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"ref20","article-title":"Qwen2.5-Coder Technical Report","author":"Hui","year":"2024"},{"key":"ref21","article-title":"NVIDIA Nsight Computing","year":"2020"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1145\/3636534.3649391"},{"key":"ref23","article-title":"Multi-Process Service","year":"2021"},{"key":"ref24","article-title":"Sharegpt-4o: Comprehensive multimodal annotations with gpt-4o","author":"Cui","year":"2024"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1145\/3746027.3755688"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.906"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/icassp48485.2024.10448484"},{"key":"ref28","article-title":"NVIDIA Nsight Systems","year":"2020"}],"event":{"name":"2026 IEEE International Symposium on Circuits and Systems (ISCAS)","location":"Shanghai, China","start":{"date-parts":[[2026,5,24]]},"end":{"date-parts":[[2026,5,28]]}},"container-title":["2026 IEEE International Symposium on Circuits and Systems (ISCAS)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11561899\/11561804\/11562930.pdf?arnumber=11562930","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,19]],"date-time":"2026-06-19T05:55:33Z","timestamp":1781848533000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11562930\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5,24]]},"references-count":28,"URL":"https:\/\/doi.org\/10.1109\/iscas66217.2026.11562930","relation":{},"subject":[],"published":{"date-parts":[[2026,5,24]]}}}