{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,3]],"date-time":"2026-06-03T08:11:02Z","timestamp":1780474262641,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":48,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,9,25]],"date-time":"2025-09-25T00:00:00Z","timestamp":1758758400000},"content-version":"vor","delay-in-days":94,"URL":"http:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["IIS-2107200"],"award-info":[{"award-number":["IIS-2107200"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["CNS-2038658"],"award-info":[{"award-number":["CNS-2038658"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,6,23]]},"DOI":"10.1145\/3711875.3729145","type":"proceedings-article","created":{"date-parts":[[2025,10,2]],"date-time":"2025-10-02T19:30:22Z","timestamp":1759433422000},"page":"196-208","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["DAF: An Efficient End-to-End Dynamic Activation Framework for on-Device DNN Training"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9710-6116","authenticated-orcid":false,"given":"Renyuan","family":"Liu","sequence":"first","affiliation":[{"name":"George Mason University, Fairfax, Virginia, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-9376-0880","authenticated-orcid":false,"given":"Yuyang","family":"Leng","sequence":"additional","affiliation":[{"name":"George Mason University, Fairfax, Virginia, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-8663-0698","authenticated-orcid":false,"given":"Kaiyan","family":"Liu","sequence":"additional","affiliation":[{"name":"Geogre Mason University, Fairfax, Virginia, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2877-2665","authenticated-orcid":false,"given":"Shaohan","family":"Hu","sequence":"additional","affiliation":[{"name":"JPMorganChase, New York, New York, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5912-5620","authenticated-orcid":false,"given":"Chun-Fu (Richard)","family":"Chen","sequence":"additional","affiliation":[{"name":"JPMorganChase, New York, New York, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2843-6941","authenticated-orcid":false,"given":"Peijun","family":"Zhao","sequence":"additional","affiliation":[{"name":"JPMorganChase, New York, New York, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8515-2622","authenticated-orcid":false,"given":"Heechul","family":"Yun","sequence":"additional","affiliation":[{"name":"University of Kansas, Lawrence, Kansas, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7446-1430","authenticated-orcid":false,"given":"Shuochao","family":"Yao","sequence":"additional","affiliation":[{"name":"George Mason University, Fairfax, Virginia, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,9,25]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/3570361.3613280"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/3570361.3613271"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3570361.3592517"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612412"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3643832.3661886"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3636534.3649379"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3636534.3690682"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/3636534.3690676"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3643832.3661894"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3498361.3538948"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3643832.3661858"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3307334.3326081"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3625687.3625798"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3495243.3560519"},{"key":"e_1_3_2_1_15_1","volume-title":"USENIX Workshop on Hot Topics in Edge Computing (HotEdge 18)","author":"A","year":"2018","unstructured":"A Privacy-Preserving deep learning approach for face recognition with edge computing. In USENIX Workshop on Hot Topics in Edge Computing (HotEdge 18), Boston, MA, July 2018. USENIX Association. URL https:\/\/www.usenix.org\/conference\/hotedge18\/presentation\/mao."},{"key":"e_1_3_2_1_16_1","first-page":"22941","article-title":"On-device training under 256kb memory","volume":"35","author":"Lin Ji","year":"2022","unstructured":"Ji Lin, Ligeng Zhu, Wei-Ming Chen, Wei-Chen Wang, Chuang Gan, and Song Han. On-device training under 256kb memory. Advances in Neural Information Processing Systems, 35:22941\u201322954, 2022.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_17_1","volume-title":"Yan Gao, Titouan Parcollet, and Nicholas Donald Lane. Zerofl: Efficient on-device training for federated learning with local sparsity. arXiv preprint arXiv:2208.02507","author":"Qiu Xinchi","year":"2022","unstructured":"Xinchi Qiu, Javier Fernandez-Marques, Pedro PB Gusmao, Yan Gao, Titouan Parcollet, and Nicholas Donald Lane. Zerofl: Efficient on-device training for federated learning with local sparsity. arXiv preprint arXiv:2208.02507, 2022."},{"key":"e_1_3_2_1_18_1","first-page":"1684","volume-title":"International Conference on Machine Learning","author":"Gao Yuanxiang","unstructured":"Yuanxiang Gao, Li Chen, and Baochun Li. Spotlight: Optimizing device placement for training deep neural networks. In International Conference on Machine Learning, pages 1676\u20131684. PMLR, 2018."},{"key":"e_1_3_2_1_19_1","first-page":"1813","volume-title":"International Conference on Machine Learning","author":"Chen Jianfei","unstructured":"Jianfei Chen, Lianmin Zheng, Zhewei Yao, Dequan Wang, Ion Stoica, Michael Mahoney, and Joseph Gonzalez. Actnn: Reducing training memory footprint via 2-bit activation compressed training. In International Conference on Machine Learning, pages 1803\u20131813. PMLR, 2021."},{"key":"e_1_3_2_1_20_1","first-page":"36057","volume-title":"International Conference on Machine Learning","author":"Wang Guanchu","unstructured":"Guanchu Wang, Zirui Liu, Zhimeng Jiang, Ninghao Liu, Na Zou, and Xia Hu. Division: memory efficient training via dual activation precision. In International Conference on Machine Learning, pages 36036\u201336057. PMLR, 2023."},{"key":"e_1_3_2_1_21_1","volume-title":"et al. Flexpoint: An adaptive numerical format for efficient training of deep neural networks. Advances in neural information processing systems, 30","author":"K\u00f6ster Urs","year":"2017","unstructured":"Urs K\u00f6ster, Tristan Webb, Xin Wang, Marcel Nassar, Arjun K Bansal, William Constable, Oguz Elibol, Scott Gray, Stewart Hall, Luke Hornof, et al. Flexpoint: An adaptive numerical format for efficient training of deep neural networks. Advances in neural information processing systems, 30, 2017."},{"key":"e_1_3_2_1_22_1","first-page":"12127","article-title":"Fractionally squeezing bit savings both temporally and spatially for efficient dnn training","volume":"33","author":"Fu Yonggan","year":"2020","unstructured":"Yonggan Fu, Haoran You, Yang Zhao, Yue Wang, Chaojian Li, Kailash Gopalakrishnan, Zhangyang Wang, and Yingyan Lin. Fractrain: Fractionally squeezing bit savings both temporally and spatially for efficient dnn training. Advances in Neural Information Processing Systems, 33:12127\u201312139, 2020.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_23_1","first-page":"14152","volume-title":"International Conference on Machine Learning","author":"Liu Xiaoxuan","unstructured":"Xiaoxuan Liu, Lianmin Zheng, Dequan Wang, Yukuo Cen, Weize Chen, Xu Han, Jianfei Chen, Zhiyuan Liu, Jie Tang, Joey Gonzalez, et al. Gact: Activation compressed training for generic network architectures. In International Conference on Machine Learning, pages 14139\u201314152. PMLR, 2022."},{"key":"e_1_3_2_1_24_1","volume-title":"Vijayalakshmi Srinivasan, and Kailash Gopalakrishnan. Pact: Parameterized clipping activation for quantized neural networks. arXiv preprint arXiv:1805.06085","author":"Choi Jungwook","year":"2018","unstructured":"Jungwook Choi, Zhuo Wang, Swagath Venkataramani, Pierce I-Jen Chuang, Vijayalakshmi Srinivasan, and Kailash Gopalakrishnan. Pact: Parameterized clipping activation for quantized neural networks. arXiv preprint arXiv:1805.06085, 2018."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00240"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3666025.3699348"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3498361.3539765"},{"key":"e_1_3_2_1_28_1","first-page":"17583","volume-title":"International Conference on Machine Learning","author":"Patil Shishir G","unstructured":"Shishir G Patil, Paras Jain, Prabal Dutta, Ion Stoica, and Joseph Gonzalez. Poet: Training neural networks on tiny devices with integrated rematerialization and paging. In International Conference on Machine Learning, pages 17573\u201317583. PMLR, 2022."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581791.3596852"},{"key":"e_1_3_2_1_30_1","volume-title":"Nvidia cub","author":"NVIDIA.","year":"2024","unstructured":"NVIDIA. Nvidia cub, 2024. URL https:\/\/nvidia.github.io\/cccl\/cub\/."},{"key":"e_1_3_2_1_31_1","first-page":"645","volume-title":"Proceedings, Part IV 14","author":"He Kaiming","unstructured":"Kaiming He, Xiangyu Zhang, Shaoqing Ren, and Jian Sun. Identity mappings in deep residual networks. In Computer Vision-ECCV 2016: 14th European Conference, Amsterdam, The Netherlands, October 11\u201314, 2016, Proceedings, Part IV 14, pages 630\u2013645. Springer, 2016."},{"key":"e_1_3_2_1_32_1","volume-title":"Roberta: A robustly optimized bert pretraining approach. arXiv preprint arXiv:1907.11692, 364","author":"Liu Yinhan","year":"2019","unstructured":"Yinhan Liu. Roberta: A robustly optimized bert pretraining approach. arXiv preprint arXiv:1907.11692, 364, 2019."},{"issue":"8","key":"e_1_3_2_1_33_1","first-page":"9","article-title":"Language models are unsupervised multitask learners","volume":"1","author":"Radford Alec","year":"2019","unstructured":"Alec Radford, Jeffrey Wu, Rewon Child, David Luan, Dario Amodei, Ilya Sutskever, et al. Language models are unsupervised multitask learners. OpenAI blog, 1(8):9, 2019.","journal-title":"OpenAI blog"},{"key":"e_1_3_2_1_34_1","volume-title":"Lora: Low-rank adaptation of large language models. arXiv preprint arXiv:2106.09685","author":"Hu Edward J","year":"2021","unstructured":"Edward J Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, Lu Wang, and Weizhu Chen. Lora: Low-rank adaptation of large language models. arXiv preprint arXiv:2106.09685, 2021."},{"key":"e_1_3_2_1_35_1","volume-title":"Throughput of native arithmetic instructions","author":"NVIDIA.","year":"2024","unstructured":"NVIDIA. Throughput of native arithmetic instructions, 2024. URL https:\/\/docs.nvidia.com\/cuda\/cuda-c-programming-guide\/tarithmetic-instructions-throughput-native-arithmetic-instructions."},{"key":"e_1_3_2_1_36_1","unstructured":"ARM. Instruction throughput and latency 2024. URL https:\/\/developer.arm.com\/documentation\/100400\/0002\/floating-point-unit-programmers-model\/instruction-throughput-and-latency."},{"key":"e_1_3_2_1_37_1","volume-title":"et al. Tensorflow: Large-scale machine learning on heterogeneous distributed systems. arXiv preprint arXiv:1603.04467","author":"Abadi Mart\u00edn","year":"2016","unstructured":"Mart\u00edn Abadi, Ashish Agarwal, Paul Barham, Eugene Brevdo, Zhifeng Chen, Craig Citro, Greg S Corrado, Andy Davis, Jeffrey Dean, Matthieu Devin, et al. Tensorflow: Large-scale machine learning on heterogeneous distributed systems. arXiv preprint arXiv:1603.04467, 2016."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1111\/j.1475-3995.2011.00840.x"},{"key":"e_1_3_2_1_39_1","unstructured":"CIFAR-10 Benchmark. Canadian institute for advanced research 10 classes URL https:\/\/paperswithcode.com\/dataset\/cifar-10."},{"key":"e_1_3_2_1_40_1","unstructured":"CIFAR-100 Benchmark. Canadian institute for advanced research 100 classes URL https:\/\/paperswithcode.com\/dataset\/cifar-100."},{"key":"e_1_3_2_1_41_1","unstructured":"ImageNet1K Benchmark. Imagenet: A large-scale hierarchical image database URL https:\/\/paperswithcode.com\/dataset\/imagenet."},{"key":"e_1_3_2_1_42_1","unstructured":"STS Benchmark. Semantic textual similarity URL https:\/\/paperswithcode.com\/dataset\/sts-benchmark."},{"key":"e_1_3_2_1_43_1","unstructured":"MRPC Benchmark. Microsoft research paraphrase corpus URL https:\/\/paperswithcode.com\/dataset\/mrpc."},{"key":"e_1_3_2_1_44_1","unstructured":"SST-2 Benchmark. The stanford sentiment treebank URL https:\/\/paperswithcode.com\/dataset\/sst-2."},{"key":"e_1_3_2_1_45_1","unstructured":"E2E Benchmark. End-to-end nlg challenge URL https:\/\/paperswithcode.com\/dataset\/e2e."},{"key":"e_1_3_2_1_46_1","unstructured":"WebNLG Benchmark. Creating training corpora for nlg micro-planners . URL https:\/\/paperswithcode.com\/dataset\/webnlg."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00204"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1145\/3005745.3005750"}],"event":{"name":"MobiSys '25: 23rd Annual International Conference on Mobile Systems, Applications and Services","location":"Hilton Anaheim Anaheim CA USA","acronym":"MobiSys '25","sponsor":["SIGMOBILE ACM Special Interest Group on Mobility of Systems, Users, Data and Computing","SIGOPS ACM Special Interest Group on Operating Systems"]},"container-title":["Proceedings of the 23rd Annual International Conference on Mobile Systems, Applications and Services"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3711875.3729145","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3711875.3729145","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,2]],"date-time":"2025-10-02T19:32:38Z","timestamp":1759433558000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3711875.3729145"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,23]]},"references-count":48,"alternative-id":["10.1145\/3711875.3729145","10.1145\/3711875"],"URL":"https:\/\/doi.org\/10.1145\/3711875.3729145","relation":{},"subject":[],"published":{"date-parts":[[2025,6,23]]},"assertion":[{"value":"2025-09-25","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}