{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T20:18:14Z","timestamp":1783455494932,"version":"3.55.0"},"reference-count":118,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"8","license":[{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. on Mobile Comput."],"published-print":{"date-parts":[[2026,8]]},"DOI":"10.1109\/tmc.2026.3672496","type":"journal-article","created":{"date-parts":[[2026,3,10]],"date-time":"2026-03-10T19:55:43Z","timestamp":1773172543000},"page":"12735-12749","source":"Crossref","is-referenced-by-count":0,"title":["FwdLLM+: Accelerating Forward-Only FedLLM With Low-Rank Perturbations"],"prefix":"10.1109","volume":"25","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-0103-3457","authenticated-orcid":false,"given":"Chen","family":"Peng","sequence":"first","affiliation":[{"name":"State Key Laboratory of Networking and Switching Technology, Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6271-6993","authenticated-orcid":false,"given":"Mengwei","family":"Xu","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Networking and Switching Technology, Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-0209-8421","authenticated-orcid":false,"given":"Zhenyan","family":"Lu","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Networking and Switching Technology, Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-4327-1920","authenticated-orcid":false,"given":"Wei","family":"Liu","sequence":"additional","affiliation":[{"name":"Xiaomi Corporation, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7245-1298","authenticated-orcid":false,"given":"Shangguang","family":"Wang","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Networking and Switching Technology, Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2728-8273","authenticated-orcid":false,"given":"Nicholas D.","family":"Lane","sequence":"additional","affiliation":[{"name":"Department of Computer Science and Technology, University of Cambridge, Cambridge, U.K."}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3301-7074","authenticated-orcid":false,"given":"Qibo","family":"Sun","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Networking and Switching Technology, Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2751-2500","authenticated-orcid":false,"given":"Dongqi","family":"Cai","sequence":"additional","affiliation":[{"name":"Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1145\/3287075"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1145\/3581791.3596853"},{"key":"ref3","article-title":"LLM as a system service on mobile devices","author":"Yin","year":"2024"},{"key":"ref4","first-page":"2158","article-title":"Mobile{BERT}: A compact task-agnostic {BERT} for resource-limited devices","volume-title":"Proc. 58th Annu. Meeting Assoc. Comput. Linguistics","author":"Sun"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1145\/3636534.3649361"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1145\/3570361.3592505"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/icassp48485.2024.10447454"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1145\/3570361.3613277"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095356"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1145\/3510033"},{"key":"ref11","first-page":"934","article-title":"Can public large language models help private cross-device federated learning?","volume-title":"Proc. Findings Assoc. Comput. Linguistics","author":"Wang"},{"key":"ref12","article-title":"Accurate, large minibatch SGD: Training ImageNet in 1 hour","author":"Goyal","year":"2017"},{"key":"ref13","first-page":"1731","article-title":"Train longer, generalize better: Closing the generalization gap in large batch training of neural networks","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"30","author":"Hoffer","year":"2017"},{"key":"ref14","article-title":"The limit of the batch size","author":"You","year":"2020"},{"key":"ref15","article-title":"The cap principle for LLM serving","author":"Zeng","year":"2024"},{"key":"ref16","article-title":"A survey of resource-efficient LLM and multimodal foundation models","author":"Xu","year":"2024"},{"key":"ref17","article-title":"LLaMA: Open and efficient foundation language models","author":"Touvron","year":"2023"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1145\/3498361.3538928"},{"key":"ref19","first-page":"11285","article-title":"Tiny{TL}: Reduce memory, not parameters for efficient on-device learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"33","author":"Cai"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1145\/3386901.3388947"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1145\/3373376.3378505"},{"key":"ref22","article-title":"Android: Low memory killer daemon","year":"2022"},{"key":"ref23","article-title":"Select TensorFlow operators","year":"2026"},{"key":"ref24","article-title":"Federated learning: Collaborative machine learning without centralized training data","year":"2017"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/5.58337"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1038\/nature14539"},{"key":"ref27","first-page":"579","article-title":"{FwdLLM}: Efficient federated finetuning of large language models with perturbed inferences","volume-title":"Proc. USENIX Annu. Tech. Conf.","author":"Xu"},{"key":"ref28","article-title":"Gradients without backpropagation","author":"Baydin","year":"2022"},{"key":"ref29","first-page":"1","article-title":"Scaling forward gradient with local losses","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Ren"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1214\/aos\/1176345632"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2020.3003837"},{"key":"ref32","first-page":"385","article-title":"Online convex optimization in the bandit setting: Gradient descent without a gradient","volume-title":"Proc. 16th Annu. ACM-SIAM Symp. Discrete Algorithms","author":"Flaxman"},{"key":"ref33","first-page":"4171","article-title":"{BERT}: Pre-training of deep bidirectional transformers for language understanding","volume-title":"Proc. Conf. North Amer. Chapter Assoc. Comput. Linguistics, Hum. Lang. Technol.","volume":"1","author":"Devlin"},{"key":"ref34","article-title":"Roberta: A robustly optimized bert pretraining approach","author":"Liu","year":"2019"},{"key":"ref35","article-title":"OPT: Open pre-trained transformer language models","author":"Zhang","year":"2022"},{"key":"ref36","article-title":"Federated large language model: A position paper","author":"Chen","year":"2023"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1145\/3495243.3560545"},{"key":"ref38","first-page":"1","article-title":"LoRA: Low-rank adaptation of large language models","volume-title":"Int. Conf. Learn. Representations","author":"Hu","year":"2021"},{"key":"ref39","first-page":"46","article-title":"{A}dapter{H}ub: A framework for adapting transformers","volume-title":"Proc. Conf. Empirical Methods Natural Lang. Process., System Demonstrations","author":"Pfeiffer"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1145\/3560815"},{"key":"ref41","first-page":"873","article-title":"End the senseless killing: Improving memory management for mobile operating systems","volume-title":"Proc. USENIX Annu. Tech. Conf.","author":"Lebeck","year":"2020"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/IWCMC.2017.7986619"},{"key":"ref43","article-title":"AI-benchmark ranking","author":"ETH Zurich","year":"2024"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1145\/3498361.3538948"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1145\/3447993.3483278"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1145\/3495243.3517017"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1145\/3498361.3538917"},{"key":"ref48","first-page":"19","article-title":"Oort: Efficient federated learning via guided participant selection","volume-title":"Proc. 15th {USENIX} Symp. Operating Syst. Des. Implementation","author":"Lai"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.findings-naacl.13"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-acl.75"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00361"},{"key":"ref52","first-page":"11814","article-title":"FedScale: Benchmarking model and system performance of federated learning at scale","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Lai","year":"2022"},{"key":"ref53","first-page":"12876","article-title":"FjORD: Fair and accurate federated learning under heterogeneous targets with ordered dropout","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Horvath","year":"2021"},{"key":"ref54","article-title":"DPZV: Elevating the tradeoff between privacy and utility in zeroth-ordervertical federated learning","author":"Zhang","year":"2025"},{"key":"ref55","first-page":"1569","article-title":"Unlocking the power of differentially private zeroth-order optimization for fine-tuning {LLMs}","volume-title":"Proc. 34th USENIX Secur. Symp.","author":"Bao","year":"2025"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1145\/3133956.3133982"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-032-16165-9_15"},{"key":"ref58","first-page":"1","article-title":"Private fine-tuning of large language models with zeroth-order optimization","author":"Tang","year":"2025","journal-title":"Trans. Mach. Learn. Res."},{"key":"ref59","first-page":"1","article-title":"FedFwd: Federated learning without backpropagation","volume-title":"Proc. Federated Learn. Anal. Practice, Algorithms, Syst., Appl., Opportunities","author":"Park","year":"2023"},{"key":"ref60","first-page":"1701","article-title":"Fine-tuning happens in tiny subspaces: Exploring intrinsic task-specific subspaces of pre-trained language models","volume-title":"Proc. 61st Annu. Meeting Assoc. Comput. Linguistics","volume":"1","author":"Zhang"},{"key":"ref61","first-page":"61\\,121\u201361\\,143","article-title":"{G}a{L}ore: Memory-efficient {LLM} training by gradient low-rank projection","volume-title":"Proc. 41st Int. Conf. Mach. Learn.","volume":"235","author":"Zhao"},{"key":"ref62","first-page":"53038","article-title":"Fine-tuning language models with just forward passes","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"36","author":"Malladi","year":"2023"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1007\/s10107-015-0871-8"},{"key":"ref64","first-page":"59\\,173\u201359\\,190","article-title":"Revisiting zeroth-order optimization for memory-efficient {LLM} fine-tuning: A benchmark","volume-title":"Proc. 41st Int. Conf. Mach. Learn.","volume":"235","author":"Zhang"},{"key":"ref65","first-page":"7204","article-title":"ZO-AdaMM: Zeroth-order adaptive momentum method for black-box optimization","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"32","author":"Chen","year":"2019"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.21236\/ADA164453"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1017\/9781108679930"},{"key":"ref68","first-page":"1","article-title":"Enhancing zeroth-order fine-tuning for language models with low-rank structures","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Chen"},{"key":"ref69","article-title":"functorch","year":"2026"},{"key":"ref70","article-title":"AutoGPTQ","year":"2023"},{"key":"ref71","article-title":"LiteRT (formerly TensorFlow Lite)","year":"2024"},{"key":"ref72","article-title":"llama.cpp","year":"2023"},{"key":"ref73","first-page":"1","article-title":"{OPTQ}: Accurate quantization for generative pre-trained transformers","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Frantar"},{"key":"ref74","first-page":"1","article-title":"{B}it{F}it: Simple parameter-efficient fine-tuning for transformer-based masked language-models","volume-title":"Proc. 60th Annu. Meeting Assoc. Comput. Linguistics","volume":"2","author":"Ben Zaken"},{"key":"ref75","first-page":"38","article-title":"Transformers: State-of-the-art natural language processing","volume-title":"Proc. Conf. Empirical Methods Natural Lang. Process., System Demonstrations","author":"Wolf"},{"key":"ref76","first-page":"649","article-title":"Character-level convolutional networks for text classification","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"28","author":"Zhang","year":"2015"},{"key":"ref77","first-page":"2383","article-title":"{SQuAD}: 100,000+ questions for machine comprehension of text","volume-title":"Proc. Conf. Empirical Methods Natural Lang. Process","author":"Rajpurkar"},{"key":"ref78","first-page":"3266","article-title":"Superglue: A stickier benchmark for general-purpose language understanding systems","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"32","author":"Wang","year":"2019"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D13-1170"},{"key":"ref80","doi-asserted-by":"publisher","DOI":"10.1007\/11736790_9"},{"key":"ref81","article-title":"The winograd schema challenge","volume-title":"Proc. 13th Int. Conf. Princ. Knowl. Representation Reasoning","author":"Levesque","year":"2012"},{"issue":"2","key":"ref82","first-page":"107","article-title":"The commitmentbank: Investigating projection in naturally occurring discourse","volume-title":"Proc. Sinn und Bedeutung","volume":"23","author":"Marneffe","year":"2019"},{"key":"ref83","first-page":"1267","article-title":"{W}i{C}: The word-in-context dataset for evaluating context-sensitive meaning representations","volume-title":"Proc. Conf. North Amer. Chapter Assoc. Comput. Linguistics, Hum. Lang. Technol.","volume":"1","author":"Pilehvar"},{"key":"ref84","first-page":"90","article-title":"Choice of plausible alternatives: An evaluation of commonsense causal reasoning","volume-title":"Proc. AAAI Spring Symp.: Log. Formalizations Commonsense Reason.","author":"Roemmele","year":"2011"},{"key":"ref85","article-title":"Jetson TX2 Module","year":"2017"},{"key":"ref86","first-page":"8026","article-title":"{PyTorch}: An imperative style, high-performance deep learning library","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"32","author":"Paszke"},{"key":"ref87","first-page":"2790","article-title":"Parameter-efficient transfer learning for NLP","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Houlsby","year":"2019"},{"key":"ref88","article-title":"Federated learning of deep networks using model averaging","author":"McMahan","year":"2016"},{"key":"ref89","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01386"},{"key":"ref90","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.162"},{"key":"ref91","doi-asserted-by":"publisher","DOI":"10.1109\/ISQED57927.2023.10129294"},{"key":"ref92","doi-asserted-by":"publisher","DOI":"10.1093\/acprof:oso\/9780198723844.001.0001"},{"key":"ref93","article-title":"Attention is all you need","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"30","author":"Vaswani","year":"2017"},{"key":"ref94","article-title":"DistilBERT, a distilled version of BERT: Smaller, faster, cheaper and lighter","author":"Sanh","year":"2019"},{"key":"ref95","first-page":"9782","article-title":"DynaBERT: Dynamic BERT with adaptive width and depth","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"33","author":"Hou","year":"2020"},{"key":"ref96","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.537"},{"key":"ref97","first-page":"6035","article-title":"TAG: Gradient attack on transformer-based language models","volume-title":"Proc. Conf. Empirical Methods Natural Lang. Process.","author":"Deng","year":"2021"},{"key":"ref98","first-page":"3600","article-title":"Deep leakage from gradients","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"32","author":"Zhu","year":"2019"},{"key":"ref99","doi-asserted-by":"publisher","DOI":"10.52202\/068431-0555"},{"key":"ref100","article-title":"When foundation model meets federated learning: Motivations, challenges, and future directions","author":"Zhuang","year":"2023"},{"key":"ref101","first-page":"1","article-title":"Towards a unified view of parameter-efficient transfer learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"He","year":"2022"},{"key":"ref102","first-page":"1","article-title":"Gradient following without back-propagation in layered networks","volume-title":"Proc. 1st Int. Conf. Neural Nets","volume":"2","author":"Barto","year":"2022"},{"key":"ref103","article-title":"The forward-forward algorithm: Some preliminary investigations","author":"Hinton","year":"2022"},{"key":"ref104","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i04.5950"},{"key":"ref105","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.259"},{"key":"ref106","article-title":"Does federated learning really need backpropagation?","author":"Feng","year":"2023"},{"key":"ref107","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73226-3_6"},{"key":"ref108","doi-asserted-by":"publisher","DOI":"10.23919\/ecc64448.2024.10590749"},{"key":"ref109","first-page":"374","article-title":"Towards federated learning at scale: System design","volume-title":"Proc. Mach. Learn. Syst.","volume":"1","author":"Bonawitz","year":"2019"},{"key":"ref110","doi-asserted-by":"publisher","DOI":"10.1145\/3442381.3449851"},{"key":"ref111","first-page":"5325","article-title":"Error compensated quantized SGD and its applications to large-scale distributed optimization","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Wu","year":"2018"},{"key":"ref112","first-page":"560","article-title":"signSGD: Compressed optimisation for non-convex problems","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Bernstein","year":"2018"},{"key":"ref113","doi-asserted-by":"publisher","DOI":"10.1109\/ICC.2019.8761315"},{"key":"ref114","doi-asserted-by":"publisher","DOI":"10.1109\/TWC.2020.3031503"},{"key":"ref115","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM42981.2021.9488906"},{"key":"ref116","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOMWKSHPS51825.2021.9484642"},{"key":"ref117","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM42981.2021.9488723"},{"key":"ref118","doi-asserted-by":"publisher","DOI":"10.1145\/3589334.3645341"}],"container-title":["IEEE Transactions on Mobile Computing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/7755\/11595771\/11429542.pdf?arnumber=11429542","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T19:47:08Z","timestamp":1783453628000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11429542\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8]]},"references-count":118,"journal-issue":{"issue":"8"},"URL":"https:\/\/doi.org\/10.1109\/tmc.2026.3672496","relation":{},"ISSN":["1536-1233","1558-0660","2161-9875"],"issn-type":[{"value":"1536-1233","type":"print"},{"value":"1558-0660","type":"electronic"},{"value":"2161-9875","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,8]]}}}