{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,17]],"date-time":"2026-08-17T16:10:14Z","timestamp":1786983014292,"version":"build-2736575974"},"reference-count":103,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2026,1,29]],"date-time":"2026-01-29T00:00:00Z","timestamp":1769644800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,29]],"date-time":"2026-01-29T00:00:00Z","timestamp":1769644800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"RGC Theme-based Research Scheme","award":["T43-513\/23-N"],"award-info":[{"award-number":["T43-513\/23-N"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["CCF Trans. Pervasive Comp. Interact."],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1007\/s42486-025-00227-7","type":"journal-article","created":{"date-parts":[[2026,1,29]],"date-time":"2026-01-29T08:54:56Z","timestamp":1769676896000},"page":"181-210","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Edge large language models: a comprehensive survey"],"prefix":"10.1007","volume":"8","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4727-4856","authenticated-orcid":false,"given":"Shan","family":"Jiang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xuecheng","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mingjin","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Changfu","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Guocheng","family":"Liao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jianguo","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiannong","family":"Cao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,1,29]]},"reference":[{"key":"227_CR1","unstructured":"Abadi, M., Barham, P., Chen, J., Chen, Z., Davis, A., Dean, J., Devin, M., Ghemawat, S., Irving, G., Isard, M., et\u00a0al.: Tensorflow: a system for large-scale machine learning. In: 12th USENIX Symposium on Operating Systems Design and Implementation (OSDI 16), pp. 265\u2013283 (2016)"},{"key":"227_CR2","doi-asserted-by":"crossref","unstructured":"Acharya, K., Velasquez, A., Song, H.H.: A survey on symbolic knowledge distillation of large language models. IEEE Trans. Artif. Intell. (2024)","DOI":"10.1109\/TAI.2024.3428519"},{"key":"227_CR3","doi-asserted-by":"crossref","unstructured":"Afzal, M.Z., Ali, S.A., Stricker, D., Eisert, P., Hilsmann, A., Perez-Marcos, D., Bianchi, M., Crottaz-Herbette, S., De\u00a0Ioris, R., Mangina, E., et al.: Next generation xr systems-large language models meet augmented and virtual reality. IEEE Comput. Gr. Appl. (2025)","DOI":"10.1109\/MCG.2025.3548554"},{"key":"227_CR4","doi-asserted-by":"crossref","unstructured":"Ainslie, J., Lee-Thorp, J., De\u00a0Jong, M., Zemlyanskiy, Y., Lebr\u00f3n, F., Sanghai, S.: Gqa: Training generalized multi-query transformer models from multi-head checkpoints. arXiv preprint arXiv:2305.13245 (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.298"},{"key":"227_CR5","unstructured":"Arivazhagan, M.G., Aggarwal, V., Singh, A.K., Choudhary, S.: Federated learning with personalization layers. arXiv preprint arXiv:1912.00818 (2019)"},{"key":"227_CR6","doi-asserted-by":"crossref","unstructured":"Behravan, M., Matkovi\u0107, K., Gra\u010danin, D.: Generative ai for context-aware 3d object creation using vision-language models in augmented reality. In: 2025 IEEE International Conference on Artificial Intelligence and eXtended and Virtual Reality (AIxVR), pp. 73\u201381 (2025). IEEE","DOI":"10.1109\/AIxVR63409.2025.00018"},{"key":"227_CR7","unstructured":"Beltagy, I., Peters, M.E., Cohan, A.: Longformer: The long-document transformer. arXiv preprint arXiv:2004.05150 (2020)"},{"key":"227_CR8","unstructured":"Bengio, Y., Ducharme, R., Vincent, P., Jauvin, C.: A neural probabilistic language model. J. Mach. Learn. Res. 3(Feb), 1137\u20131155 (2003)"},{"key":"227_CR9","first-page":"129","volume":"2","author":"D Blalock","year":"2020","unstructured":"Blalock, D., Gonzalez Ortiz, J.J., Frankle, J., Guttag, J.: What is the state of neural network pruning? Proceedings of machine learning and systems 2, 129\u2013146 (2020)","journal-title":"Proc. Mach. Learn. Syst."},{"key":"227_CR10","unstructured":"Brown, T.B., Mann, B., Ryder, N., Subbiah, M., Kaplan, J., Dhariwal, P., Neelakantan, A., Shyam, P., Sastry, G., Askell, A., Agarwal, S., Herbert-Voss, A., Krueger, G., Henighan, T., Child, R., Ramesh, A., Ziegler, D.M., Wu, J., Winter, C., Hesse, C., Chen, M., Sigler, E., Litwin, M., Gray, S., Chess, B., Clark, J., Berner, C., McCandlish, S., Radford, A., Sutskever, I., Amodei, D.: Language models are few-shot learners. In: Annual Conference on Neural Information Processing Systems (NeurIPS 20) (2020)"},{"issue":"2","key":"227_CR11","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3743127","volume":"58","author":"S Chaudhari","year":"2025","unstructured":"Chaudhari, S., Aggarwal, P., Murahari, V., Rajpurohit, T., Kalyan, A., Narasimhan, K., Deshpande, A., Silva, B.: Rlhf deciphered: A critical analysis of reinforcement learning from human feedback for llms. ACM Comput. Surv. 58(2), 1\u201337 (2025)","journal-title":"ACM Comput. Surv."},{"issue":"2","key":"227_CR12","doi-asserted-by":"publisher","first-page":"594","DOI":"10.1109\/TCC.2024.3381646","volume":"12","author":"X Chen","year":"2024","unstructured":"Chen, X., Cao, J., Sahni, Y., Jiang, S., Liang, Z.: Dynamic task offloading in edge computing based on dependency-aware reinforcement learning. IEEE Transactions on Cloud Computing 12(2), 594\u2013608 (2024)","journal-title":"IEEE Trans. Cloud Comput."},{"issue":"5","key":"227_CR13","doi-asserted-by":"publisher","first-page":"263","DOI":"10.1109\/MNET.2025.3539338","volume":"39","author":"H Chen","year":"2025","unstructured":"Chen, H., Deng, W., Yang, S., Xu, J., Jiang, Z., Ngai, E.C., Liu, J., Liu, X.: Towards edge general intelligence via large language models: Opportunities and challenges. IEEE Network 39(5), 263\u2013271 (2025)","journal-title":"IEEE Netw."},{"issue":"4","key":"227_CR14","doi-asserted-by":"publisher","first-page":"42","DOI":"10.1007\/s11280-024-01276-1","volume":"27","author":"J Chen","year":"2024","unstructured":"Chen, J., Liu, Z., Huang, X., Wu, C., Liu, Q., Jiang, G., Pu, Y., Lei, Y., Chen, X., Wang, X., et al.: When large language models meet personalization: Perspectives of challenges and opportunities. World Wide Web 27(4), 42 (2024)","journal-title":"World Wide Web"},{"key":"227_CR15","unstructured":"Chen, Y., Qian, S., Tang, H., Lai, X., Liu, Z., Han, S., Jia, J.: Longlora: Efficient fine-tuning of long-context large language models. arXiv preprint arXiv:2309.12307 (2023)"},{"key":"227_CR16","unstructured":"Chen, C., Borgeaud, S., Irving, G., Lespiau, J.-B., Sifre, L., Jumper, J.: Accelerating large language model decoding with speculative sampling. arXiv preprint arXiv:2302.01318 (2023)"},{"key":"227_CR17","unstructured":"Chen, T., Moreau, T., Jiang, Z., Zheng, L., Yan, E., Shen, H., Cowan, M., Wang, L., Hu, Y., Ceze, L., et\u00a0al.: Tvm: An automated end-to-end optimizing compiler for deep learning. In: 13th USENIX Symposium on Operating Systems Design and Implementation (OSDI 18), pp. 578\u2013594 (2018)"},{"key":"227_CR18","unstructured":"Chetlur, S., Woolley, C., Vandermersch, P., Cohen, J., Tran, J., Catanzaro, B., Shelhamer, E.: cudnn: Efficient primitives for deep learning. arXiv preprint arXiv:1410.0759 (2014)"},{"key":"227_CR19","unstructured":"Child, R., Gray, S., Radford, A., Sutskever, I.: Generating long sequences with sparse transformers. arXiv preprint arXiv:1904.10509 (2019)"},{"key":"227_CR20","unstructured":"Chiu, H.-k., Hachiuma, R., Wang, C.-Y., Smith, S.F., Wang, Y.-C.F., Chen, M.-H.: V2v-llm: Vehicle-to-vehicle cooperative autonomous driving with multi-modal large language models. arXiv preprint arXiv:2502.09980 (2025)"},{"key":"227_CR21","unstructured":"Choromanski, K., Likhosherstov, V., Dohan, D., Song, X., Gane, A., Sarlos, T., Hawkins, P., Davis, J., Mohiuddin, A., Kaiser, L., et al.: Rethinking attention with performers. arXiv preprint arXiv:2009.14794 (2020)"},{"key":"227_CR22","unstructured":"Cobbe, K., Kosaraju, V., Bavarian, M., Chen, M., Jun, H., Kaiser, L., Plappert, M., Tworek, J., Hilton, J., Nakano, R., et al.: Training verifiers to solve math word problems. arXiv preprint arXiv:2110.14168 (2021)"},{"key":"227_CR23","doi-asserted-by":"publisher","DOI":"10.1016\/j.compind.2024.104129","volume":"162","author":"S Colabianchi","year":"2024","unstructured":"Colabianchi, S., Costantino, F., Sabetta, N.: Assessment of a large language model based digital intelligent assistant in assembly manufacturing. Comput. Ind. 162, 104129 (2024)","journal-title":"Comput. Ind."},{"issue":"6","key":"227_CR24","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3712001","volume":"57","author":"BC Das","year":"2025","unstructured":"Das, B.C., Amini, M.H., Wu, Y.: Security and privacy challenges of large language models: A survey. ACM Comput. Surv. 57(6), 1\u201339 (2025)","journal-title":"ACM Comput. Surv."},{"key":"227_CR25","doi-asserted-by":"publisher","first-page":"10088","DOI":"10.52202\/075280-0441","volume":"36","author":"T Dettmers","year":"2023","unstructured":"Dettmers, T., Pagnoni, A., Holtzman, A., Zettlemoyer, L.: Qlora: Efficient finetuning of quantized llms. Adv. Neural. Inf. Process. Syst. 36, 10088\u201310115 (2023)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"issue":"2","key":"227_CR26","doi-asserted-by":"publisher","first-page":"1141","DOI":"10.1007\/s10845-023-02294-y","volume":"36","author":"H Fan","year":"2025","unstructured":"Fan, H., Liu, X., Fuh, J.Y.H., Lu, W.F., Li, B.: Embodied intelligence in manufacturing: leveraging large language models for autonomous industrial robotics. J. Intell. Manuf. 36(2), 1141\u20131157 (2025)","journal-title":"J. Intell. Manuf."},{"key":"227_CR27","doi-asserted-by":"publisher","first-page":"5799","DOI":"10.1109\/OJCOMS.2024.3456549","volume":"5","author":"O ,","year":"2024","unstructured":"Friha, O., Ferrag, M.A., Kantarci, B., Cakmak, B., Ozgun, A., Ghoualmi-Zine, N.: Llm-based edge intelligence: A comprehensive survey on architectures, applications, security and trustworthiness. IEEE Open Journal of the Communications Society 5, 5799\u20135856 (2024)","journal-title":"IEEE Open J. Commun. Soc."},{"key":"227_CR28","unstructured":"Fu, Y., Xue, L., Huang, Y., Brabete, A.-O., Ustiugov, D., Patel, Y., Mai, L.: Serverlessllm: Low-latency serverless inference for large language models. In: 18th USENIX Symposium on Operating Systems Design and Implementation (OSDI 24), pp. 135\u2013153 (2024)"},{"key":"227_CR29","doi-asserted-by":"crossref","unstructured":"Gao, Z., Zhang, Z., Guo, Y., Gong, Y.: Federated adaptive fine-tuning of large language models with heterogeneous quantization and lora. In: IEEE INFOCOM 2025-IEEE Conference on Computer Communications, pp. 1\u201310 (2025). IEEE","DOI":"10.1109\/INFOCOM55648.2025.11044641"},{"issue":"6","key":"227_CR30","doi-asserted-by":"publisher","first-page":"1789","DOI":"10.1007\/s11263-021-01453-z","volume":"129","author":"J Gou","year":"2021","unstructured":"Gou, J., Yu, B., Maybank, S.J., Tao, D.: Knowledge distillation: A survey. Int. J. Comput. Vision 129(6), 1789\u20131819 (2021)","journal-title":"Int. J. Comput. Vision"},{"issue":"10","key":"227_CR31","doi-asserted-by":"publisher","first-page":"2222","DOI":"10.1109\/TNNLS.2016.2582924","volume":"28","author":"K Greff","year":"2016","unstructured":"Greff, K., Srivastava, R.K., Koutn\u00edk, J., Steunebrink, B.R., Schmidhuber, J.: Lstm: A search space odyssey. IEEE transactions on neural networks and learning systems 28(10), 2222\u20132232 (2016)","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"227_CR32","doi-asserted-by":"crossref","unstructured":"Gu, Z., Fan, Q., Sun, L., Liu, Y., Ye, X.: Vflair-llm: A comprehensive framework and benchmark for split learning of llms. In: Proceedings of the 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining V. 2, pp. 5470\u20135481 (2025)","DOI":"10.1145\/3711896.3737411"},{"key":"227_CR33","unstructured":"Gu, A., Dao, T.: Mamba: Linear-time sequence modeling with selective state spaces. In: First Conference on Language Modeling, pp. 1\u201332 (2024)"},{"key":"227_CR34","unstructured":"Guu, K., Lee, K., Tung, Z., Pasupat, P., Chang, M.: Retrieval augmented language model pre-training. In: International Conference on Machine Learning, pp. 3929\u20133938 (2020). PMLR"},{"key":"227_CR35","unstructured":"Hendrycks, D., Burns, C., Basart, S., Zou, A., Mazeika, M., Song, D., Steinhardt, J.: Measuring massive multitask language understanding. arXiv preprint arXiv:2009.03300 (2020)"},{"key":"227_CR36","unstructured":"Hinton, G., Vinyals, O., Dean, J.: Distilling the knowledge in a neural network. arXiv preprint arXiv:1503.02531 (2015)"},{"key":"227_CR37","doi-asserted-by":"crossref","unstructured":"Hong, J., Lee, Y., Kim, D.H., Choi, D., Yoon, Y.-J., Lee, G.-c., Lee, Z., Kim, J.: A context-aware onboarding agent for metaverse powered by large language models. In: Proceedings of the 2024 ACM Designing Interactive Systems Conference, pp. 1857\u20131874 (2024)","DOI":"10.1145\/3643834.3661579"},{"key":"227_CR38","first-page":"1270","volume":"37","author":"C Hooper","year":"2024","unstructured":"Hooper, C., Kim, S., Mohammadzadeh, H., Mahoney, M.W., Shao, Y.S., Keutzer, K., Gholami, A.: Kvquant: Towards 10 million context length llm inference with kv cache quantization. Adv. Neural. Inf. Process. Syst. 37, 1270\u20131303 (2024)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"227_CR39","unstructured":"Houlsby, N., Giurgiu, A., Jastrzebski, S., Morrone, B., De\u00a0Laroussilhe, Q., Gesmundo, A., Attariyan, M., Gelly, S.: Parameter-efficient transfer learning for nlp. In: International Conference on Machine Learning, pp. 2790\u20132799 (2019). PMLR"},{"issue":"2","key":"227_CR40","first-page":"3","volume":"1","author":"EJ Hu","year":"2022","unstructured":"Hu, E.J., Shen, Y., Wallis, P., Allen-Zhu, Z., Li, Y., Wang, S., Wang, L., Chen, W., et al.: Lora: Low-rank adaptation of large language models. ICLR 1(2), 3 (2022)","journal-title":"ICLR"},{"key":"227_CR41","unstructured":"Hu, Q., Ye, Z., Wang, Z., Wang, G., Zhang, M., Chen, Q., Sun, P., Lin, D., Wang, X., Luo, Y., et\u00a0al.: Characterization of large language model development in the datacenter. In: 21st USENIX Symposium on Networked Systems Design and Implementation (NSDI 24), pp. 709\u2013729 (2024)"},{"issue":"11","key":"227_CR42","doi-asserted-by":"publisher","first-page":"182","DOI":"10.3390\/data10110182","volume":"10","author":"S Jiang","year":"2025","unstructured":"Jiang, S.: Big data sharing: A comprehensive survey. Data 10(11), 182 (2025)","journal-title":"Data"},{"key":"227_CR43","doi-asserted-by":"publisher","first-page":"72","DOI":"10.1016\/j.ins.2023.03.121","volume":"635","author":"S Jiang","year":"2023","unstructured":"Jiang, S., Cao, J., Wu, H., Chen, K., Liu, X.: Privacy-preserving and efficient data sharing for blockchain-based intelligent transportation systems. Inf. Sci. 635, 72\u201385 (2023)","journal-title":"Inf. Sci."},{"key":"227_CR44","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2025.103752","volume":"127","author":"S Jiang","year":"2026","unstructured":"Jiang, S., Chai, W., Zhang, M., Cao, J., Xuan, S., Shen, J.: Verifying energy generation via edge llm for web3-based decentralized clean energy networks. Information Fusion 127, 103752 (2026)","journal-title":"Inf. Fus."},{"key":"227_CR45","unstructured":"Jiang, A.Q., Sablayrolles, A., Roux, A., Mensch, A., Savary, B., Bamford, C., Chaplot, D.S., Casas, D.d.l., Hanna, E.B., Bressand, F., et al.: Mixtral of experts. arXiv preprint arXiv:2401.04088 (2024)"},{"key":"227_CR46","unstructured":"Karimireddy, S.P., Kale, S., Mohri, M., Reddi, S., Stich, S., Suresh, A.T.: Scaffold: Stochastic controlled averaging for federated learning. In: International Conference on Machine Learning, pp. 5132\u20135143 (2020). PMLR"},{"key":"227_CR47","unstructured":"Katharopoulos, A., Vyas, A., Pappas, N., Fleuret, F.: Transformers are rnns: Fast autoregressive transformers with linear attention. In: International Conference on Machine Learning, pp. 5156\u20135165 (2020). PMLR"},{"issue":"1","key":"227_CR48","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3643505","volume":"8","author":"E King","year":"2024","unstructured":"King, E., Yu, H., Lee, S., Julien, C.: Sasha: creative goal-oriented reasoning in smart homes with large language models. Proceedings of the ACM on Interactive, Mobile, Wearable and Ubiquitous Technologies 8(1), 1\u201338 (2024)","journal-title":"Proc. ACM Interact. Mobile Wear. Ubiquit. Technol."},{"key":"227_CR49","doi-asserted-by":"crossref","unstructured":"Kong, R., Li, Y., Wang, W., Kong, L., Liu, Y.: Serving moe models on resource-constrained edge devices via dynamic expert swapping. IEEE Transactions on Computers (2025)","DOI":"10.1109\/TC.2025.3575905"},{"key":"227_CR50","doi-asserted-by":"crossref","unstructured":"Kwon, W., Li, Z., Zhuang, S., Sheng, Y., Zheng, L., Yu, C.H., Gonzalez, J., Zhang, H., Stoica, I.: Efficient memory management for large language model serving with pagedattention. In: Proceedings of the 29th Symposium on Operating Systems Principles, pp. 611\u2013626 (2023)","DOI":"10.1145\/3600006.3613165"},{"key":"227_CR51","first-page":"429","volume":"2","author":"T Li","year":"2020","unstructured":"Li, T., Sahu, A.K., Zaheer, M., Sanjabi, M., Talwalkar, A., Smith, V.: Federated optimization in heterogeneous networks. Proceedings of Machine learning and systems 2, 429\u2013450 (2020)","journal-title":"Proc. Mach. Learn. Syst."},{"key":"227_CR52","unstructured":"Li, X.L., Liang, P.: Prefix-tuning: Optimizing continuous prompts for generation. arXiv preprint arXiv:2101.00190 (2021)"},{"key":"227_CR53","doi-asserted-by":"crossref","unstructured":"Li, N., Guo, S., Zhang, T., Li, M., Hong, Z., Zhou, Q., Yuan, X., Zhang, H.: The moe-empowered edge llms deployment: Architecture, challenges, and opportunities. arXiv preprint arXiv:2502.08381 (2025)","DOI":"10.1109\/MCOM.001.2400717"},{"key":"227_CR54","doi-asserted-by":"crossref","unstructured":"Liang, Z., Cao, J., Jiang, S., Xu, H.: Hierarchical reinforcement learning with partner modeling for distributed multi-agent cooperation. IEEE Trans. Parallel Distrib. Syst., 1\u201313 (2024)","DOI":"10.1109\/TPDS.2024.3457153"},{"issue":"9","key":"227_CR55","doi-asserted-by":"publisher","first-page":"52","DOI":"10.1109\/MCOM.001.2400764","volume":"63","author":"Z Lin","year":"2025","unstructured":"Lin, Z., Qu, G., Chen, Q., Chen, X., Chen, Z., Huang, K.: Pushing large language models to the 6g edge: Vision, challenges, and opportunities. IEEE Commun. Mag. 63(9), 52\u201359 (2025)","journal-title":"IEEE Commun. Mag."},{"key":"227_CR56","first-page":"87","volume":"6","author":"J Lin","year":"2024","unstructured":"Lin, J., Tang, J., Tang, H., Yang, S., Chen, W.-M., Wang, W.-C., Xiao, G., Dang, X., Gan, C., Han, S.: Awq: Activation-aware weight quantization for on-device llm compression and acceleration. Proceedings of machine learning and systems 6, 87\u2013100 (2024)","journal-title":"Proc. Mach. Learn. Syst."},{"key":"227_CR57","doi-asserted-by":"crossref","unstructured":"Lin, X., Wang, W., Li, Y., Yang, S., Feng, F., Wei, Y., Chua, T.-S.: Data-efficient fine-tuning for llm-based recommendation. In: Proceedings of the 47th International ACM SIGIR Conference on Research and Development in Information Retrieval, pp. 365\u2013374 (2024)","DOI":"10.1145\/3626772.3657807"},{"key":"227_CR58","unstructured":"Lin, Z., Hu, X., Zhang, Y., Chen, Z., Fang, Z., Chen, X., Li, A., Vepakomma, P., Gao, Y.: Splitlora: A split parameter-efficient fine-tuning framework for large language models. arXiv preprint arXiv:2407.00952 (2024)"},{"key":"227_CR59","unstructured":"Lin, C.-Y.: Rouge: A package for automatic evaluation of summaries. In: Text Summarization Branches Out, pp. 74\u201381 (2004)"},{"key":"227_CR60","doi-asserted-by":"crossref","unstructured":"Lin, S., Hilton, J., Evans, O.: Truthfulqa: Measuring how models mimic human falsehoods. In: Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics, pp. 3214\u20133252 (2022)","DOI":"10.18653\/v1\/2022.acl-long.229"},{"issue":"11","key":"227_CR61","doi-asserted-by":"publisher","first-page":"15869","DOI":"10.1109\/JIOT.2025.3529934","volume":"12","author":"P Liu","year":"2025","unstructured":"Liu, P., He, Q., Chen, Y., Jiang, S.: Efficient multikeyword searchable and verifiable data sharing for cloud-edge collaboration intelligent transportation systems. IEEE Internet Things J. 12(11), 15869\u201315882 (2025)","journal-title":"IEEE Internet Things J."},{"key":"227_CR62","unstructured":"Liu, Y., Ott, M., Goyal, N., Du, J., Joshi, M., Chen, D., Levy, O., Lewis, M., Zettlemoyer, L., Stoyanov, V.: Roberta: A robustly optimized bert pretraining approach. arXiv preprint arXiv:1907.11692 (2019)"},{"key":"227_CR63","doi-asserted-by":"crossref","unstructured":"Liu, X., Ji, K., Fu, Y., Tam, W.L., Du, Z., Yang, Z., Tang, J.: P-tuning v2: Prompt tuning can be comparable to fine-tuning universally across scales and tasks. arXiv preprint arXiv:2110.07602 (2021)","DOI":"10.18653\/v1\/2022.acl-short.8"},{"key":"227_CR64","doi-asserted-by":"crossref","unstructured":"Liu, Z., Oguz, B., Zhao, C., Chang, E., Stock, P., Mehdad, Y., Shi, Y., Krishnamoorthi, R., Chandra, V.: Llm-qat: Data-free quantization aware training for large language models. arXiv preprint arXiv:2305.17888 (2023)","DOI":"10.18653\/v1\/2024.findings-acl.26"},{"key":"227_CR65","unstructured":"Liu, Z., Zhao, C., Iandola, F., Lai, C., Tian, Y., Fedorov, I., Xiong, Y., Chang, E., Shi, Y., Krishnamoorthi, R., et\u00a0al.: Mobilellm: Optimizing sub-billion parameter language models for on-device use cases. In: Forty-first International Conference on Machine Learning (2024)"},{"key":"227_CR66","doi-asserted-by":"crossref","unstructured":"Luo, H., Liu, Y., Zhang, R., Wang, J., Sun, G., Niyato, D., Yu, H., Xiong, Z., Wang, X., Shen, X.: Toward edge general intelligence with multiple-large language model (multi-llm): architecture, trust, and orchestration. IEEE Trans. Cognit. Commun. Netw. (2025)","DOI":"10.1109\/TCCN.2025.3612760"},{"issue":"1","key":"227_CR67","doi-asserted-by":"publisher","first-page":"416","DOI":"10.1109\/COMST.2017.2771153","volume":"20","author":"C Mouradian","year":"2017","unstructured":"Mouradian, C., Naboulsi, D., Yangui, S., Glitho, R.H., Morrow, M.J., Polakos, P.A.: A comprehensive survey on fog computing: State-of-the-art and research challenges. IEEE communications surveys & tutorials 20(1), 416\u2013464 (2017)","journal-title":"IEEE Commun. Surv. Tutor."},{"issue":"5","key":"227_CR68","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3744746","volume":"16","author":"H Naveed","year":"2025","unstructured":"Naveed, H., Khan, A.U., Qiu, S., Saqib, M., Anwar, S., Usman, M., Akhtar, N., Barnes, N., Mian, A.: A comprehensive overview of large language models. ACM Transactions on Intelligent Systems and Technology 16(5), 1\u201372 (2025)","journal-title":"ACM Trans. Intell. Syst. Technol."},{"issue":"10","key":"227_CR69","doi-asserted-by":"publisher","first-page":"2061","DOI":"10.3390\/electronics14102061","volume":"14","author":"G Palma","year":"2025","unstructured":"Palma, G., Cecchi, G., Rizzo, A.: Large language models for predictive maintenance in the leather tanning industry: Multimodal anomaly detection in compressors. Electronics 14(10), 2061 (2025)","journal-title":"Electronics"},{"key":"227_CR70","doi-asserted-by":"crossref","unstructured":"Papineni, K., Roukos, S., Ward, T., Zhu, W.-J.: Bleu: a method for automatic evaluation of machine translation. In: Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics, pp. 311\u2013318 (2002)","DOI":"10.3115\/1073083.1073135"},{"key":"227_CR71","doi-asserted-by":"crossref","unstructured":"Qin, R., Xia, J., Jia, Z., Jiang, M., Abbasi, A., Zhou, P., Hu, J., Shi, Y.: Enabling on-device large language model personalization with self-supervised data selection and synthesis. In: Proceedings of the 61st ACM\/IEEE Design Automation Conference, pp. 1\u20136 (2024)","DOI":"10.1145\/3649329.3655665"},{"key":"227_CR72","doi-asserted-by":"crossref","unstructured":"Qiu, W., Zhou, Y., Wang, J., Sheng, Q.Z., Cui, L.: Flm-topk: Expediting federated large language model tuning by sparsifying intervalized gradients. In: IEEE INFOCOM 2025-IEEE Conference on Computer Communications, pp. 1\u201310 (2025). IEEE","DOI":"10.1109\/INFOCOM55648.2025.11044514"},{"key":"227_CR73","doi-asserted-by":"crossref","unstructured":"Qu, G., Chen, Q., Wei, W., Lin, Z., Chen, X., Huang, K.: Mobile edge intelligence for large language models: A contemporary survey. IEEE Commun. Surv. Tutor. (2025)","DOI":"10.36227\/techrxiv.172115025.57884352\/v1"},{"key":"227_CR74","first-page":"53728","volume":"36","author":"R Rafailov","year":"2023","unstructured":"Rafailov, R., Sharma, A., Mitchell, E., Manning, C.D., Ermon, S., Finn, C.: Direct preference optimization: Your language model is secretly a reward model. Adv. Neural. Inf. Process. Syst. 36, 53728\u201353741 (2023)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"227_CR75","doi-asserted-by":"crossref","unstructured":"Shao, H., Hu, Y., Wang, L., Song, G., Waslander, S.L., Liu, Y., Li, H.: Lmdrive: Closed-loop end-to-end driving with large language models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15120\u201315130 (2024)","DOI":"10.1109\/CVPR52733.2024.01432"},{"key":"227_CR76","unstructured":"Shazeer, N.: Fast transformer decoding: One write-head is all you need. arXiv preprint arXiv:1911.02150 (2019)"},{"issue":"10","key":"227_CR77","doi-asserted-by":"publisher","first-page":"140","DOI":"10.1109\/MCOM.001.2300550","volume":"62","author":"Y Shen","year":"2024","unstructured":"Shen, Y., Shao, J., Zhang, X., Lin, Z., Pan, H., Li, D., Zhang, J., Letaief, K.B.: Large language models empowered autonomous edge ai for connected intelligence. IEEE Commun. Mag. 62(10), 140\u2013146 (2024)","journal-title":"IEEE Commun. Mag."},{"key":"227_CR78","doi-asserted-by":"crossref","unstructured":"Talmor, A., Herzig, J., Lourie, N., Berant, J.: Commonsenseqa: A question answering challenge targeting commonsense knowledge. In: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, pp. 4149\u20134158 (2019)","DOI":"10.18653\/v1\/N19-1421"},{"issue":"12","key":"227_CR79","doi-asserted-by":"publisher","first-page":"10706","DOI":"10.1109\/TMC.2024.3379501","volume":"23","author":"T Tan","year":"2024","unstructured":"Tan, T., Cao, G.: Thermal-aware scheduling for deep learning on mobile devices with npu. IEEE Trans. Mob. Comput. 23(12), 10706\u201310719 (2024)","journal-title":"IEEE Trans. Mob. Comput."},{"issue":"8","key":"227_CR80","doi-asserted-by":"publisher","first-page":"1930","DOI":"10.1038\/s41591-023-02448-8","volume":"29","author":"AJ Thirunavukarasu","year":"2023","unstructured":"Thirunavukarasu, A.J., Ting, D.S.J., Elangovan, K., Gutierrez, L., Tan, T.F., Ting, D.S.W.: Large language models in medicine. Nat. Med. 29(8), 1930\u20131940 (2023)","journal-title":"Nat. Med."},{"key":"227_CR81","unstructured":"Touvron, H., Lavril, T., Izacard, G., Martinet, X., Lachaux, M.-A., Lacroix, T., Rozi\u00e8re, B., Goyal, N., Hambro, E., Azhar, F., et al.: Llama: Open and efficient foundation language models. arXiv preprint arXiv:2302.13971 (2023)"},{"issue":"5","key":"227_CR82","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3708495","volume":"57","author":"J Tu","year":"2025","unstructured":"Tu, J., Yang, L., Cao, J.: Distributed machine learning in edge computing: Challenges, solutions and future directions. ACM Comput. Surv. 57(5), 1\u201337 (2025)","journal-title":"ACM Comput. Surv."},{"key":"227_CR83","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, \u0141., Polosukhin, I.: Attention is all you need. Adv. Neural Inform. Process. Syst. 30 (2017)"},{"issue":"10","key":"227_CR84","doi-asserted-by":"publisher","first-page":"2878","DOI":"10.1038\/s41591-024-03148-7","volume":"30","author":"P Wan","year":"2024","unstructured":"Wan, P., Huang, Z., Tang, W., Nie, Y., Pei, D., Deng, S., Chen, J., Zhou, Y., Duan, H., Chen, Q., et al.: Outpatient reception via collaboration between nurses and a large language model: a randomized controlled trial. Nat. Med. 30(10), 2878\u20132885 (2024)","journal-title":"Nat. Med."},{"key":"227_CR85","doi-asserted-by":"crossref","unstructured":"Wang, Y., Cao, J., Jiang, S., Wang, S., Liu, H., Liang, Z.: Efficient and atomic cross-blockchain transaction processing for decentralized web3 applications. In: 2024 IEEE Smart World Congress (SWC), pp. 155\u2013162 (2024). IEEE","DOI":"10.1109\/SWC62898.2024.00056"},{"key":"227_CR86","unstructured":"Wang, S., Li, B.Z., Khabsa, M., Fang, H., Ma, H.: Linformer: Self-attention with linear complexity. arXiv preprint arXiv:2006.04768 (2020)"},{"key":"227_CR87","doi-asserted-by":"crossref","unstructured":"Wang, C., Pan, D., Kumari, S., Wu, M.-E., Jiang, H.: Large language models driven health text information analysis in consumer electronics. IEEE Trans. Consum. Electron. (2024)","DOI":"10.1109\/TCE.2024.3516006"},{"key":"227_CR88","doi-asserted-by":"crossref","unstructured":"Wang, S., Deng, J., Li, Q., Wu, J., Zhao, Z.: Performance analysis on the applications of large language models: A case for elderly care. In: 2024 IEEE International Conference on High Performance Computing and Communications (HPCC), pp. 145\u2013151 (2024). IEEE","DOI":"10.1109\/HPCC64274.2024.00029"},{"key":"227_CR89","unstructured":"Wu, X., Liu, X., Niu, J., Wang, H., Tang, S., Zhu, G.: Fedlora: When personalized federated learning meets low-rank adaptation (2024)"},{"key":"227_CR90","doi-asserted-by":"crossref","unstructured":"Wu, T., Li, M., Qu, Y., Wang, H., Wei, Z., Cao, J.: Joint uav deployment and edge association for energy-efficient federated learning. IEEE Trans. Cognit. Commun. Netw. (2025)","DOI":"10.1109\/TCCN.2025.3543365"},{"key":"227_CR91","doi-asserted-by":"crossref","unstructured":"Xu, Z., Chen, T., Huang, Z., Xing, Y., Chen, S.: Personalizing driver agent using large language models for driving safety and smarter human\u2013machine interactions. IEEE Intell. Transp. Syst. Mag. (2025)","DOI":"10.1109\/MITS.2025.3551736"},{"key":"227_CR92","doi-asserted-by":"crossref","unstructured":"Yan, H., Li, Y.: Generative ai for intelligent transportation systems: Road transportation perspective. ACM Comput. Surv. (2025)","DOI":"10.1145\/3719290"},{"key":"227_CR93","unstructured":"Yi, R., Guo, L., Wei, S., Zhou, A., Wang, S., Xu, M.: Edgemoe: Fast on-device inference of moe-based large language models. arXiv preprint arXiv:2308.14352 (2023)"},{"issue":"7","key":"227_CR94","doi-asserted-by":"publisher","first-page":"1235","DOI":"10.1162\/neco_a_01199","volume":"31","author":"Y Yu","year":"2019","unstructured":"Yu, Y., Si, X., Hu, C., Zhang, J.: A review of recurrent neural networks: Lstm cells and network architectures. Neural Comput. 31(7), 1235\u20131270 (2019)","journal-title":"Neural Comput."},{"key":"227_CR95","doi-asserted-by":"crossref","unstructured":"Yu, Z., Wang, Z., Li, Y., Gao, R., Zhou, X., Bommu, S.R., Zhao, Y., Lin, Y.: Edge-llm: Enabling efficient large language model adaptation on edge devices via unified compression and adaptive layer voting. In: Proceedings of the 61st ACM\/IEEE Design Automation Conference, pp. 1\u20136 (2024)","DOI":"10.1145\/3649329.3658473"},{"key":"227_CR96","doi-asserted-by":"crossref","unstructured":"Zellers, R., Holtzman, A., Bisk, Y., Farhadi, A., Choi, Y.: Hellaswag: Can a machine really finish your sentence? In: Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics, pp. 4791\u20134800 (2019)","DOI":"10.18653\/v1\/P19-1472"},{"issue":"10","key":"227_CR97","doi-asserted-by":"publisher","first-page":"13119","DOI":"10.1109\/JIOT.2024.3524255","volume":"12","author":"M Zhang","year":"2025","unstructured":"Zhang, M., Shen, X., Cao, J., Cui, Z., Jiang, S.: Edgeshard: Efficient llm inference via collaborative edge computing. IEEE Internet Things J. 12(10), 13119\u201313131 (2025)","journal-title":"IEEE Internet Things J."},{"issue":"4","key":"227_CR98","doi-asserted-by":"publisher","first-page":"1449","DOI":"10.1109\/TNET.2022.3217280","volume":"31","author":"Y Zhang","year":"2022","unstructured":"Zhang, Y., Wang, W., Ren, J., Huang, J., He, S., Zhang, Y.: Efficient revenue-based mec server deployment and management in mobile edge-cloud computing. IEEE\/ACM Trans. Networking 31(4), 1449\u20131462 (2022)","journal-title":"IEEE\/ACM Trans. Netw."},{"key":"227_CR99","doi-asserted-by":"crossref","unstructured":"Zhang, M., Cao, J., Sahni, Y., Chen, X., Jiang, S.: Resource-efficient parallel split learning in heterogeneous edge computing. In: 2024 International Conference on Computing, Networking and Communications, pp. 794\u2013798 (2024)","DOI":"10.1109\/ICNC59896.2024.10556386"},{"key":"227_CR100","doi-asserted-by":"publisher","first-page":"46","DOI":"10.1016\/j.comcom.2023.11.024","volume":"214","author":"L Zhao","year":"2024","unstructured":"Zhao, L., Yang, Q., Huang, H., Guo, L., Jiang, S.: Intelligent wireless sensing driven metaverse: A survey. Comput. Commun. 214, 46\u201356 (2024)","journal-title":"Comput. Commun."},{"issue":"8","key":"227_CR101","first-page":"1","volume":"57","author":"Y Zheng","year":"2025","unstructured":"Zheng, Y., Chen, Y., Qian, B., Shi, X., Shu, Y., Chen, J.: A review on edge large language models: Design, execution, and applications. ACM Comput. Surv. 57(8), 1\u201335 (2025)","journal-title":"ACM Comput. Surv."},{"key":"227_CR102","unstructured":"Zheng, L., Jia, C., Sun, M., Wu, Z., Yu, C.H., Haj-Ali, A., Wang, Y., Yang, J., Zhuo, D., Sen, K., et\u00a0al.: Ansor: Generating high-performance tensor programs for deep learning. In: 14th USENIX Symposium on Operating Systems Design and Implementation (OSDI 20), pp. 863\u2013879 (2020)"},{"key":"227_CR103","doi-asserted-by":"crossref","unstructured":"Zhuang, W., Pan, X., Wen, S., Yu, W., Li, X., Bao, J.: Large language model-enabled multi-agent self-organised approach for personalised rehabilitation assistive devices design. Int. J. Prod. Res. 1\u201330 (2025)","DOI":"10.1080\/00207543.2025.2509155"}],"container-title":["CCF Transactions on Pervasive Computing and Interaction"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s42486-025-00227-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s42486-025-00227-7","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s42486-025-00227-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,10]],"date-time":"2026-06-10T07:53:41Z","timestamp":1781078021000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s42486-025-00227-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,1,29]]},"references-count":103,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2026,6]]}},"alternative-id":["227"],"URL":"https:\/\/doi.org\/10.1007\/s42486-025-00227-7","relation":{},"ISSN":["2524-521X","2524-5228"],"issn-type":[{"value":"2524-521X","type":"print"},{"value":"2524-5228","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,1,29]]},"assertion":[{"value":"1 October 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 December 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"29 January 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"On behalf of all authors, the corresponding author states that there is no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}