{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,25]],"date-time":"2025-09-25T20:40:02Z","timestamp":1758832802113,"version":"3.44.0"},"reference-count":69,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2025,7,7]],"date-time":"2025-07-07T00:00:00Z","timestamp":1751846400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,7,7]],"date-time":"2025-07-07T00:00:00Z","timestamp":1751846400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"Institute of Information & Communications Technology Planning & Evaluation","award":["RS-2024-00392332","RS-2024-00392332","RS-2024-00392332"],"award-info":[{"award-number":["RS-2024-00392332","RS-2024-00392332","RS-2024-00392332"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Netw Syst Manage"],"published-print":{"date-parts":[[2025,10]]},"DOI":"10.1007\/s10922-025-09962-9","type":"journal-article","created":{"date-parts":[[2025,7,7]],"date-time":"2025-07-07T05:02:44Z","timestamp":1751864564000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["DRAFAS: Dynamic Resource Allocation for AI-Native Services"],"prefix":"10.1007","volume":"33","author":[{"given":"Nguyen Van","family":"Tu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sukhyun","family":"Nam","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lizhuang","family":"Tan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"James Won-ki","family":"Hong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,7,7]]},"reference":[{"key":"9962_CR1","unstructured":"Gartner: Forecast analysis: artificial intelligence software, 2023\u20132027, worldwide. https:\/\/www.gartner.com\/en\/documents\/4925331. Accessed 2024-12-10"},{"key":"9962_CR2","unstructured":"Precedence Research: Artificial intelligence-as-a-service market. https:\/\/www.precedenceresearch.com\/artificial-intelligence-as-a-service-market. Accessed 2024-12-10"},{"key":"9962_CR3","doi-asserted-by":"publisher","first-page":"96","DOI":"10.1109\/MWC.001.2100338","volume":"29","author":"W Wu","year":"2022","unstructured":"Wu, W., et al.: Ai-native network slicing for 6G networks. IEEE Wirel. Commun. 29, 96\u2013103 (2022)","journal-title":"IEEE Wirel. Commun."},{"key":"9962_CR4","unstructured":"Microsoft azure: Cloud computing services. https:\/\/azure.microsoft.com\/. Accessed 2024-12-10"},{"key":"9962_CR5","unstructured":"Amazon web services (aws): Cloud computing services. https:\/\/aws.amazon.com\/. Accessed 2024-12-10"},{"key":"9962_CR6","unstructured":"Google cloud platform: Cloud computing services. https:\/\/cloud.google.com\/. Accessed: 2024-12-10"},{"key":"9962_CR7","unstructured":"Open5gs: Open source 5G core network. https:\/\/open5gs.org\/. Accessed 2024-12-10"},{"key":"9962_CR8","unstructured":"Ueransim: Open source 5G ue and ran (gnodeb) implementation. https:\/\/github.com\/aligungr\/UERANSIM. Accessed 2024-12-10"},{"key":"9962_CR9","doi-asserted-by":"crossref","unstructured":"Tu, N.V.: Drafas: dynamic resource allocation for AI-native services. https:\/\/github.com\/tu-nv\/drafas. Accessed 2024-12-10","DOI":"10.1007\/s10922-025-09962-9"},{"key":"9962_CR10","first-page":"98","volume":"2","author":"P Yu","year":"2020","unstructured":"Yu, P., Chowdhury, M.: Fine-grained gpu sharing primitives for deep learning applications. Proc. Mach. Learn. Syst. 2, 98\u2013111 (2020)","journal-title":"Proc. Mach. Learn. Syst."},{"issue":"6","key":"9962_CR11","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3638757","volume":"56","author":"Z Ye","year":"2024","unstructured":"Ye, Z., et al.: Deep learning workload scheduling in gpu datacenters: a survey. ACM Comput. Surv. 56(6), 1\u201338 (2024)","journal-title":"ACM Comput. Surv."},{"key":"9962_CR12","unstructured":"Nvidia: Improving GPU utilization in Kubernetes. https:\/\/developer.nvidia.com\/blog\/improving-gpu-utilization-in-kubernetes. Accessed 2024-12-10"},{"key":"9962_CR13","unstructured":"Nvidia: Multi-Process Service. https:\/\/docs.nvidia.com\/deploy\/mps\/index.html. Accessed 2024-12-10"},{"key":"9962_CR14","unstructured":"Nvidia: Multi-Instance GPU. https:\/\/docs.nvidia.com\/datacenter\/tesla\/mig-user-guide\/index.html. Accessed 2024-12-10"},{"key":"9962_CR15","doi-asserted-by":"publisher","first-page":"94","DOI":"10.1109\/MCOM.2017.1600951","volume":"55","author":"X Foukas","year":"2017","unstructured":"Foukas, X., Patounas, G., Elmokashfi, A., Marina, M.K.: Network slicing in 5G: survey and challenges. IEEE Commun. Mag. 55, 94\u2013100 (2017)","journal-title":"IEEE Commun. Mag."},{"key":"9962_CR16","unstructured":"Farrel, A., et\u00a0al.: A framework for network slices in networks built from IETF technologies. RFC 9543 (2024). https:\/\/www.rfc-editor.org\/info\/rfc9543. Accessed 2024-12-10"},{"issue":"11","key":"9962_CR17","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3571072","volume":"55","author":"L-H Shen","year":"2023","unstructured":"Shen, L.-H., Feng, K.-T., Hanzo, L.: Five facets of 6G: research challenges and opportunities. ACM Comput. Surv. 55(11), 1\u201339 (2023)","journal-title":"ACM Comput. Surv."},{"key":"9962_CR18","doi-asserted-by":"crossref","unstructured":"Wu, X., Xu, H., Wang, Y.: Irina: accelerating dnn inference with efficient online scheduling. In: 4th Asia-Pacific Workshop on Networking, APNet \u201920, pp. 36\u201343. Association for Computing Machinery, New York (2020)","DOI":"10.1145\/3411029.3411035"},{"key":"9962_CR19","doi-asserted-by":"crossref","unstructured":"Dhakal, A., Kulkarni, S.G., Ramakrishnan, K.K.: Gslice: controlled spatial sharing of gpus for a scalable inference platform. In: Proceedings of the 11th ACM Symposium on Cloud Computing, SoCC \u201920, pp. 492\u2013506. Association for Computing Machinery, New York (2020)","DOI":"10.1145\/3419111.3421284"},{"key":"9962_CR20","unstructured":"Zhang, H., Tang, Y., Khandelwal, A., Stoica, I.: SHEPHERD: serving DNNs in the wild. In: 20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23), pp. 787\u2013808. USENIX Association, Boston (2023). https:\/\/www.usenix.org\/conference\/nsdi23\/presentation\/zhang-hong"},{"key":"9962_CR21","first-page":"20","volume":"4","author":"J Cho","year":"2022","unstructured":"Cho, J., Zad Tootaghaj, D., Cao, L., Sharma, P.: Sla-driven ml inference framework for clouds with heterogeneous accelerators. Proc. Mach. Learn. Syst. 4, 20\u201332 (2022)","journal-title":"Proc. Mach. Learn. Syst."},{"key":"9962_CR22","unstructured":"NVIDIA: Triton inference server. https:\/\/github.com\/triton-inference-server\/server. Accessed 2024-12-10"},{"key":"9962_CR23","doi-asserted-by":"crossref","unstructured":"Luan, Y., Chen, X., Zhao, H., Yang, Z., Dai, Y.: Sched2: scheduling deep learning training via deep reinforcement learning. In: 2019 IEEE Global Communications Conference (GLOBECOM), pp. 1\u20137 (2019)","DOI":"10.1109\/GLOBECOM38437.2019.9014110"},{"key":"9962_CR24","doi-asserted-by":"crossref","unstructured":"Wang, H., Liu, Z., Shen, H.: Job scheduling for large-scale machine learning clusters. In: Proceedings of the 16th International Conference on Emerging Networking EXperiments and Technologies, CoNEXT \u201920, pp. 108\u2013120. Association for Computing Machinery, New York (2020)","DOI":"10.1145\/3386367.3432588"},{"key":"9962_CR25","unstructured":"Mnih, V., et\u00a0al.: Playing atari with deep reinforcement learning. CoRR (2013)"},{"key":"9962_CR26","doi-asserted-by":"crossref","unstructured":"van Hasselt, H., Guez, A., Silver, D.: Deep reinforcement learning with double q-learning. CoRR (2015)","DOI":"10.1609\/aaai.v30i1.10295"},{"key":"9962_CR27","unstructured":"Mnih, V., et\u00a0al.: Asynchronous methods for deep reinforcement learning. In: International Conference on Machine Learning, pp. 1928\u20131937. PMLR (2016)"},{"key":"9962_CR28","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., Radford, A., Klimov, O.: Proximal policy optimization algorithms (2017). CoRR arXiv:1707.06347"},{"key":"9962_CR29","doi-asserted-by":"crossref","unstructured":"Williams, R.J.: Simple statistical gradient-following algorithms for connectionist reinforcement learning. In: Reinforcement Learning, pp. 5\u201332 (1992)","DOI":"10.1007\/978-1-4615-3618-5_2"},{"key":"9962_CR30","unstructured":"Schulman, J., Levine, S., Abbeel, P., Jordan, M., Moritz, P.: Trust region policy optimization. In: International Conference on Machine Learning, pp. 1889\u20131897. PMLR (2015)"},{"key":"9962_CR31","unstructured":"Haarnoja, T., Zhou, A., Abbeel, P., Levine, S.: Soft actor-critic: off-policy maximum entropy deep reinforcement learning with a stochastic actor. In: International Conference on Machine Learning, pp. 1861\u20131870. PMLR (2018)"},{"key":"9962_CR32","unstructured":"Fujimoto, S., Hoof, H., Meger, D.: Addressing function approximation error in actor-critic methods. In: International Conference on Machine Learning, pp. 1587\u20131596. PMLR (2018)"},{"key":"9962_CR33","unstructured":"Huang, S., Onta\u00f1\u00f3n, S.: A closer look at invalid action masking in policy gradient algorithms. CoRR (2020).arXiv:2006.14171"},{"key":"9962_CR34","first-page":"321","volume-title":"Multi-Agent Reinforcement Learning: A Selective Overview of Theories and Algorithms","author":"K Zhang","year":"2021","unstructured":"Zhang, K., Yang, Z., Ba\u015far, T.: Multi-Agent Reinforcement Learning: A Selective Overview of Theories and Algorithms, pp. 321\u2013384. Springer International Publishing, Cham (2021)"},{"issue":"315","key":"9962_CR35","first-page":"1","volume":"24","author":"S Hu","year":"2023","unstructured":"Hu, S., et al.: Marllib: s scalable and efficient multi-agent reinforcement learning library. J. Mach. Learn. Res. 24(315), 1\u201323 (2023)","journal-title":"J. Mach. Learn. Res."},{"key":"9962_CR36","doi-asserted-by":"crossref","unstructured":"Tan, M.: Multi-agent reinforcement learning: Independent vs. cooperative agents. In: Proceedings of the Tenth International Conference on Machine Learning, pp. 330\u2013337 (1993)","DOI":"10.1016\/B978-1-55860-307-3.50049-6"},{"key":"9962_CR37","doi-asserted-by":"crossref","unstructured":"Arabnejad, H., Pahl, C., Jamshidi, P., Estrada, G.: A comparison of reinforcement learning techniques for fuzzy cloud auto-scaling. In: 2017 17th IEEE\/ACM International Symposium on Cluster, Cloud and Grid Computing (CCGRID), pp. 64\u201373 (2017)","DOI":"10.1109\/CCGRID.2017.15"},{"key":"9962_CR38","unstructured":"Openstack: Open source cloud computing infrastructure. https:\/\/www.openstack.org\/. Accessed 2024-12-10"},{"issue":"6","key":"9962_CR39","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3475991","volume":"20","author":"K Ray","year":"2021","unstructured":"Ray, K., Banerjee, A.: Horizontal auto-scaling for multi-access edge computing using safe reinforcement learning. ACM Trans. Embed. Comput. Syst. 20(6), 1\u201333 (2021)","journal-title":"ACM Trans. Embed. Comput. Syst."},{"key":"9962_CR40","doi-asserted-by":"crossref","unstructured":"Lee, D., Yoo, J.-H., Hong, J. W.-K.: Deep q-networks based auto-scaling for service function chaining. In: 2020 16th International Conference on Network and Service Management (CNSM), pp. 1\u20139 (2020)","DOI":"10.23919\/CNSM50824.2020.9269107"},{"key":"9962_CR41","doi-asserted-by":"crossref","unstructured":"Soto, P., et\u00a0al.: Towards autonomous vnf auto-scaling using deep reinforcement learning. In: 2021 Eighth International Conference on Software Defined Systems (SDS), pp. 01\u201308 (2021)","DOI":"10.1109\/SDS54264.2021.9731854"},{"key":"9962_CR42","doi-asserted-by":"crossref","unstructured":"Gabriela, P., Lee, D., Tu, N.\u00a0V., Hong, J. W.-K.: Machine learning-based auto-scaler for video conferencing systems. In: 2021 IEEE 7th International Conference on Network Softwarization (NetSoft), pp. 142\u2013150 (2021)","DOI":"10.1109\/NetSoft51509.2021.9492728"},{"key":"9962_CR43","doi-asserted-by":"crossref","unstructured":"Gan, Z., Lin, R., Zou, H.: Adaptive auto-scaling in mobile edge computing: a deep reinforcement learning approach. In: 2022 2nd International Conference on Consumer Electronics and Computer Engineering (ICCECE), pp. 586\u2013591 (2022)","DOI":"10.1109\/ICCECE54139.2022.9712801"},{"key":"9962_CR44","doi-asserted-by":"publisher","first-page":"122229","DOI":"10.1109\/ACCESS.2020.3006502","volume":"8","author":"T Li","year":"2020","unstructured":"Li, T., Zhu, X., Liu, X.: An end-to-end network slicing algorithm based on deep q-learning for 5g network. IEEE Access 8, 122229\u2013122240 (2020)","journal-title":"IEEE Access"},{"key":"9962_CR45","doi-asserted-by":"crossref","unstructured":"Liu, Q., Choi, N., Han, T.: Atlas: automate online service configuration in network slicing. In: Proceedings of the 18th International Conference on Emerging Networking EXperiments and Technologies, CoNEXT \u201922, pp. 140\u2013155. Association for Computing Machinery, New York (2022)","DOI":"10.1145\/3555050.3569115"},{"key":"9962_CR46","doi-asserted-by":"crossref","unstructured":"D\u2019Oro, S., et\u00a0al.: Sl-edge: network slicing at the edge. In: Proceedings of the Twenty-First International Symposium on Theory, Algorithmic Foundations, and Protocol Design for Mobile Networks and Mobile Computing, Mobihoc \u201920, pp. 1\u201310. Association for Computing Machinery, New York (2020)","DOI":"10.1145\/3397166.3409133"},{"key":"9962_CR47","doi-asserted-by":"publisher","unstructured":"Liu, Q., Choi, N., Han, T.: Onslicing: online end-to-end network slicing with reinforcement learning. In: Proceedings of the 17th International Conference on Emerging Networking EXperiments and Technologies, CoNEXT \u201921, pp. 141\u2013153. Association for Computing Machinery, New York (2021). https:\/\/doi.org\/10.1145\/3485983.3494850","DOI":"10.1145\/3485983.3494850"},{"key":"9962_CR48","doi-asserted-by":"crossref","unstructured":"Stojkovic, J., Zhang, C., \u00cd\u00f1igo Goiri, Torrellas, J., Choukse, E.: Dynamollm: designing llm inference clusters for performance and energy efficiency (2024). arXiv:2408.00741","DOI":"10.1109\/HPCA61900.2025.00102"},{"key":"9962_CR49","unstructured":"Wang, Z., Gwon, C., Oates, T., Iezzi, A.: Automated cloud provisioning on aws using deep reinforcement learning (2017). arXiv:1709.04305"},{"key":"9962_CR50","first-page":"1","volume":"22","author":"A Raffin","year":"2021","unstructured":"Raffin, A., et al.: Stable-baselines3: reliable reinforcement learning implementations. J. Mach. Learn. Res. 22, 1\u20138 (2021)","journal-title":"J. Mach. Learn. Res."},{"key":"9962_CR51","unstructured":"Nair, V., Hinton, G.E.: Rectified linear units improve restricted Boltzmann machines. In: Proceedings of the 27th International Conference on International Conference on Machine Learning, ICML\u201910, pp. 807\u2013814. Omnipress, Madison (2010)"},{"key":"9962_CR52","unstructured":"Simpy: Discrete event simulation for python. https:\/\/gitlab.com\/team-simpy\/simpy. Accessed 2024-12-10"},{"key":"9962_CR53","unstructured":"Traefik: The cloud-native application proxy. https:\/\/traefik.io\/. Accessed 2024-12-10"},{"key":"9962_CR54","unstructured":"K3s: Lightweight kubernetes distribution. https:\/\/k3s.io\/. Accessed 2024-12-10"},{"key":"9962_CR55","unstructured":"Nvidia container toolkit: Build and run containers leveraging Nvidia gpus. https:\/\/github.com\/NVIDIA\/nvidia-container-toolkit. Accessed 2024-12-10"},{"key":"9962_CR56","unstructured":"Envoy proxy: Cloud-native high-performance edge\/middle\/service proxy. https:\/\/www.envoyproxy.io\/. Accessed 2024-12-10"},{"key":"9962_CR57","unstructured":"The istio service mesh. https:\/\/istio.io\/. Accessed 2024-12-10"},{"key":"9962_CR58","unstructured":"kube-proxy: The kubernetes network proxy. https:\/\/kubernetes.io\/docs\/reference\/command-line-tools-reference\/kube-proxy\/. Accessed 2024-12-10"},{"key":"9962_CR59","unstructured":"Prometheus: Monitoring system & time series database. https:\/\/prometheus.io\/. Accessed 2024-12-10"},{"key":"9962_CR60","unstructured":"asyncio: Asynchronous i\/o framework. https:\/\/docs.python.org\/3\/library\/asyncio.html. Accessed 2024-12-10"},{"key":"9962_CR61","unstructured":"Ollama: Get up and running with large language models. https:\/\/ollama.com\/. Accessed 2024-12-10"},{"key":"9962_CR62","unstructured":"Llama: Open-source AI models. https:\/\/ai.meta.com\/. Accessed 2024-12-10"},{"key":"9962_CR63","unstructured":". Pytorch: Tensors and dynamic neural networks in Python with strong GPU acceleration. https:\/\/github.com\/pytorch\/pytorch. Accessed 2024-12-10"},{"key":"9962_CR64","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"9962_CR65","unstructured":"Coqui tts: Open-source text-to-speech framework. https:\/\/coqui.ai\/. Accessed 2024-12-10"},{"key":"9962_CR66","unstructured":"Krizhevsky, A.: Learning multiple layers of features from tiny images. Technical Report, University of Toronto (2009)"},{"key":"9962_CR67","unstructured":"Chatgpt: Language model for conversational AI. https:\/\/chatgpt.com. Accessed 2024-12-10"},{"key":"9962_CR68","doi-asserted-by":"crossref","unstructured":"Akiba, T., Sano, S., Yanase, T., Ohta, T., Koyama, M.: Optuna: a next-generation hyperparameter optimization framework. In: The 25th ACM SIGKDD International Conference on Knowledge Discovery & Data Mining, pp. 2623\u20132631 (2019)","DOI":"10.1145\/3292500.3330701"},{"key":"9962_CR69","unstructured":"Iperf3: A tcp, udp, and sctp network bandwidth measurement tool. https:\/\/iperf.fr\/. Accessed 2024-12-10"}],"container-title":["Journal of Network and Systems Management"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10922-025-09962-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10922-025-09962-9\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10922-025-09962-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,25]],"date-time":"2025-09-25T20:03:17Z","timestamp":1758830597000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10922-025-09962-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,7,7]]},"references-count":69,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2025,10]]}},"alternative-id":["9962"],"URL":"https:\/\/doi.org\/10.1007\/s10922-025-09962-9","relation":{},"ISSN":["1064-7570","1573-7705"],"issn-type":[{"type":"print","value":"1064-7570"},{"type":"electronic","value":"1573-7705"}],"subject":[],"published":{"date-parts":[[2025,7,7]]},"assertion":[{"value":"13 February 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 June 2025","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 June 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 July 2025","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"86"}}