{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,23]],"date-time":"2025-08-23T05:26:29Z","timestamp":1755926789653,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":38,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,8,10]],"date-time":"2023-08-10T00:00:00Z","timestamp":1691625600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,8,10]]},"DOI":"10.1145\/3589013.3596676","type":"proceedings-article","created":{"date-parts":[[2023,8,11]],"date-time":"2023-08-11T07:53:54Z","timestamp":1691740434000},"page":"19-26","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["mSIRM: Cost-Efficient and SLO-aware ML Load Balancing on Fog and Multi-Cloud Network"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-2575-6436","authenticated-orcid":false,"given":"Chetan","family":"Phalak","sequence":"first","affiliation":[{"name":"TCS Research, Mumbai, India"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-7398-0749","authenticated-orcid":false,"given":"Dheeraj","family":"Chahal","sequence":"additional","affiliation":[{"name":"TCS Research, Mumbai, India"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2720-3276","authenticated-orcid":false,"given":"Manju","family":"Ramesh","sequence":"additional","affiliation":[{"name":"TCS Research, Mumbai, India"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3712-1784","authenticated-orcid":false,"given":"Rekha","family":"Singhal","sequence":"additional","affiliation":[{"name":"TCS Research, New York, NY, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,8,11]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"2023. Amazon Lambda. https:\/\/aws.amazon.com\/lambda\/"},{"key":"e_1_3_2_1_2_1","unstructured":"2023. Amazon Lambda Pricing. https:\/\/aws.amazon.com\/lambda\/pricing\/"},{"key":"e_1_3_2_1_3_1","unstructured":"2023. Amazon SageMaker. https:\/\/aws.amazon.com\/sagemaker\/"},{"key":"e_1_3_2_1_4_1","unstructured":"2023. Amazon Web Services. https:\/\/aws.amazon.com\/"},{"key":"e_1_3_2_1_5_1","unstructured":"2023. AWS Lambda timeout. https:\/\/aws.amazon.com\/about-aws\/whats-new\/2018\/10\/aws-lambda-supports-functions-that-can-run-up-to-15-minutes\/"},{"key":"e_1_3_2_1_6_1","unstructured":"2023. Azure Machine Learning. https:\/\/azure.microsoft.com\/en-in\/services\/machine-learning\/"},{"key":"e_1_3_2_1_7_1","unstructured":"2023. Cloud Functions. https:\/\/cloud.google.com\/functions"},{"key":"e_1_3_2_1_8_1","unstructured":"2023. Google Cloud. https:\/\/cloud.google.com\/"},{"key":"e_1_3_2_1_9_1","unstructured":"2023. Introduction to Azure Functions. https:\/\/learn.microsoft.com\/en-us\/azure\/azure-functions\/functions-overview"},{"key":"e_1_3_2_1_10_1","unstructured":"2023. JMeter. https:\/\/jmeter.apache.org\/"},{"key":"e_1_3_2_1_11_1","unstructured":"2023. Microsoft Azure. https:\/\/azure.microsoft.com\/en-in\/"},{"key":"e_1_3_2_1_12_1","unstructured":"2023. Vertex AI. https:\/\/cloud.google.com\/vertex-ai"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/SC41405.2020.00073"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.14778\/3547305.3547313"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1007\/s12083-021-01125-2"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/Confluence47617.2020.9057799"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3447545.3451184"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/3452413.3464789"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3526060.3535458"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA51647.2021.00049"},{"key":"e_1_3_2_1_21_1","volume-title":"Prashanth Thinakaran, Bikash Sharma, Mahmut Taylan Kandemir, and Chita R Das.","author":"Gunasekaran Jashwant Raj","year":"2022","unstructured":"Jashwant Raj Gunasekaran, Cyan Subhra Mishra, Prashanth Thinakaran, Bikash Sharma, Mahmut Taylan Kandemir, and Chita R Das. 2022. Cocktail: A multidimensional optimization for model serving in cloud. In USENIX NSDI. 1041--1057."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CLOUD.2019.00043"},{"key":"e_1_3_2_1_23_1","volume-title":"NISER: normalized item and session representations with graph neural networks. arXiv preprint arXiv:1909.04276","author":"Gupta Priyanka","year":"2019","unstructured":"Priyanka Gupta, Diksha Garg, Pankaj Malhotra, Lovekesh Vig, and Gautam M Shroff. 2019. NISER: normalized item and session representations with graph neural networks. arXiv preprint arXiv:1909.04276 (2019)."},{"key":"e_1_3_2_1_24_1","volume-title":"Inferall: Coordinated Optimization for Machine Learning Inference Serving in Public Cloud.","author":"Kumar Pramod","year":"2021","unstructured":"Pramod Kumar. 2021. Inferall: Coordinated Optimization for Machine Learning Inference Serving in Public Cloud. (2021)."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/3458817.3476168"},{"key":"e_1_3_2_1_26_1","volume-title":"MLProxy: SLA-Aware Reverse Proxy for Machine Learning Inference Serving on Serverless Computing Platforms. arXiv preprint arXiv:2202.11243","author":"Mahmoudi Nima","year":"2022","unstructured":"Nima Mahmoudi and Hamzeh Khazaei. 2022. MLProxy: SLA-Aware Reverse Proxy for Machine Learning Inference Serving on Serverless Computing Platforms. arXiv preprint arXiv:2202.11243 (2022)."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/i-Society.2014.7009018"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/COMSNETS56262.2023.10041384"},{"key":"e_1_3_2_1_29_1","volume-title":"Abolfazl Toroghi Haghighat, and Amin Keshavarzi","author":"Shahidani Fatemeh Ramezani","year":"2023","unstructured":"Fatemeh Ramezani Shahidani, Arezoo Ghasemi, Abolfazl Toroghi Haghighat, and Amin Keshavarzi. 2023. Task scheduling in edge-fog-cloud architecture: a multi-objective load balancing approach using reinforcement learning algorithm. Computing (2023), 1--23."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/IC2E52221.2021.00028"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.compeleceng.2022.107789"},{"key":"e_1_3_2_1_32_1","volume-title":"Splice: An Automated Framework for Cost-and Performance-Aware Blending of Cloud Services. In 2022 22nd IEEE International Symposium on Cluster, Cloud and Internet Computing (CCGrid). IEEE, 119--128","author":"Son Myungjun","year":"2022","unstructured":"Myungjun Son, Shruti Mohanty, Jashwant Raj Gunasekaran, Aman Jain, Mahmut Taylan Kandemir, George Kesidis, and Bhuvan Urgaonkar. 2022. Splice: An Automated Framework for Cost-and Performance-Aware Blending of Cloud Services. In 2022 22nd IEEE International Symposium on Cluster, Cloud and Internet Computing (CCGrid). IEEE, 119--128."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1007\/s12652-020-01768-8"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.future.2019.10.043"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/IPCCC47392.2019.8958770"},{"key":"e_1_3_2_1_36_1","volume-title":"Serving and Optimizing Machine Learning Workflows on Heterogeneous Infrastructures. arXiv preprint arXiv:2205.04713","author":"Wu Yongji","year":"2022","unstructured":"Yongji Wu, Matthew Lentz, Danyang Zhuo, and Yao Lu. 2022. Serving and Optimizing Machine Learning Workflows on Heterogeneous Infrastructures. arXiv preprint arXiv:2205.04713 (2022)."},{"key":"e_1_3_2_1_37_1","volume-title":"Dynamic resource allocation for load balancing in fog environment. Wireless Communications and Mobile Computing 2018","author":"Xu Xiaolong","year":"2018","unstructured":"Xiaolong Xu, Shucun Fu, Qing Cai, Wei Tian, Wenjie Liu, Wanchun Dou, Xingming Sun, and Alex X Liu. 2018. Dynamic resource allocation for load balancing in fog environment. Wireless Communications and Mobile Computing 2018 (2018)."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCC.2020.3006751"}],"event":{"name":"HPDC '23: The 32nd International Symposium on High-Performance Parallel and Distributed Computing","sponsor":["SIGHPC ACM Special Interest Group on High Performance Computing, Special Interest Group on High Performance Computing","SIGARCH ACM Special Interest Group on Computer Architecture"],"location":"Orlando FL USA","acronym":"HPDC '23"},"container-title":["Proceedings of the 13th Workshop on AI and Scientific Computing at Scale using Flexible Computing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3589013.3596676","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3589013.3596676","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T16:36:15Z","timestamp":1750178175000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3589013.3596676"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,8,10]]},"references-count":38,"alternative-id":["10.1145\/3589013.3596676","10.1145\/3589013"],"URL":"https:\/\/doi.org\/10.1145\/3589013.3596676","relation":{},"subject":[],"published":{"date-parts":[[2023,8,10]]},"assertion":[{"value":"2023-08-11","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}