{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T04:10:32Z","timestamp":1750219832266,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":25,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,9,14]],"date-time":"2023-09-14T00:00:00Z","timestamp":1694649600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,9,14]]},"DOI":"10.1145\/3604915.3610249","type":"proceedings-article","created":{"date-parts":[[2023,9,14]],"date-time":"2023-09-14T22:40:23Z","timestamp":1694731223000},"page":"1071-1074","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["From Research to Production: Towards Scalable and Sustainable Neural Recommendation Models on Commodity CPU Hardware"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5042-2856","authenticated-orcid":false,"given":"Anshumali","family":"Shrivastava","sequence":"first","affiliation":[{"name":"Rice University\/ThirdAI, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-9836-7714","authenticated-orcid":false,"given":"Vihan","family":"Lakshman","sequence":"additional","affiliation":[{"name":"Research, ThirdAI, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-0410-2977","authenticated-orcid":false,"given":"Tharun","family":"Medini","sequence":"additional","affiliation":[{"name":"ThirdAI, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-5301-6421","authenticated-orcid":false,"given":"Nicholas","family":"Meisburger","sequence":"additional","affiliation":[{"name":"ThirdAI Corp, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-6907-2236","authenticated-orcid":false,"given":"Joshua","family":"Engels","sequence":"additional","affiliation":[{"name":"ThirdAI, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-9485-2065","authenticated-orcid":false,"given":"David","family":"Torres Ramos","sequence":"additional","affiliation":[{"name":"ThirdAI, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4021-0016","authenticated-orcid":false,"given":"Benito","family":"Geordie","sequence":"additional","affiliation":[{"name":"ThirdAI, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-8422-1733","authenticated-orcid":false,"given":"Pratik","family":"Pranav","sequence":"additional","affiliation":[{"name":"ThirdAI, India"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-6692-1092","authenticated-orcid":false,"given":"Shubh","family":"Gupta","sequence":"additional","affiliation":[{"name":"ThirdAI, India"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-3429-4348","authenticated-orcid":false,"given":"Yashwanth","family":"Adunukota","sequence":"additional","affiliation":[{"name":"ThirdAI, India"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-3147-5214","authenticated-orcid":false,"given":"Siddharth","family":"Jain","sequence":"additional","affiliation":[{"name":"ThirdAI, India"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,9,14]]},"reference":[{"volume-title":"2021 IEEE International Symposium on High-Performance Computer Architecture (HPCA). IEEE, 802\u2013814","author":"Acun Bilge","key":"e_1_3_2_1_1_1","unstructured":"Bilge Acun, Matthew Murphy, Xiaodong Wang, Jade Nie, Carole-Jean Wu, and Kim Hazelwood. [n. d.]. Understanding training efficiency of deep learning recommendation models at scale. In 2021 IEEE International Symposium on High-Performance Computer Architecture (HPCA). IEEE, 802\u2013814."},{"key":"e_1_3_2_1_2_1","volume-title":"International Conference on Learning Representations.","author":"Chen Beidi","year":"2020","unstructured":"Beidi Chen, Zichang Liu, Binghui Peng, Zhaozhuo Xu, Jonathan\u00a0Lingjie Li, Tri Dao, Zhao Song, Anshumali Shrivastava, and Christopher Re. 2020. MONGOOSE: A learnable LSH framework for efficient neural network training. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_3_1","first-page":"291","article-title":"Slide: In defense of smart algorithms over hardware acceleration for large-scale deep learning systems","volume":"2","author":"Chen Beidi","year":"2020","unstructured":"Beidi Chen, Tharun Medini, James Farwell, Charlie Tai, Anshumali Shrivastava, 2020. Slide: In defense of smart algorithms over hardware acceleration for large-scale deep learning systems. Proceedings of Machine Learning and Systems 2 (2020), 291\u2013306.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_4_1","unstructured":"Criteo. 2014. Criteo Conversion Logs Dataset. https:\/\/ailab.criteo.com\/ressources\/."},{"key":"e_1_3_2_1_5_1","first-page":"762","article-title":"Random Offset Block Embedding (ROBE) for compressed embedding tables in deep learning recommendation systems","volume":"4","author":"Desai Aditya","year":"2022","unstructured":"Aditya Desai, Li Chou, and Anshumali Shrivastava. 2022. Random Offset Block Embedding (ROBE) for compressed embedding tables in deep learning recommendation systems. Proceedings of Machine Learning and Systems 4 (2022), 762\u2013778.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_6_1","first-page":"33961","article-title":"The trade-offs of model size in large recommendation models: 100GB to 10MB Criteo-tb DLRM model","volume":"35","author":"Desai Aditya","year":"2022","unstructured":"Aditya Desai and Anshumali Shrivastava. 2022. The trade-offs of model size in large recommendation models: 100GB to 10MB Criteo-tb DLRM model. Advances in Neural Information Processing Systems 35 (2022), 33961\u201333972.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_7_1","volume-title":"The lottery ticket hypothesis: Finding sparse, trainable neural networks. arXiv preprint arXiv:1803.03635","author":"Frankle Jonathan","year":"2018","unstructured":"Jonathan Frankle and Michael Carbin. 2018. The lottery ticket hypothesis: Finding sparse, trainable neural networks. arXiv preprint arXiv:1803.03635 (2018)."},{"key":"e_1_3_2_1_8_1","volume-title":"Pruning neural networks at initialization: Why are we missing the mark?arXiv preprint arXiv:2009.08576","author":"Frankle Jonathan","year":"2020","unstructured":"Jonathan Frankle, Gintare\u00a0Karolina Dziugaite, Daniel\u00a0M Roy, and Michael Carbin. 2020. Pruning neural networks at initialization: Why are we missing the mark?arXiv preprint arXiv:2009.08576 (2020)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/HPCA47549.2020.00047"},{"key":"e_1_3_2_1_10_1","volume-title":"Deep compression: Compressing deep neural networks with pruning, trained quantization and huffman coding. arXiv preprint arXiv:1510.00149","author":"Han Song","year":"2015","unstructured":"Song Han, Huizi Mao, and William\u00a0J Dally. 2015. Deep compression: Compressing deep neural networks with pruning, trained quantization and huffman coding. arXiv preprint arXiv:1510.00149 (2015)."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/2505515.2505665"},{"key":"e_1_3_2_1_12_1","unstructured":"Michael Kan. 2021. Inside the GPU shortage: Why you still can\u2019t buy a graphics card. https:\/\/www.pcmag.com\/news\/inside-the-gpu-shortage-why-you-still-cant-buy-a-graphics-card"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3447548.3467101"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3534678.3539164"},{"key":"e_1_3_2_1_15_1","first-page":"336","article-title":"Mlperf training benchmark","volume":"2","author":"Mattson Peter","year":"2020","unstructured":"Peter Mattson, Christine Cheng, Gregory Diamos, Cody Coleman, Paulius Micikevicius, David Patterson, Hanlin Tang, Gu-Yeon Wei, Peter Bailis, Victor Bittorf, 2020. Mlperf training benchmark. Proceedings of Machine Learning and Systems 2 (2020), 336\u2013349.","journal-title":"Proceedings of Machine Learning and Systems"},{"key":"e_1_3_2_1_16_1","volume-title":"BOLT: An Automated Deep Learning Framework for Training and Deploying Large-Scale Neural Networks on Commodity CPU Hardware. arXiv preprint arXiv:2303.17727","author":"Meisburger Nicholas","year":"2023","unstructured":"Nicholas Meisburger, Vihan Lakshman, Benito Geordie, Joshua Engels, David\u00a0Torres Ramos, Pratik Pranav, Benjamin Coleman, Benjamin Meisburger, Shubh Gupta, Yashwanth Adunukota, 2023. BOLT: An Automated Deep Learning Framework for Training and Deploying Large-Scale Neural Networks on Commodity CPU Hardware. arXiv preprint arXiv:2303.17727 (2023)."},{"key":"e_1_3_2_1_17_1","volume-title":"An updated duet model for passage re-ranking. arXiv preprint arXiv:1903.07666","author":"Mitra Bhaskar","year":"2019","unstructured":"Bhaskar Mitra and Nick Craswell. 2019. An updated duet model for passage re-ranking. arXiv preprint arXiv:1903.07666 (2019)."},{"key":"e_1_3_2_1_18_1","volume-title":"Deep learning recommendation model for personalization and recommendation systems. arXiv preprint arXiv:1906.00091","author":"Naumov Maxim","year":"2019","unstructured":"Maxim Naumov, Dheevatsa Mudigere, Hao-Jun\u00a0Michael Shi, Jianyu Huang, Narayanan Sundaraman, Jongsoo Park, Xiaodong Wang, Udit Gupta, Carole-Jean Wu, Alisson\u00a0G Azzolini, 2019. Deep learning recommendation model for personalization and recommendation systems. arXiv preprint arXiv:1906.00091 (2019)."},{"key":"e_1_3_2_1_19_1","volume-title":"Deep Learning Recommendation Model for Personalization and Recommendation Systems. CoRR abs\/1906.00091","author":"Naumov Maxim","year":"2019","unstructured":"Maxim Naumov, Dheevatsa Mudigere, Hao-Jun\u00a0Michael Shi, Jianyu Huang, Narayanan Sundaraman, Jongsoo Park, Xiaodong Wang, Udit Gupta, Carole-Jean Wu, Alisson\u00a0G. Azzolini, Dmytro Dzhulgakov, Andrey Mallevich, Ilia Cherniavskii, Yinghai Lu, Raghuraman Krishnamoorthi, Ansha Yu, Volodymyr Kondratenko, Stephanie Pereira, Xianjie Chen, Wenlin Chen, Vijay Rao, Bill Jia, Liang Xiong, and Misha Smelyanskiy. 2019. Deep Learning Recommendation Model for Personalization and Recommendation Systems. CoRR abs\/1906.00091 (2019). https:\/\/arxiv.org\/abs\/1906.00091"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3292500.3330759"},{"key":"e_1_3_2_1_21_1","volume-title":"Passage Re-ranking with BERT. arXiv preprint arXiv:1901.04085","author":"Nogueira Rodrigo","year":"2019","unstructured":"Rodrigo Nogueira and Kyunghyun Cho. 2019. Passage Re-ranking with BERT. arXiv preprint arXiv:1901.04085 (2019)."},{"key":"e_1_3_2_1_22_1","unstructured":"Oleg Rybakov Vijai Mohan Avishkar Misra Scott LeGrand Rejith Joseph Kiuk Chung Siddharth Singh Qian You Eric Nalisnick and Runfei Luo. 2018. The effectiveness of a two-layer neural network for recommendations. (2018)."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3097983.3098035"},{"key":"e_1_3_2_1_24_1","volume-title":"Energy and policy considerations for deep learning in NLP. arXiv preprint arXiv:1906.02243","author":"Strubell Emma","year":"2019","unstructured":"Emma Strubell, Ananya Ganesh, and Andrew McCallum. 2019. Energy and policy considerations for deep learning in NLP. arXiv preprint arXiv:1906.02243 (2019)."},{"key":"e_1_3_2_1_25_1","volume-title":"The end of moore\u2019s law: A new beginning for information technology. Computing in science & engineering 19, 2","author":"Theis N","year":"2017","unstructured":"Thomas\u00a0N Theis and H-S\u00a0Philip Wong. 2017. The end of moore\u2019s law: A new beginning for information technology. Computing in science & engineering 19, 2 (2017), 41\u201350."}],"event":{"name":"RecSys '23: Seventeenth ACM Conference on Recommender Systems","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web","SIGAI ACM Special Interest Group on Artificial Intelligence","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data","SIGIR ACM Special Interest Group on Information Retrieval","SIGCHI ACM Special Interest Group on Computer-Human Interaction","SIGecom Special Interest Group on Economics and Computation"],"location":"Singapore Singapore","acronym":"RecSys '23"},"container-title":["Proceedings of the 17th ACM Conference on Recommender Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3604915.3610249","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3604915.3610249","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T16:46:35Z","timestamp":1750178795000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3604915.3610249"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,9,14]]},"references-count":25,"alternative-id":["10.1145\/3604915.3610249","10.1145\/3604915"],"URL":"https:\/\/doi.org\/10.1145\/3604915.3610249","relation":{},"subject":[],"published":{"date-parts":[[2023,9,14]]},"assertion":[{"value":"2023-09-14","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}