{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,6]],"date-time":"2026-07-06T11:03:27Z","timestamp":1783335807261,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":38,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,8,24]],"date-time":"2024-08-24T00:00:00Z","timestamp":1724457600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,8,25]]},"DOI":"10.1145\/3637528.3671542","type":"proceedings-article","created":{"date-parts":[[2024,8,25]],"date-time":"2024-08-25T04:55:12Z","timestamp":1724561712000},"page":"4784-4792","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":12,"title":["Large Scale Generative AI Text Applied to Sports and Music"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2167-5133","authenticated-orcid":false,"given":"Aaron","family":"Baughman","sequence":"first","affiliation":[{"name":"IBM, RTP, NC, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9738-4850","authenticated-orcid":false,"given":"Eduardo","family":"Morales","sequence":"additional","affiliation":[{"name":"IBM, Atlanta, GA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-0510-3342","authenticated-orcid":false,"given":"Rahul","family":"Agarwal","sequence":"additional","affiliation":[{"name":"IBM, New York, NY, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-0797-3926","authenticated-orcid":false,"given":"Gozde","family":"Akay","sequence":"additional","affiliation":[{"name":"IBM, Fredericton, NB, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6399-0679","authenticated-orcid":false,"given":"Rogerio","family":"Feris","sequence":"additional","affiliation":[{"name":"IBM, Boston, MA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-9741-2808","authenticated-orcid":false,"given":"Tony","family":"Johnson","sequence":"additional","affiliation":[{"name":"IBM, Atlanta, GA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-3680-9191","authenticated-orcid":false,"given":"Stephen","family":"Hammer","sequence":"additional","affiliation":[{"name":"IBM, Atlanta, GA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2524-2068","authenticated-orcid":false,"given":"Leonid","family":"Karlinsky","sequence":"additional","affiliation":[{"name":"IBM, Boston, MA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,8,24]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/219717.219748"},{"key":"e_1_3_2_2_2_1","volume-title":"https:\/\/www.asimovinstitute.org\/neural-network-zoo\/","author":"Asimov Institute","year":"2024","unstructured":"Asimov Institute, https:\/\/www.asimovinstitute.org\/neural-network-zoo\/, 14 Jan 2024."},{"key":"e_1_3_2_2_3_1","volume-title":"The Pile: An 800GB Dataset of Diverse Text for Language Modeling\". arXiv abs\/2101.00027","author":"Gao Leo","year":"2020","unstructured":"Leo Gao, et al. \"The Pile: An 800GB Dataset of Diverse Text for Language Modeling\". arXiv abs\/2101.00027 (2020). https:\/\/arxiv.org\/abs\/2101.00027"},{"key":"e_1_3_2_2_4_1","unstructured":"\"Why we built an AI supercomputer in the cloud\". 07 Feb 2023. https:\/\/research.ibm.com\/blog\/AI-supercomputer-Vela-GPU-cluster"},{"key":"e_1_3_2_2_5_1","volume-title":"An overview\". arXiv pdf\/1404.7828","author":"J. Schmuidhuber","year":"2015","unstructured":"J. Schmuidhuber \"Deep learning in neural networks: An overview\". arXiv pdf\/1404.7828 (2015). https:\/\/arxiv.org\/pdf\/1404.7828.pdf"},{"key":"e_1_3_2_2_6_1","volume-title":"Generative Adversarial Networks\", arXiv abs\/1406.2661","author":"Goodfellow Ian","year":"2014","unstructured":"Ian Goodfellow, Jean Pouget-Abadie, Mehdi Mirza, and et al. \"Generative Adversarial Networks\", arXiv abs\/1406.2661 (2014). https:\/\/arxiv.org\/abs\/1406.2661"},{"key":"e_1_3_2_2_7_1","volume-title":"Attention Is All You Need\", arXiv abs\/1706.03762","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, and et al. \"Attention Is All You Need\", arXiv abs\/1706.03762 (2017). https:\/\/arxiv.org\/abs\/1706.03762"},{"key":"e_1_3_2_2_8_1","first-page":"1","article-title":"Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer","volume":"21","author":"Raffel Colin","year":"2020","unstructured":"Colin Raffel, Noam Shaker, Adam Roberts, Katherine Lee, Sharan Narang, Michael Matena, Yanqi Zhou, Wei Li, Peter J. Liu. \"Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer\". Journal of Machine Learning Research 21 (2020) 1--67.","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_2_2_9_1","volume-title":"Pre-training of Deep Bidirectional Transformers for Language Understanding\". arXiv abs\/1810.04805","author":"Devlin Jacob","year":"2018","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, Dristinal Toutanova. \"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding\". arXiv abs\/1810.04805 (2018). https:\/\/arxiv.org\/abs\/1810.04805"},{"key":"e_1_3_2_2_11_1","volume-title":"PaLM: Scaling Language Modeling with Pathways\". arXiv abs\/2204.02311","author":"Chowdhery Aakanksha","year":"2022","unstructured":"Aakanksha Chowdhery, Sharan Narang, Jacob Devlin, and et al. \"PaLM: Scaling Language Modeling with Pathways\". arXiv abs\/2204.02311 (2022). https:\/\/arxiv.org\/abs\/2204.02311"},{"key":"e_1_3_2_2_12_1","unstructured":"IBM Research. \"Granite Foundation Models\" 30 Nov 2023. https:\/\/www.ibm.com\/downloads\/cas\/X9W4O6BM"},{"key":"e_1_3_2_2_13_1","volume-title":"BLOOM: A 176B-Parameter Open-Access Multilingual Language Model\", arXiv abs\/2211.05100","author":"Scao Teven Le","year":"2022","unstructured":"Teven Le Scao, Angela Fan, Christopher Akiki, and et al. \"BLOOM: A 176B-Parameter Open-Access Multilingual Language Model\", arXiv abs\/2211.05100 (2022). https:\/\/arxiv.org\/abs\/2211.05100"},{"key":"e_1_3_2_2_14_1","volume-title":"ICLR","author":"Wei Jason","year":"2022","unstructured":"Jason Wei, Maarten Bosma, Vincent Zhao, Kelvin Guu, Adams Wei Yu, Brian Lester, Nan Du, Andrew M. Dai, Quoc V. Le. \"Finetuned Language Models are Zero-Shot Learners\". ICLR 2022."},{"key":"e_1_3_2_2_15_1","first-page":"10661","article-title":"The remarkable, yet not extraordinary, human brain as a scaled-up primate brain and its associated cost","volume":"109","author":"Herculano-Houzel Suzana","year":"2012","unstructured":"Suzana Herculano-Houzel. \"The remarkable, yet not extraordinary, human brain as a scaled-up primate brain and its associated cost\", PNAS Research Article, Biological Sciences, Vol 109 pp 10661--10668, 22 June 2012.","journal-title":"PNAS Research Article, Biological Sciences"},{"key":"e_1_3_2_2_16_1","volume-title":"Llama 2: Open Foundation and Fine-Tuned Chat Models\". arXiv abs\/2307.09288","author":"Touvron Hugo","year":"2023","unstructured":"Hugo Touvron, Louis Martin, Kevin Stone , et al. \"Llama 2: Open Foundation and Fine-Tuned Chat Models\". arXiv abs\/2307.09288 (2023). https:\/\/arxiv.org\/abs\/2307.09288."},{"key":"e_1_3_2_2_17_1","volume-title":"Distilling Step-by-Step! Outperforming Larger Language Models with Less Training Data and Smaller Model Sizes\", arXiv abs\/2305.02301","author":"Hsieh Cheng-Yu","year":"2023","unstructured":"Cheng-Yu Hsieh, Chun-Liang Li, Chih-Kuan Yeh, and et al. \"Distilling Step-by-Step! Outperforming Larger Language Models with Less Training Data and Smaller Model Sizes\", arXiv abs\/2305.02301 (2023). https:\/\/arxiv.org\/abs\/2305.02301"},{"key":"e_1_3_2_2_18_1","unstructured":"IBM Data and AI Team \"How foundation models and data stores unlock the business potential of generative AI\" 1 Aug 2023. https:\/\/www.ibm.com\/blog\/how-foundation-models-and-data-stores-unlock-the-business-potential-of-generative-ai\/"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1038\/s42256-023-00626--4"},{"key":"e_1_3_2_2_20_1","volume-title":"ICLR","author":"Press Ofir","year":"2022","unstructured":"Ofir Press, Noah Smith, and Mike Lewis, \"Train Short, Test Long: Attention with Linear Biases Enables Input Length Extrapolation\", ICLR 2022."},{"key":"e_1_3_2_2_21_1","volume-title":"arXiv abs\/2005.11401","author":"Lewis Patrick","year":"2021","unstructured":"Patrick Lewis and Ethan Perez. \"Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks\", arXiv abs\/2005.11401 (2021). https:\/\/arxiv.org\/abs\/2005.11401"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"crossref","unstructured":"Seetharami Seelam. \"Hardware-Middleware System co-design for flexible training of foundation models in the cloud\". Keynote Middleware 2022. https:\/\/middleware-conf.github.io\/2022\/keynote-speakers\/","DOI":"10.1145\/3568161.3568317"},{"key":"e_1_3_2_2_23_1","volume-title":"Language models are few-shot learners\". Advances in neural information processing systems, 33: 1877--1901","author":"Brown Tom","year":"2020","unstructured":"Tom Brown, Benjamin Mann, Nick Ryder, Melanie Subbiah, Jared D Kaplan, Prafulla Dhariwal, Arvind Neelakantan, Prenav Shyam, Girish Sastry, Amanda Alkell, et al. \"Language models are few-shot learners\". Advances in neural information processing systems, 33:1877--1901, 2020."},{"key":"e_1_3_2_2_24_1","volume-title":"A Survey on ChatGPT and Beyond\". arXiv abs\/2304.13712","author":"Yang Jingfeng","year":"2023","unstructured":"Jingfeng Yang, Hongye Jin, Ruixiang Tang, Xiaotian Han, Qizhang Feng, Haoming Jiang, Bing Yin, Xia Hu. \"Harnessing the Power of LLMs in Practice: A Survey on ChatGPT and Beyond\". arXiv abs\/2304.13712 (2023). https:\/\/arxiv.org\/abs\/2304.13712."},{"key":"e_1_3_2_2_25_1","volume-title":"arXiv abs\/2201.05273","author":"Li Junyi","year":"2022","unstructured":"Junyi Li, Tianyi Tang, Wayne Zhao, Jian-Yun Nie, Ji-Rong Wen. \"Pre-trained Language Models for Text Generation\". arXiv abs\/2201.05273 (2022). https:\/\/arxiv.org\/abs\/2201.05273"},{"key":"e_1_3_2_2_26_1","volume-title":"arXiv abs\/2307.07164","author":"Wang Liang","year":"2023","unstructured":"Liang Wang, Nan Yang, Guru Wei. \"Learning to Retrieve In-Context Examples for Large Language Models\". arXiv abs\/2307.07164 (2023). https:\/\/arxiv.org\/abs\/2307.07164"},{"key":"e_1_3_2_2_27_1","volume-title":"Near-linear Scaling for Training Gigantic Model on Public Cloud\". arXiv abs\/2205.00119","author":"Zhang Zhen","year":"2022","unstructured":"Zhen Zhang, Shuai Zheng, Yida Wang, Justin Chiu, George Karypis, Trishul Chilimbi, Mu Li, Xin Jin. \"MiCS: Near-linear Scaling for Training Gigantic Model on Public Cloud\". arXiv abs\/2205.00119 (2022). https:\/\/arxiv.org\/abs\/2205.00119"},{"key":"e_1_3_2_2_28_1","volume-title":"HuggingFace's Transformers: State-of-the-art Natural Language Processing\". arXiv abs\/1910.03771","author":"Wolf Thomas","year":"2020","unstructured":"Thomas Wolf, Lysandre Debug, et al. \"HuggingFace's Transformers: State-of-the-art Natural Language Processing\". arXiv abs\/1910.03771 (2020). https:\/\/arxiv.org\/abs\/1910.03771"},{"key":"e_1_3_2_2_29_1","volume-title":"PyTorch: An Imperative Style","author":"Paszke Adam","year":"1912","unstructured":"Adam Paszke, Sam Gross, et al. \"PyTorch: An Imperative Style, High-Performance Deep Learning Library\". arXiv abs\/1912.01703 (2019). https:\/\/arxiv.org\/abs\/1912.01703"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.1038\/s42256-023-00626--4"},{"key":"e_1_3_2_2_31_1","volume-title":"Philip Wallis et al. \"LoRA: Low-Rank Adaptation of Large Language Models\", arXiv abs\/2106.09685","author":"Hu Edward","year":"2021","unstructured":"Edward Hu, Yelong Shen, Philip Wallis et al. \"LoRA: Low-Rank Adaptation of Large Language Models\", arXiv abs\/2106.09685 (2021). https:\/\/arxiv.org\/abs\/2106.09685"},{"key":"e_1_3_2_2_32_1","first-page":"4582","volume-title":"11th International Joint Conference on Natural Language Processing","author":"Li Xiang Lisa","year":"2021","unstructured":"Xiang Lisa Li, Percy Liang. \"Prefix-Tuning: Optimizing Continuous Prompts for Generation\", 11th International Joint Conference on Natural Language Processing, pp 4582--4597 1--6 Aug 2021."},{"key":"e_1_3_2_2_33_1","volume-title":"Why Can GPT Learn In-Context\" Langauge Models Implicitly Perform Gradient Descent as Meta-Optimizer\", arXiv abs\/2212.10559","author":"Dai Damai","year":"2023","unstructured":"Damai Dai, Yutao Sun, Li Dong, et al. \"Why Can GPT Learn In-Context\" Langauge Models Implicitly Perform Gradient Descent as Meta-Optimizer\", arXiv abs\/2212.10559 (2023). https:\/\/arxiv.org\/abs\/2212.10559"},{"key":"e_1_3_2_2_34_1","volume-title":"SELF-INSTRUCT: Aligning Language Models with Self-Generated Instructions\", arXiv abs\/2212.10560","author":"Wang Yizhong","year":"2023","unstructured":"Yizhong Wang, Yeganeh Kordi, Swaroop Mishra, et al. \"SELF-INSTRUCT: Aligning Language Models with Self-Generated Instructions\", arXiv abs\/2212.10560 (2023). https:\/\/arxiv.org\/abs\/2212.10560"},{"key":"e_1_3_2_2_35_1","volume-title":"Efficient Large-Scale Language Model Training of GPU Clusters Using Megatron-LM\". arXiv abs\/2104.04473","author":"Narayanan Deepak","year":"2021","unstructured":"Deepak Narayanan, Mohammad Shoeybi, Jared Casper, and et al. \"Efficient Large-Scale Language Model Training of GPU Clusters Using Megatron-LM\". arXiv abs\/2104.04473 (2021). https:\/\/arxiv.org\/abs\/2104.04473"},{"key":"e_1_3_2_2_36_1","volume-title":"Low-cost Training of Massive Deep Learning Models\", abs\/2111.04007","author":"Athlur Sanjith","year":"2021","unstructured":"Sanjith Athlur, Natika Saran, Mathias Sivathanu, et al. \"Varuna: Scalable, Low-cost Training of Massive Deep Learning Models\", abs\/2111.04007 (2021). https:\/\/arxiv.org\/abs\/2111.04007"},{"key":"e_1_3_2_2_37_1","volume-title":"A Comprehensive Performance Study of Large Language Models on Novel AI Accelerators\", arXiv abs\/2310.04607","author":"Emani Muali","year":"2023","unstructured":"Muali Emani, Sam Foreman, Varuni Sastry, and et al. \"A Comprehensive Performance Study of Large Language Models on Novel AI Accelerators\", arXiv abs\/2310.04607 (2023). https:\/\/arxiv.org\/abs\/2310.04607"},{"key":"e_1_3_2_2_38_1","unstructured":"Elizabeth Reid. \"Supercharging Search with generative AI\" 10 May 2023 https:\/\/blog.google\/products\/search\/generative-ai-search\/"},{"key":"e_1_3_2_2_39_1","volume-title":"Survey of Hallucination in Natural Language Generation\", arXiv abs\/2202.03629","author":"Ji Ziwei","year":"2022","unstructured":"Ziwei Ji, Nayeon Lee, Rita, Rita Frieske, et al. \"Survey of Hallucination in Natural Language Generation\", arXiv abs\/2202.03629 (2022). https:\/\/arxiv.org\/abs\/2202.03629. https:\/\/arxiv.org\/abs\/2005.14165"}],"event":{"name":"KDD '24: The 30th ACM SIGKDD Conference on Knowledge Discovery and Data Mining","location":"Barcelona Spain","acronym":"KDD '24","sponsor":["SIGMOD ACM Special Interest Group on Management of Data","SIGKDD ACM Special Interest Group on Knowledge Discovery in Data"]},"container-title":["Proceedings of the 30th ACM SIGKDD Conference on Knowledge Discovery and Data Mining"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3637528.3671542","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3637528.3671542","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T00:04:19Z","timestamp":1750291459000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3637528.3671542"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,8,24]]},"references-count":38,"alternative-id":["10.1145\/3637528.3671542","10.1145\/3637528"],"URL":"https:\/\/doi.org\/10.1145\/3637528.3671542","relation":{},"subject":[],"published":{"date-parts":[[2024,8,24]]},"assertion":[{"value":"2024-08-24","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}