{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T09:15:55Z","timestamp":1783242955144,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":28,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T00:00:00Z","timestamp":1783209600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,5]]},"DOI":"10.1145\/3805760.3814902","type":"proceedings-article","created":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T08:46:13Z","timestamp":1783241173000},"page":"129-134","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["When AI Coding Assistants Leak Training Data: A Study of LLM Memorization in Code Generation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-9958-5836","authenticated-orcid":false,"given":"Xiaoyu","family":"Cheng","sequence":"first","affiliation":[{"name":"University of Waterloo, Department of Electrical and Computer Engineering, Waterloo, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3756-4673","authenticated-orcid":false,"given":"Kundi","family":"Yao","sequence":"additional","affiliation":[{"name":"Ontario Tech University, Faculty of Engineering and Applied Science, Oshawa, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1529-3216","authenticated-orcid":false,"given":"Pengyu","family":"Nie","sequence":"additional","affiliation":[{"name":"University of Waterloo, Department of Electrical and Computer Engineering, Waterloo, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6222-7444","authenticated-orcid":false,"given":"Weiyi","family":"Shang","sequence":"additional","affiliation":[{"name":"University of Waterloo, Department of Electrical and Computer Engineering, Waterloo, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,5]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","unstructured":"Ali Al-Kaswan Maliheh Izadi and Arie van Deursen. 2023. Targeted Attack on GPT-Neo for the SATML Language Model Data Extraction Challenge. doi:10. 48550\/arXiv.2302.07735 arXiv:2302.07735 [cs]. 10.48550\/arXiv.2302.07735","DOI":"10.48550\/arXiv.2302.07735"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/3597503.3639133"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","unstructured":"Nicholas Carlini Chang Liu \u00dalfar Erlingsson Jernej Kos and Dawn Song. 2019. The Secret Sharer: Evaluating and Testing Unintended Memorization in Neural Networks. doi:10.48550\/arXiv.1802.08232 arXiv:1802.08232 [cs]. 10.48550\/arXiv.1802.08232","DOI":"10.48550\/arXiv.1802.08232"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2012.07805"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","unstructured":"DeepSeek-AI Daya Guo Dejian Yang and et al. 2025. DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning. doi:10.48550\/arXiv. 2501.12948 arXiv:2501.12948 [cs]. 10.48550\/arXiv.2501.12948","DOI":"10.48550\/arXiv"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","unstructured":"Yijiang River Dong Hongzhou Lin Mikhail Belkin Ramon Huerta and Ivan Vuli\u0107. 2024. Unmemorization in Large Language Models via Self-Distillation and Deliberate Imagination. doi:10.48550\/arXiv.2402.10052 arXiv:2402.10052 [cs] version: 1. 10.48550\/arXiv.2402.10052","DOI":"10.48550\/arXiv.2402.10052"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","unstructured":"Vitaly Feldman. 2021. Does Learning Require Memorization? A Short Tale about a Long Tail. doi:10.48550\/arXiv.1906.05271 arXiv:1906.05271 [cs]. 10.48550\/arXiv.1906.05271","DOI":"10.48550\/arXiv.1906.05271"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/2790755.2790797"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2101.00027"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","unstructured":"Aaron Grattafiori Abhimanyu Dubey Abhinav Jauhri and et al. 2024. The Llama 3 Herd of Models. doi:10.48550\/arXiv.2407.21783 arXiv:2407.21783 [cs]. 10.48550\/arXiv.2407.21783","DOI":"10.48550\/arXiv.2407.21783"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","unstructured":"Yufei Guo Muzhe Guo Juntao Su Zhou Yang Mengqiu Zhu Hongfei Li Mengyang Qiu and Shuo Shuo Liu. 2024. Bias in Large Language Models: Origin Evaluation and Mitigation. doi:10.48550\/arXiv.2411.10915 arXiv:2411.10915 [cs]. 10.48550\/arXiv.2411.10915","DOI":"10.48550\/arXiv.2411.10915"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","unstructured":"Abhimanyu Hans Yuxin Wen Neel Jain John Kirchenbauer Hamid Kazemi Prajwal Singhania Siddharth Singh Gowthami Somepalli Jonas Geiping Abhinav Bhatele and Tom Goldstein. 2024. Be like a Goldfish Don't Memorize! Mitigating Memorization in Generative LLMs. doi:10.48550\/arXiv.2406.10209 arXiv:2406.10209 [cs]. 10.48550\/arXiv.2406.10209","DOI":"10.48550\/arXiv.2406.10209"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/2902362"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","unstructured":"Geoffrey Hinton Oriol Vinyals and Jeff Dean. 2015. Distilling the Knowledge in a Neural Network. doi:10.48550\/arXiv.1503.02531 arXiv:1503.02531 [stat]. 10.48550\/arXiv.1503.02531","DOI":"10.48550\/arXiv.1503.02531"},{"key":"e_1_3_2_1_15_1","volume-title":"Effective code membership inference for code completion models via adversarial prompts. arXiv preprint arXiv:2511.15107","author":"Jiang Yuan","year":"2025","unstructured":"Yuan Jiang, Zehao Li, Shan Huang, Christoph Treude, Xiaohong Su, and Tiantian Wang. 2025. Effective code membership inference for code completion models via adversarial prompts. arXiv preprint arXiv:2511.15107 (2025)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","unstructured":"Katherine Lee Daphne Ippolito Andrew Nystrom Chiyuan Zhang Douglas Eck Chris Callison-Burch and Nicholas Carlini. 2022. Deduplicating Training Data Makes Language Models Better. doi:10.48550\/arXiv.2107.06499 arXiv:2107.06499 [cs]. 10.48550\/arXiv.2107.06499","DOI":"10.48550\/arXiv.2107.06499"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1126\/science.abq1158"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2402.19173"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"crossref","first-page":"15121","DOI":"10.52202\/075280-0665","article-title":"Data portraits: Recording foundation model training data","volume":"36","author":"Marone Marc","year":"2023","unstructured":"Marc Marone and Benjamin Van Durme. 2023. Data portraits: Recording foundation model training data. Advances in Neural Information Processing Systems 36 (2023), 15121-15135.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_20_1","volume-title":"2025 IEEE\/ACM 47th International Conference on Software Engineering (ICSE). IEEE, 2880-2892","author":"Nie Yuqing","year":"2025","unstructured":"Yuqing Nie, Chong Wang, Kailong Wang, Guoai Xu, Guosheng Xu, and Haoyu Wang. 2025. Decoding secret memorization in code llms through token-level characterization. In 2025 IEEE\/ACM 47th International Conference on Software Engineering (ICSE). IEEE, 2880-2892."},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","unstructured":"Rafiqul Rabin Sean McGregor and Nick Judd. 2025. Malicious and Unintentional Disclosure Risks in Large Language Models for Code Generation. doi:10.48550\/ arXiv.2503.22760 arXiv:2503.22760 [cs]. 10.48550\/arXiv.2503.22760","DOI":"10.48550\/arXiv.2503.22760"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","unstructured":"Fabio Salerno Ali Al-Kaswan and Maliheh Izadi. 2025. How Much Do Code Language Models Remember? An Investigation on Data Extraction Attacks before and after Fine-tuning. doi:10.48550\/arXiv.2501.17501 arXiv:2501.17501 [cs]. 10.48550\/arXiv.2501.17501","DOI":"10.48550\/arXiv.2501.17501"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","unstructured":"Yashothara Shanmugarasa Ming Ding M. A. P. Chamikara and Thierry Rakotoarivelo. 2025. SoK: The Privacy Paradox of Large Language Models: Advancements Privacy Risks and Mitigation. doi:10.1145\/3708821.3733888 arXiv:2506.12699 [cs]. 10.1145\/3708821.3733888","DOI":"10.1145\/3708821.3733888"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/SP.2017.41"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2310.01424"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","unstructured":"Weiwei Xu Kai Gao Hao He and Minghui Zhou. 2025. LiCoEval: Evaluating LLMs on License Compliance in Code Generation. doi:10.48550\/arXiv.2408.02487 arXiv:2408.02487 [cs] version: 3. 10.48550\/arXiv.2408.02487","DOI":"10.48550\/arXiv.2408.02487"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","unstructured":"Xiaohan Xu Ming Li Chongyang Tao Tao Shen Reynold Cheng Jinyang Li Can Xu Dacheng Tao and Tianyi Zhou. 2024. A Survey on Knowledge Distillation of Large Language Models. doi:10.48550\/arXiv.2402.13116 arXiv:2402.13116 [cs]. 10.48550\/arXiv.2402.13116","DOI":"10.48550\/arXiv.2402.13116"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3597503.3639074"}],"event":{"name":"AIware '26: 3rd ACM International Conference on AI-Powered Software","location":"Montreal QC Canada","acronym":"AIware '26","sponsor":["SIGSOFT ACM Special Interest Group on Software Engineering"]},"container-title":["Proceedings of the 3rd ACM International Conference on AI-Powered Software"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3805760.3814902","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T08:50:16Z","timestamp":1783241416000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805760.3814902"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,5]]},"references-count":28,"alternative-id":["10.1145\/3805760.3814902","10.1145\/3805760"],"URL":"https:\/\/doi.org\/10.1145\/3805760.3814902","relation":{},"subject":[],"published":{"date-parts":[[2026,7,5]]},"assertion":[{"value":"2026-07-05","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}