{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,29]],"date-time":"2026-07-29T16:00:42Z","timestamp":1785340842581,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":46,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,4,12]],"date-time":"2026-04-12T00:00:00Z","timestamp":1775952000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,4,12]]},"DOI":"10.1145\/3794763.3794813","type":"proceedings-article","created":{"date-parts":[[2026,7,29]],"date-time":"2026-07-29T15:18:58Z","timestamp":1785338338000},"page":"232-244","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["An Empirical Study on Influence-Based Pretraining Data Selection for Code Large Language Models"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-0384-6613","authenticated-orcid":false,"given":"Chengli","family":"Xing","sequence":"first","affiliation":[{"name":"Peking University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-8422-4522","authenticated-orcid":false,"given":"Zhengran","family":"Zeng","sequence":"additional","affiliation":[{"name":"Peking University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-0967-1333","authenticated-orcid":false,"given":"Gexiang","family":"Fang","sequence":"additional","affiliation":[{"name":"Peking University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1756-7746","authenticated-orcid":false,"given":"Rui","family":"Xie","sequence":"additional","affiliation":[{"name":"Peking University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9331-4716","authenticated-orcid":false,"given":"Wei","family":"Ye","sequence":"additional","affiliation":[{"name":"Peking University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8576-2674","authenticated-orcid":false,"given":"Shikun","family":"Zhang","sequence":"additional","affiliation":[{"name":"Peking University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,29]]},"reference":[{"key":"e_1_3_3_1_2_2","unstructured":"[n. d.]. An Empirical Study on Data Influence-Based Pretraining Data Selection for Code Large Language Models. https:\/\/github.com\/ZZR0\/DIScore. Accessed: 2025-10-23.."},{"key":"e_1_3_3_1_3_2","unstructured":"2023. ChatGPT. Website. https:\/\/openai.com\/blog\/chatgpt."},{"key":"e_1_3_3_1_4_2","unstructured":"Loubna\u00a0Ben Allal Raymond Li Denis Kocetkov Chenghao Mou Christopher Akiki Carlos\u00a0Munoz Ferrandis Niklas Muennighoff Mayank Mishra Alex Gu Manan Dey et\u00a0al. 2023. SantaCoder: don\u2019t reach for the stars! arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2301.03988 (2023)."},{"key":"e_1_3_3_1_5_2","unstructured":"Zachary Ankner Cody Blakeney Kartik Sreenivasan Max Marion Matthew\u00a0L Leavitt and Mansheej Paul. 2024. Perplexed by Perplexity: Perplexity-Based Data Pruning With Small Reference Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2405.20541 (2024)."},{"key":"e_1_3_3_1_6_2","unstructured":"Jacob Austin Augustus Odena Maxwell Nye Maarten Bosma Henryk Michalewski David Dohan Ellen Jiang Carrie Cai Michael Terry Quoc Le et\u00a0al. 2021. Program Synthesis with Large Language Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2108.07732 (2021)."},{"key":"e_1_3_3_1_7_2","unstructured":"Mark Chen Jerry Tworek Heewoo Jun Qiming Yuan Henrique\u00a0Pond\u00e9 de Oliveira\u00a0Pinto Jared Kaplan Harrison Edwards Yuri Burda Nicholas Joseph Greg Brockman Alex Ray Raul Puri Gretchen Krueger Michael Petrov Heidy Khlaaf Girish Sastry Pamela Mishkin Brooke Chan Scott Gray Nick Ryder Mikhail Pavlov Alethea Power Lukasz Kaiser Mohammad Bavarian Clemens Winter Philippe Tillet Felipe\u00a0Petroski Such Dave Cummings Matthias Plappert Fotios Chantzis Elizabeth Barnes Ariel Herbert-Voss William\u00a0Hebgen Guss Alex Nichol Alex Paino Nikolas Tezak Jie Tang Igor Babuschkin Suchir Balaji Shantanu Jain William Saunders Christopher Hesse Andrew\u00a0N. Carr Jan Leike Joshua Achiam Vedant Misra Evan Morikawa Alec Radford Matthew Knight Miles Brundage Mira Murati Katie Mayer Peter Welinder Bob McGrew Dario Amodei Sam McCandlish Ilya Sutskever and Wojciech Zaremba. 2021. Evaluating Large Language Models Trained on Code. CoRR abs\/2107.03374 (2021). arXiv:https:\/\/arXiv.org\/abs\/2107.03374https:\/\/arxiv.org\/abs\/2107.03374"},{"key":"e_1_3_3_1_8_2","unstructured":"Logan Engstrom Axel Feldmann and Aleksander Madry. 2024. Dsdm: Model-aware dataset selection with datamodels. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2401.12926 (2024)."},{"key":"e_1_3_3_1_9_2","doi-asserted-by":"publisher","unstructured":"Sarah Fakhoury Saikat Chakraborty Madan Musuvathi and Shuvendu\u00a0K. Lahiri. 2023. Towards Generating Functionally Correct Code Edits from Natural Language Issue Descriptions. CoRR abs\/2304.03816 (2023). arXiv:https:\/\/arXiv.org\/abs\/2304.0381610.48550\/ARXIV.2304.03816","DOI":"10.48550\/ARXIV.2304.03816"},{"key":"e_1_3_3_1_10_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICSE-FOSE59343.2023.00008"},{"key":"e_1_3_3_1_11_2","unstructured":"Vitaly Feldman and Chiyuan Zhang. 2020. What neural networks memorize and why: Discovering the long tail via influence estimation. Advances in Neural Information Processing Systems 33 (2020) 2881\u20132891."},{"key":"e_1_3_3_1_12_2","doi-asserted-by":"crossref","unstructured":"Shuzheng Gao Cuiyun Gao Yulan He Jichuan Zeng Lunyiu Nie Xin Xia and Michael Lyu. 2023. Code structure\u2013guided transformer for source code summarization. ACM Transactions on Software Engineering and Methodology 32 1 (2023) 1\u201332.","DOI":"10.1145\/3522674"},{"key":"e_1_3_3_1_13_2","unstructured":"Suriya Gunasekar Yi Zhang Jyoti Aneja Caio C\u00e9sar\u00a0Teodoro Mendes Allie Del\u00a0Giorno Sivakanth Gopi Mojan Javaheripi Piero Kauffmann Gustavo de Rosa Olli Saarikivi et\u00a0al. 2023. Textbooks are all you need. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2306.11644 (2023)."},{"key":"e_1_3_3_1_14_2","unstructured":"Daya Guo Qihao Zhu Dejian Yang Zhenda Xie Kai Dong Wentao Zhang Guanting Chen Xiao Bi Yu Wu YK Li et\u00a0al. 2024. DeepSeek-Coder: When the Large Language Model Meets Programming\u2013The Rise of Code Intelligence. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2401.14196 (2024)."},{"key":"e_1_3_3_1_15_2","unstructured":"Shengding Hu Yuge Tu Xu Han Chaoqun He Ganqu Cui Xiang Long Zhi Zheng Yewei Fang Yuxiang Huang Weilin Zhao et\u00a0al. 2024. Minicpm: Unveiling the potential of small language models with scalable training strategies. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2404.06395 (2024)."},{"key":"e_1_3_3_1_16_2","unstructured":"Binyuan Hui Jian Yang Zeyu Cui Jiaxi Yang Dayiheng Liu Lei Zhang Tianyu Liu Jiajun Zhang Bowen Yu Kai Dang et\u00a0al. 2024. Qwen2. 5-coder technical report. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2409.12186 (2024)."},{"key":"e_1_3_3_1_17_2","unstructured":"Srinivasan Iyer Ioannis Konstas Alvin Cheung and Luke Zettlemoyer. 2018. Mapping language to code in programmatic context. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1808.09588 (2018)."},{"key":"e_1_3_3_1_18_2","volume-title":"The Twelfth International Conference on Learning Representations, ICLR 2024, Vienna, Austria, May 7-11, 2024","author":"Jimenez Carlos\u00a0E.","year":"2024","unstructured":"Carlos\u00a0E. Jimenez, John Yang, Alexander Wettig, Shunyu Yao, Kexin Pei, Ofir Press, and Karthik\u00a0R. Narasimhan. 2024. SWE-bench: Can Language Models Resolve Real-world Github Issues?. In The Twelfth International Conference on Learning Representations, ICLR 2024, Vienna, Austria, May 7-11, 2024. OpenReview.net. https:\/\/openreview.net\/forum?id=VTF8yNQM66"},{"key":"e_1_3_3_1_19_2","first-page":"1885","volume-title":"International conference on machine learning","author":"Koh Pang\u00a0Wei","year":"2017","unstructured":"Pang\u00a0Wei Koh and Percy Liang. 2017. Understanding black-box predictions via influence functions. In International conference on machine learning. PMLR, 1885\u20131894."},{"key":"e_1_3_3_1_20_2","unstructured":"Yuhang Lai Chengxi Li Yiming Wang Tianyi Zhang Ruiqi Zhong Luke Zettlemoyer Scott\u00a0Wen tau Yih Daniel Fried Sida Wang and Tao Yu. 2022. DS-1000: A Natural and Reliable Benchmark for Data Science Code Generation. ArXiv abs\/2211.11501 (2022)."},{"key":"e_1_3_3_1_21_2","doi-asserted-by":"crossref","unstructured":"Hugo Lauren\u00e7on Lucile Saulnier Thomas Wang Christopher Akiki Albert Villanova\u00a0del Moral Teven Le\u00a0Scao Leandro Von\u00a0Werra Chenghao Mou Eduardo Gonz\u00e1lez\u00a0Ponferrada Huu Nguyen et\u00a0al. 2022. The bigscience roots corpus: A 1.6 tb composite multilingual dataset. Advances in Neural Information Processing Systems 35 (2022) 31809\u201331826.","DOI":"10.52202\/068431-2306"},{"key":"e_1_3_3_1_22_2","unstructured":"Jinyang Li Binyuan Hui Ge Qu Jiaxi Yang Binhua Li Bowen Li Bailin Wang Bowen Qin Ruiying Geng Nan Huo et\u00a0al. 2024. Can llm already serve as a database interface? a big bench for large-scale database grounded text-to-sqls. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"publisher","unstructured":"Raymond Li Loubna\u00a0Ben Allal Yangtian Zi Niklas Muennighoff Denis Kocetkov Chenghao Mou Marc Marone Christopher Akiki Jia Li Jenny Chim Qian Liu Evgenii Zheltonozhskii Terry\u00a0Yue Zhuo Thomas Wang Olivier Dehaene Mishig Davaadorj Joel Lamy-Poirier Jo\u00e3o Monteiro Oleh Shliazhko Nicolas Gontier Nicholas Meade Armel Zebaze Ming-Ho Yee Logesh\u00a0Kumar Umapathi Jian Zhu Benjamin Lipkin Muhtasham Oblokulov Zhiruo Wang Rudra\u00a0Murthy V Jason Stillerman Siva\u00a0Sankalp Patel Dmitry Abulkhanov Marco Zocca Manan Dey Zhihan Zhang Nour Moustafa-Fahmy Urvashi Bhattacharyya Wenhao Yu Swayam Singh Sasha Luccioni Paulo Villegas Maxim Kunakov Fedor Zhdanov Manuel Romero Tony Lee Nadav Timor Jennifer Ding Claire Schlesinger Hailey Schoelkopf Jan Ebert Tri Dao Mayank Mishra Alex Gu Jennifer Robinson Carolyn\u00a0Jane Anderson Brendan Dolan-Gavitt Danish Contractor Siva Reddy Daniel Fried Dzmitry Bahdanau Yacine Jernite Carlos\u00a0Mu\u00f1oz Ferrandis Sean Hughes Thomas Wolf Arjun Guha Leandro von Werra and Harm de Vries. 2023. StarCoder: may the source be with you! CoRR abs\/2305.06161 (2023). arXiv:https:\/\/arXiv.org\/abs\/2305.0616110.48550\/ARXIV.2305.06161","DOI":"10.48550\/ARXIV.2305.06161"},{"key":"e_1_3_3_1_24_2","unstructured":"Zhe Li Wei Zhao Yige Li and Jun Sun. 2024. Do Influence Functions Work on Large Language Models? arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2409.19998 (2024)."},{"key":"e_1_3_3_1_25_2","doi-asserted-by":"publisher","unstructured":"Junwei Liu Kaixin Wang Yixuan Chen Xin Peng Zhenpeng Chen Lingming Zhang and Yiling Lou. 2024. Large Language Model-Based Agents for Software Engineering: A Survey. CoRR abs\/2409.02977 (2024). arXiv:https:\/\/arXiv.org\/abs\/2409.0297710.48550\/ARXIV.2409.02977","DOI":"10.48550\/ARXIV.2409.02977"},{"key":"e_1_3_3_1_26_2","unstructured":"Yinhan Liu. 2019. Roberta: A robustly optimized bert pretraining approach. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1907.11692 (2019)."},{"key":"e_1_3_3_1_27_2","doi-asserted-by":"publisher","unstructured":"Anton Lozhkov Raymond Li Loubna\u00a0Ben Allal Federico Cassano Joel Lamy-Poirier Nouamane Tazi Ao Tang Dmytro Pykhtar Jiawei Liu Yuxiang Wei Tianyang Liu Max Tian Denis Kocetkov Arthur Zucker Younes Belkada Zijian Wang Qian Liu Dmitry Abulkhanov Indraneil Paul Zhuang Li Wen-Ding Li Megan Risdal Jia Li Jian Zhu Terry\u00a0Yue Zhuo Evgenii Zheltonozhskii Nii Osae\u00a0Osae Dade Wenhao Yu Lucas Krau\u00df Naman Jain Yixuan Su Xuanli He Manan Dey Edoardo Abati Yekun Chai Niklas Muennighoff Xiangru Tang Muhtasham Oblokulov Christopher Akiki Marc Marone Chenghao Mou Mayank Mishra Alex Gu Binyuan Hui Tri Dao Armel Zebaze Olivier Dehaene Nicolas Patry Canwen Xu Julian\u00a0J. McAuley Han Hu Torsten Scholak S\u00e9bastien Paquet Jennifer Robinson Carolyn\u00a0Jane Anderson Nicolas Chapados and et al.2024. StarCoder 2 and The Stack v2: The Next Generation. CoRR abs\/2402.19173 (2024). arXiv:https:\/\/arXiv.org\/abs\/2402.1917310.48550\/ARXIV.2402.19173","DOI":"10.48550\/ARXIV.2402.19173"},{"key":"e_1_3_3_1_28_2","doi-asserted-by":"crossref","unstructured":"Andreas Madsen Siva Reddy and Sarath Chandar. 2022. Post-hoc interpretability for neural nlp: A survey. Comput. Surveys 55 8 (2022) 1\u201342.","DOI":"10.1145\/3546577"},{"key":"e_1_3_3_1_29_2","unstructured":"Garima Pruthi Frederick Liu Satyen Kale and Mukund Sundararajan. 2020. Estimating training data influence by tracing gradient descent. Advances in Neural Information Processing Systems 33 (2020) 19920\u201319930."},{"key":"e_1_3_3_1_30_2","unstructured":"Chen Qian Xin Cong Cheng Yang Weize Chen Yusheng Su Juyuan Xu Zhiyuan Liu and Maosong Sun. 2023. Communicative agents for software development. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2307.07924 (2023)."},{"key":"e_1_3_3_1_31_2","unstructured":"Colin Raffel Noam Shazeer Adam Roberts Katherine Lee Sharan Narang Michael Matena Yanqi Zhou Wei Li and Peter\u00a0J Liu. 2020. Exploring the limits of transfer learning with a unified text-to-text transformer. Journal of machine learning research 21 140 (2020) 1\u201367."},{"key":"e_1_3_3_1_32_2","doi-asserted-by":"publisher","unstructured":"Baptiste Rozi\u00e8re Jonas Gehring Fabian Gloeckle Sten Sootla Itai Gat Xiaoqing\u00a0Ellen Tan Yossi Adi Jingyu Liu Tal Remez J\u00e9r\u00e9my Rapin Artyom Kozhevnikov Ivan Evtimov Joanna Bitton Manish Bhatt Cristian Canton-Ferrer Aaron Grattafiori Wenhan Xiong Alexandre D\u00e9fossez Jade Copet Faisal Azhar Hugo Touvron Louis Martin Nicolas Usunier Thomas Scialom and Gabriel Synnaeve. 2023. Code Llama: Open Foundation Models for Code. CoRR abs\/2308.12950 (2023). arXiv:https:\/\/arXiv.org\/abs\/2308.1295010.48550\/ARXIV.2308.12950","DOI":"10.48550\/ARXIV.2308.12950"},{"key":"e_1_3_3_1_33_2","unstructured":"Noveen Sachdeva Benjamin Coleman Wang-Cheng Kang Jianmo Ni Lichan Hong Ed\u00a0H Chi James Caverlee Julian McAuley and Derek\u00a0Zhiyuan Cheng. 2024. How to Train Data-Efficient LLMs. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2402.09668 (2024)."},{"key":"e_1_3_3_1_34_2","unstructured":"Weisong Sun Chunrong Fang Yudu You Yun Miao Yi Liu Yuekang Li Gelei Deng Shenghan Huang Yuchen Chen Quanjun Zhang et\u00a0al. 2023. Automatic code summarization via chatgpt: How far are we? arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2305.12865 (2023)."},{"key":"e_1_3_3_1_35_2","first-page":"5998","volume-title":"Advances in Neural Information Processing Systems 30: Annual Conference on Neural Information Processing Systems 2017, December 4-9, 2017, Long Beach, CA, USA","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan\u00a0N. Gomez, Lukasz Kaiser, and Illia Polosukhin. 2017. Attention is All you Need. In Advances in Neural Information Processing Systems 30: Annual Conference on Neural Information Processing Systems 2017, December 4-9, 2017, Long Beach, CA, USA, Isabelle Guyon, Ulrike von Luxburg, Samy Bengio, Hanna\u00a0M. Wallach, Rob Fergus, S.\u00a0V.\u00a0N. Vishwanathan, and Roman Garnett (Eds.). 5998\u20136008. https:\/\/proceedings.neurips.cc\/paper\/2017\/hash\/3f5ee243547dee91fbd053c1c4a845aa-Abstract.html"},{"key":"e_1_3_3_1_36_2","unstructured":"Alexander Wettig Aatmik Gupta Saumya Malik and Danqi Chen. 2024. Qurating: Selecting high-quality data for training language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2402.09739 (2024)."},{"key":"e_1_3_3_1_37_2","unstructured":"Chunqiu\u00a0Steven Xia Yinlin Deng Soren Dunn and Lingming Zhang. 2024. Agentless: Demystifying llm-based software engineering agents. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2407.01489 (2024)."},{"key":"e_1_3_3_1_38_2","unstructured":"Mengzhou Xia Sadhika Malladi Suchin Gururangan Sanjeev Arora and Danqi Chen. 2024. Less: Selecting influential data for targeted instruction tuning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2402.04333 (2024)."},{"key":"e_1_3_3_1_39_2","volume-title":"The Thirteenth International Conference on Learning Representations, ICLR 2025, Singapore, April 24-28, 2025","author":"Ye Jiasheng","year":"2025","unstructured":"Jiasheng Ye, Peiju Liu, Tianxiang Sun, Jun Zhan, Yunhua Zhou, and Xipeng Qiu. 2025. Data Mixing Laws: Optimizing Data Mixtures by Predicting Language Modeling Performance. In The Thirteenth International Conference on Learning Representations, ICLR 2025, Singapore, April 24-28, 2025. OpenReview.net. https:\/\/openreview.net\/forum?id=jjCB27TMK3"},{"key":"e_1_3_3_1_40_2","doi-asserted-by":"publisher","unstructured":"Hao Yu Bo Shen Dezhi Ran Jiaxin Zhang Qi Zhang Yuchi Ma Guangtai Liang Ying Li Tao Xie and Qianxiang Wang. 2023. CoderEval: A Benchmark of Pragmatic Code Generation with Generative Pre-trained Models. CoRR abs\/2302.00288 (2023). arXiv:https:\/\/arXiv.org\/abs\/2302.0028810.48550\/ARXIV.2302.00288","DOI":"10.48550\/ARXIV.2302.00288"},{"key":"e_1_3_3_1_41_2","unstructured":"Zichun Yu Spandan Das and Chenyan Xiong. 2024. MATES: Model-Aware Data Selection for Efficient Pretraining with Data Influence Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.06046 (2024)."},{"key":"e_1_3_3_1_42_2","unstructured":"Jerrold\u00a0H Zar. 2014. Spearman rank correlation: overview. Wiley StatsRef: Statistics Reference Online (2014)."},{"key":"e_1_3_3_1_43_2","doi-asserted-by":"publisher","DOI":"10.1145\/3650212.3652115"},{"key":"e_1_3_3_1_44_2","doi-asserted-by":"crossref","unstructured":"Fengji Zhang Bei Chen Yue Zhang Jin Liu Daoguang Zan Yi Mao Jian-Guang Lou and Weizhu Chen. 2023. Repocoder: Repository-level code completion through iterative retrieval and generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2303.12570 (2023).","DOI":"10.18653\/v1\/2023.emnlp-main.151"},{"key":"e_1_3_3_1_45_2","unstructured":"Peiyuan Zhang Guangtao Zeng Tianduo Wang and Wei Lu. 2024. Tinyllama: An open-source small language model. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2401.02385 (2024)."},{"key":"e_1_3_3_1_46_2","doi-asserted-by":"publisher","unstructured":"Ziyin Zhang Chaoyu Chen Bingchang Liu Cong Liao Zi Gong Hang Yu Jianguo Li and Rui Wang. 2023. A Survey on Language Models for Code. CoRR abs\/2311.07989 (2023). arXiv:https:\/\/arXiv.org\/abs\/2311.0798910.48550\/ARXIV.2311.07989","DOI":"10.48550\/ARXIV.2311.07989"},{"key":"e_1_3_3_1_47_2","unstructured":"Wayne\u00a0Xin Zhao Kun Zhou Junyi Li Tianyi Tang Xiaolei Wang Yupeng Hou Yingqian Min Beichen Zhang Junjie Zhang Zican Dong et\u00a0al. 2023. A survey of large language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2303.18223 (2023)."}],"event":{"name":"ICPC '26: 34th IEEE\/ACM International Conference on Program Comprehension","location":"Rio de Janeiro , Brazil","acronym":"ICPC '26","sponsor":["SIGSOFT ACM Special Interest Group on Software Engineering"]},"container-title":["Proceedings of the 2026 34th IEEE\/ACM International Conference on Program Comprehension"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3794763.3794813","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,29]],"date-time":"2026-07-29T15:21:36Z","timestamp":1785338496000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3794763.3794813"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,12]]},"references-count":46,"alternative-id":["10.1145\/3794763.3794813","10.1145\/3794763"],"URL":"https:\/\/doi.org\/10.1145\/3794763.3794813","relation":{},"subject":[],"published":{"date-parts":[[2026,4,12]]},"assertion":[{"value":"2026-07-29","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}