{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T21:15:30Z","timestamp":1783804530227,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":90,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,9,11]],"date-time":"2024-09-11T00:00:00Z","timestamp":1726012800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"NSF (National Science Foundation)","award":["1801751 and 1956364"],"award-info":[{"award-number":["1801751 and 1956364"]}]},{"name":"UC Noyce Initiative","award":[""],"award-info":[{"award-number":[""]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,9,11]]},"DOI":"10.1145\/3650212.3680342","type":"proceedings-article","created":{"date-parts":[[2024,9,11]],"date-time":"2024-09-11T11:44:25Z","timestamp":1726055065000},"page":"1061-1072","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":12,"title":["UniTSyn: A Large-Scale Dataset Capable of Enhancing the Prowess of Large Language Models for Program Testing"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5389-7128","authenticated-orcid":false,"given":"Yifeng","family":"He","sequence":"first","affiliation":[{"name":"University of California at Davis, Davis, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7900-3439","authenticated-orcid":false,"given":"Jiabo","family":"Huang","sequence":"additional","affiliation":[{"name":"Tencent, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0648-0255","authenticated-orcid":false,"given":"Yuyang","family":"Rong","sequence":"additional","affiliation":[{"name":"University of California at Davis, Davis, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0709-4877","authenticated-orcid":false,"given":"Yiwen","family":"Guo","sequence":"additional","affiliation":[{"name":"Unaffiliated, n.n., China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-2715-0410","authenticated-orcid":false,"given":"Ethan","family":"Wang","sequence":"additional","affiliation":[{"name":"University of California at Davis, Davis, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4072-0710","authenticated-orcid":false,"given":"Hao","family":"Chen","sequence":"additional","affiliation":[{"name":"University of California at Davis, Davis, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,9,11]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Wasi Ahmad Saikat Chakraborty Baishakhi Ray and Kai-Wei Chang. 2020."},{"key":"e_1_3_2_1_2_1","volume-title":"InProceedings of the 58th Annual Meeting of the Association for Computational Linguistics.","author":"A","unstructured":"A transformer-based approach for source code summarization. InProceedings of the 58th Annual Meeting of the Association for Computational Linguistics."},{"key":"e_1_3_2_1_3_1","volume-title":"ttps:\/\/doi.org \/10.18653\/v1\/","author":"Association for Computational Linguistics, Online, ( July 2020 ) h.","year":"2020","unstructured":"Association for Computational Linguistics, Online, ( July 2020 ) h. ttps:\/\/doi.org \/10.18653\/v1\/ 2020.acl-main. 449."},{"key":"e_1_3_2_1_4_1","unstructured":"Loubna Ben Allal Raymond Li Denis Kocetkov Chenghao Mou Christopher Akiki et al. 2023. Santacoder: don't reach for the stars !arXiv preprint arXiv: 2301. 03988."},{"key":"e_1_3_2_1_5_1","unstructured":"Dimitrios Athanasiou Ariadi Nugroho Joost Visser and Andy Zaidman. 2014."},{"key":"e_1_3_2_1_6_1","first-page":"1100","article-title":"code quality and its relation to issue handling performanceI","volume":"40","author":"Test","unstructured":"Test code quality and its relation to issue handling performanceI. EEE Transactions on Software Engineering, 40, 11, 1100-1125.","journal-title":"EEE Transactions on Software Engineering"},{"key":"e_1_3_2_1_7_1","unstructured":"Moritz Beller Georgios Gousios Annibale Panichella and Andy Zaidman. 2015."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/2786805.2786843"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICSE"},{"key":"e_1_3_2_1_10_1","unstructured":"Bei Chen Fengji Zhang Anh Nguyen Daoguang Zan Zeqi Lin et al. 2023."},{"key":"e_1_3_2_1_11_1","volume-title":"Henrique Ponde de Oliveira Pinto, et al","author":"Chen Mark","year":"2021","unstructured":"Mark Chen, Jerry Tworek, Heewoo Jun, Qiming Yuan, Henrique Ponde de Oliveira Pinto, et al. 2021. Evaluating large language models trained on code."},{"key":"e_1_3_2_1_12_1","unstructured":"( 2021 ). arXiv: 2107.03374 [cs.LG]."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"crossref","unstructured":"Peng Chen and Hao Chen. 2018. Angora: eficient fuzzing by principled search.","DOI":"10.1109\/SP.2018.00046"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/SP.2018.00046"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3319535.3363225"},{"key":"e_1_3_2_1_16_1","unstructured":"Yuting Chen Ting Su Chengnian Sun Zhendong Su and Jianjun Zhao. 2016."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/2980983.2908095"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"crossref","first-page":"268","DOI":"10.1145\/351240.351266","volume-title":"InProceedings of the Fifth ACM SIGPLAN International Conference on Functional Programming (ICFP '00)","author":"Claessen Koen","year":"2000","unstructured":"Koen Claessen and John Hughes. 2000. Quickcheck: a lightweight tool for random testing of haskell programs. InProceedings of the Fifth ACM SIGPLAN International Conference on Functional Programming (ICFP '00). Association for Computing Machinery, New York, NY, USA, 268-279.isbn: 1581132026."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","unstructured":"https:\/\/doi.org\/10.1145\/351240.351266. 10.1145\/351240.351266","DOI":"10.1145\/351240.351266"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"crossref","unstructured":"Daniel DeFreez Aditya V. Thakur and Cindy Rubio-Gonz\u00e1lez. 2018. Path-based function embedding and its application to error-handling specification mining.","DOI":"10.1145\/3236024.3236059"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3236024.3236059"},{"key":"e_1_3_2_1_22_1","unstructured":"1972. Chapter i: notes on structured programming. Structured Programming."},{"key":"e_1_3_2_1_23_1","unstructured":"Academic Press Ltd. GBR 1-82. isbn: 0122005503."},{"key":"e_1_3_2_1_24_1","volume-title":"Lahiri","author":"Dinella Elizabeth","year":"2022","unstructured":"Elizabeth Dinella, Gabriel Ryan, Todd Mytkowicz, and Shuvendu K. Lahiri. 2022."},{"key":"e_1_3_2_1_25_1","first-page":"2130","volume-title":"InProceedings of the 44th International Conference on Software Engineering (ICSE '22)","author":"Toga","unstructured":"Toga: a neural method for test oracle generation. InProceedings of the 44th International Conference on Software Engineering (ICSE '22). Association for Computing Machinery, Pitsburgh, Pennsylvania, 2130-2141. isbn: 9781450392211."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","unstructured":"https:\/\/doi.org\/10.1145\/3510003.3510141. 10.1145\/3510003.3510141","DOI":"10.1145\/3510003.3510141"},{"key":"e_1_3_2_1_27_1","unstructured":"Zhangyin Feng Daya Guo Duyu Tang Nan Duan Xiaocheng Feng et al. 2020."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1"},{"key":"e_1_3_2_1_29_1","volume-title":"InProceedings of the 14th USENIX Conference on Ofensive Technologies, 10-10","author":"Fioraldi Andrea","year":"2020","unstructured":"Andrea Fioraldi, Dominik Maier, Heiko Ei\u00dffeldt, and Marc Heuse. 2020. Afl++ combining incremental steps of fuzzing research. InProceedings of the 14th USENIX Conference on Ofensive Technologies, 10-10."},{"key":"e_1_3_2_1_30_1","unstructured":"Daniel Fried Armen Aghajanyan Jessy Lin Sida Wang Eric Wallace et al. 2022."},{"key":"e_1_3_2_1_31_1","unstructured":"Incoder: a generative model for code infilling and synthesis. arXiv preprint arXiv:2204. 05999."},{"key":"e_1_3_2_1_32_1","volume-title":"Language Server Protocol and Implementation","author":"Gunasinghe Nadeeshaan","unstructured":"Nadeeshaan Gunasinghe and Nipuna Marcus. 2021. Language Server Protocol and Implementation. Springer. Chap. 8."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"crossref","unstructured":"Daya Guo Shuai Lu Nan Duan Yanlin Wang Ming Zhou et al. 2022. Unixcoder: unified cross-modal pre-training for code representation. In ACL.","DOI":"10.18653\/v1\/2022.acl-long.499"},{"key":"e_1_3_2_1_34_1","volume-title":"Edmund SL Lam, and Xiaoyin Wang","author":"Hassan Foyzul","year":"2017","unstructured":"Foyzul Hassan, Shaikh Mostafa, Edmund SL Lam, and Xiaoyin Wang. 2017."},{"key":"e_1_3_2_1_35_1","volume-title":"In2017 ACM\/IEEE International Symposium on Empirical Software Engineering and Measurement (ESEM). IEEE, 38-47","author":"Automatic","unstructured":"Automatic building of java projects in software repositories: a study on feasibility and challenges. In2017 ACM\/IEEE International Symposium on Empirical Software Engineering and Measurement (ESEM). IEEE, 38-47."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3238147.3238183"},{"key":"e_1_3_2_1_37_1","unstructured":"Jiabo Huang Jianyu Zhao Yuyang Rong Yiwen Guo Yifeng He et al. 2023."},{"key":"e_1_3_2_1_38_1","unstructured":"( 2023 ). arXiv: 2309.09980 [cs.SE]."},{"key":"e_1_3_2_1_39_1","volume-title":"Codesearchnet challenge: evaluating the state of semantic code search. ( 2020 ). arXiv","author":"Husain Hamel","year":"1909","unstructured":"Hamel Husain, Ho-Hsiang Wu, Tiferet Gazit, Miltiadis Allamanis, and Marc Brockschmidt. 2020. Codesearchnet challenge: evaluating the state of semantic code search. ( 2020 ). arXiv: 1909. 09436 [cs.LG]."},{"key":"e_1_3_2_1_40_1","unstructured":"Srinivasan Iyer Ioannis Konstas Alvin Cheung and Luke Zetlemoyer. 2016."},{"key":"e_1_3_2_1_41_1","first-page":"16","volume-title":"Proceedings of the 54th Annual Meeting of the Association for Computational Linguistics. Association for Computational Linguistics","author":"Summarizing","year":"2016","unstructured":"Summarizing source code using a neural atention model. In Proceedings of the 54th Annual Meeting of the Association for Computational Linguistics. Association for Computational Linguistics, Berlin, Germany, (Aug. 2016 ) h.ttps:\/\/doi.o rg\/10.18653\/v1\/ P16-1195."},{"key":"e_1_3_2_1_42_1","unstructured":"[n. d.] Junit-quickcheck: property-based testing junit-styleh.ttps:\/\/pholser.git hub. io\/junit-quickcheck\/site\/1.0\/."},{"key":"e_1_3_2_1_43_1","volume-title":"Unit Testing Principles, Practices, and Paterns","author":"Khorikov Vladimir","unstructured":"Vladimir Khorikov. 2020. Unit Testing Principles, Practices, and Paterns. Simon and Schuster."},{"key":"e_1_3_2_1_44_1","unstructured":"Diederik P Kingma and Jimmy Ba. 2014. Adam: a method for stochastic optimization. arXiv preprint arXiv:1412. 6980."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"crossref","unstructured":"James Kirkpatrick Razvan Pascanu Neil Rabinowitz Joel Veness Guillaume Desjardins et al. 2017. Overcoming catastrophic forgeting in neural networks.","DOI":"10.1073\/pnas.1611835114"},{"key":"e_1_3_2_1_46_1","unstructured":"Proceedings of the national academy of sciences 114 13 3521-3526."},{"key":"e_1_3_2_1_47_1","unstructured":"2023. The stack: 3 TB of permissively licensed source code. Transactions on Machine Learning Research."},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"crossref","first-page":"215","DOI":"10.1109\/APSEC.2014.42","volume-title":"In2014 21st Asia-Pacific Software Engineering Conference.","volume":"1","author":"Kochhar Pavneet Singh","year":"2014","unstructured":"Pavneet Singh Kochhar, Ferdian Thung, David Lo, and Julia Lawall. 2014. An empirical study on the adequacy of testing in open source projects. In2014 21st Asia-Pacific Software Engineering Conference. Vol. 1. IEEE, 215-222."},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISSRE"},{"key":"e_1_3_2_1_50_1","volume-title":"In2019 IEEE\/ACM 41st International Conference on Software Engineering (ICSE). IEEE, 795-806","author":"LeClair Alexander","year":"2019","unstructured":"Alexander LeClair, Siyuan Jiang, and Collin McMillan. 2019. A neural model for generating natural language summaries of program subroutines. In2019 IEEE\/ACM 41st International Conference on Software Engineering (ICSE). IEEE, 795-806."},{"key":"e_1_3_2_1_51_1","volume-title":"Yangtian Zi, Niklas Muennighof, Denis Kocetkov, et al.","author":"Li Raymond","year":"2023","unstructured":"Raymond Li, Loubna Ben allal, Yangtian Zi, Niklas Muennighof, Denis Kocetkov, et al. 2023. Starcoder: may the source be with you! Transactions on Machine Learning Research."},{"key":"e_1_3_2_1_52_1","first-page":"1092","article-title":"2022. Competition-level code generation with alphacode","volume":"378","author":"Li Yujia","unstructured":"Yujia Li, David Choi, Junyoung Chung, Nate Kushman, Julian Schritwieser, et al. 2022. Competition-level code generation with alphacode. Science, 378, 6624, 1092-1097. https:\/\/www.science.org\/doi\/abs\/10.1126\/science.abq1158 eprint: https:\/\/www.science.org\/doi\/pdf\/10.1126\/science.abq1158.","journal-title":"Science"},{"key":"e_1_3_2_1_53_1","volume-title":"International Conference on Architectural Support for Programming Languages and Operating Systems (ASPLOS). (Apr. 19-23","author":"Liu Ziheng","year":"2021","unstructured":"Ziheng Liu, Shuofei Zhu, Boqin Qin, Hao Chen, and Song Linhai. 2021. Automatically detecting and fixing concurrency bugs in go software systems. In International Conference on Architectural Support for Programming Languages and Operating Systems (ASPLOS). (Apr. 19-23, 2021 )."},{"key":"e_1_3_2_1_54_1","unstructured":"Ilya Loshchilov and Frank Huter. 2016. SGDR: stochastic gradient descent with restarts. CoRR. http:\/\/arxiv.org\/abs\/1608.03983."},{"key":"e_1_3_2_1_55_1","unstructured":"Shuai Lu Daya Guo Shuo Ren Junjie Huang Alexey Svyatkovskiy et al. 2021."},{"key":"e_1_3_2_1_56_1","volume-title":"InThirty-fith Conference on Neural Information Processing Systems Datasets and Benchmarks Track (Round 1 ).","unstructured":"CodeXGLUE: a machine learning benchmark dataset for code understanding and generation. InThirty-fith Conference on Neural Information Processing Systems Datasets and Benchmarks Track (Round 1 )."},{"key":"e_1_3_2_1_57_1","volume-title":"InThe Twelfth International Conference on Learning Representations.","author":"Luo Ziyang","year":"2024","unstructured":"Ziyang Luo, Can Xu, Pu Zhao, Qingfeng Sun, Xiubo Geng, et al. 2024. Wizardcoder: empowering code large language models with evol-instruct. InThe Twelfth International Conference on Learning Representations."},{"key":"e_1_3_2_1_58_1","first-page":"43","article-title":"2019. Hypothesis: a new approach to property-based testing","volume":"4","author":"MacIver David R","year":"1891","unstructured":"David R MacIver, Zac Hatfield-Dodds, et al. 2019. Hypothesis: a new approach to property-based testing. Journal of Open Source Software, 4, 43, 1891.","journal-title":"Journal of Open Source Software"},{"key":"e_1_3_2_1_59_1","volume-title":"Language server protocol. (Jan. 11","year":"2024","unstructured":"Microsoft. 2024. Language server protocol. (Jan. 11, 2024 ). https:\/\/microsoft.git hub.io\/language-server-protocol\/."},{"key":"e_1_3_2_1_60_1","unstructured":"Nachiappan Nagappan and Thomas Ball. 2010. Evidence-based failure prediction. InMaking Software. Greg Wilson Andy Oram (Ed.) O'REILLY. Chap. 23."},{"key":"e_1_3_2_1_61_1","volume-title":"In2023 IEEE\/ACM 45th International Conference on Software Engineering (ICSE). IEEE, 2111-2123","author":"Nie Pengyu","year":"2023","unstructured":"Pengyu Nie, Rahul Banerjee, Junyi Jessy Li, Raymond J Mooney, and Milos Gligoric. 2023. Learning deep semantics for test completion. In2023 IEEE\/ACM 45th International Conference on Software Engineering (ICSE). IEEE, 2111-2123."},{"key":"e_1_3_2_1_62_1","volume-title":"InThe Eleventh International Conference on Learning Representations.","author":"Nijkamp Erik","year":"2023","unstructured":"Erik Nijkamp, Hiroaki Hayashi, Caiming Xiong, Silvio Savarese, and Yingbo Zhou. 2023. Codegen2: lessons for training llms on programming and natural languages. InThe Eleventh International Conference on Learning Representations."},{"key":"e_1_3_2_1_63_1","unstructured":"Erik Nijkamp Bo Pang Hiroaki Hayashi Lifu Tu Huan Wang et al. 2023."},{"key":"e_1_3_2_1_64_1","unstructured":"[n. d.] Parametrizing tests.https:\/\/docs.pytest. org\/en\/8.0.x\/example\/parametri ze. html."},{"key":"e_1_3_2_1_65_1","unstructured":"Ruchir Puri David S. Kung Geert Janssen Wei Zhang Giacomo Domeniconi et al. 2021. Codenet: a large-scale ai for code dataset for learning a diversity of coding tasks. InNeural Information Processing Systems (NeuralIPS)."},{"key":"e_1_3_2_1_66_1","unstructured":"Alec Radford Jefrey Wu Rewon Child David Luan Dario Amodei et al. 2019."},{"key":"e_1_3_2_1_67_1","unstructured":"Language models are unsupervised multitask learners. OpenAI blog 1 8 9."},{"key":"e_1_3_2_1_68_1","volume-title":"In2023 38th IEEE\/ACM International Conference on Automated Software Engineering. IEEE.","unstructured":"2023. Cat-lm training language models on aligned code and tests. In2023 38th IEEE\/ACM International Conference on Automated Software Engineering. IEEE."},{"key":"e_1_3_2_1_69_1","volume-title":"InProceedings of the 6th Workshop on Formal Integrated Development Environment, 3-18","author":"Rask Jonas Kjaer","year":"2021","unstructured":"Jonas Kjaer Rask, Frederik Palludan Madsen, Nick Batle, Hugo Daniel Macedo, and Peter Gorm Larsen. 2021. The specification language server protocol: a proposal for standardised lsp extensions. InProceedings of the 6th Workshop on Formal Integrated Development Environment, 3-18."},{"key":"e_1_3_2_1_70_1","volume-title":"International Conference on Security and Privacy in Communication Systems. Springer, 360-380","author":"Rong Yuyang","year":"2020","unstructured":"Yuyang Rong, Peng Chen, and Hao Chen. 2020. Integrity: finding integer errors by targeted fuzzing. In International Conference on Security and Privacy in Communication Systems. Springer, 360-380."},{"key":"e_1_3_2_1_71_1","doi-asserted-by":"crossref","unstructured":"Yuyang Rong Chibin Zhang Jianzhong Liu and Hao Chen. 2024. Valkyrie: improving fuzzing performance through deterministic techniquesJ. ournal of Systems and Software 209 111886.","DOI":"10.1016\/j.jss.2023.111886"},{"key":"e_1_3_2_1_72_1","unstructured":"2023. Code llama: open foundation models for code. ( 2023 ). arXiv: 2308.12950 [cs.CL]."},{"key":"e_1_3_2_1_73_1","doi-asserted-by":"crossref","unstructured":"Max Sch\u00e4fer Sarah Nadi Aryaz Eghbali and Frank Tip. 2023. An empirical evaluation of using large language models for automated unit test generation.","DOI":"10.1109\/TSE.2023.3334955"},{"key":"e_1_3_2_1_74_1","doi-asserted-by":"crossref","unstructured":"Kosta Serebryany. 2016. Continuous fuzzing with libfuzzer and addresssanitizer.","DOI":"10.1109\/SecDev.2016.043"},{"key":"e_1_3_2_1_75_1","doi-asserted-by":"publisher","DOI":"10.1109\/SecDev.2016.043"},{"key":"e_1_3_2_1_76_1","volume-title":"Shao Kun Deng, and Neel Sundaresan","author":"Tufano Michele","year":"2021","unstructured":"Michele Tufano, Dawn Drain, Alexey Svyatkovskiy, Shao Kun Deng, and Neel Sundaresan. 2021. Unit test case generation with transformers and focal context."},{"key":"e_1_3_2_1_77_1","volume-title":"arXiv","year":"2009","unstructured":"( 2021 ). arXiv: 2009. 05617 [cs.SE]."},{"key":"e_1_3_2_1_78_1","doi-asserted-by":"publisher","unstructured":"2024. UniTSyn.https:\/\/doi.org\/10.5281\/zenodo.12639546. 10.5281\/zenodo.12639546","DOI":"10.5281\/zenodo.12639546"},{"key":"e_1_3_2_1_79_1","unstructured":"Yue Wang Hung Le Akhilesh Gotmare Nghi Bui Junnan Li et al. 2023."},{"key":"e_1_3_2_1_80_1","volume-title":"Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, 1069-1088","unstructured":"Codet5+: open code large language models for code understanding and generation. In Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, 1069-1088."},{"key":"e_1_3_2_1_81_1","volume-title":"InProceedings of the 2021 Conference on Empirical Methods in Natural Language Processing, 8696-8708","author":"Wang Yue","year":"2021","unstructured":"Yue Wang, Weishi Wang, Shafiq Joty, and Steven CH Hoi. 2021. Codet5: identifieraware unified pre-trained encoder-decoder models for code understanding and generation. InProceedings of the 2021 Conference on Empirical Methods in Natural Language Processing, 8696-8708."},{"key":"e_1_3_2_1_82_1","doi-asserted-by":"publisher","DOI":"10.1145\/3377811.3380429"},{"key":"e_1_3_2_1_83_1","unstructured":"Weimin Xiong Yiwen Guo and Hao Chen. 2023. The program testing ability of large language models for code. arXiv preprint arXiv:2310. 05727."},{"key":"e_1_3_2_1_84_1","doi-asserted-by":"publisher","DOI":"10.1145\/1138929.1138949"},{"key":"e_1_3_2_1_85_1","volume-title":"Proceedings of the 46th IEEE\/ACM International Conference on Software Engineering, 1-12","author":"Yu Hao","year":"2024","unstructured":"Hao Yu, Bo Shen, Dezhi Ran, Jiaxin Zhang, Qi Zhang, et al. 2024. Codereval: a benchmark of pragmatic code generation with generative pre-trained models. In Proceedings of the 46th IEEE\/ACM International Conference on Software Engineering, 1-12."},{"key":"e_1_3_2_1_86_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-acl.678"},{"key":"e_1_3_2_1_87_1","doi-asserted-by":"publisher","DOI":"10.1109\/ASE"},{"key":"e_1_3_2_1_88_1","volume-title":"Proceedings of the 29th ACM SIGKDD Conference on Knowledge Discovery and Data Mining, 5673-5684","author":"Zheng Qinkai","year":"2023","unstructured":"Qinkai Zheng, Xiao Xia, Xu Zou, Yuxiao Dong, Shan Wang, et al. 2023. Codegeex: a pre-trained model for code generation with multilingual benchmarking on humaneval-x. In Proceedings of the 29th ACM SIGKDD Conference on Knowledge Discovery and Data Mining, 5673-5684."},{"key":"e_1_3_2_1_89_1","volume-title":"2023 IEEE\/ACM 45th International Conference on Software Engineering (ICSE). IEEE, 2324-2335","author":"Zhu Hao-Nan","year":"2023","unstructured":"Hao-Nan Zhu and Cindy Rubio-Gonz\u00e1lez. 2023. On the reproducibility of software defect datasets. In 2023 IEEE\/ACM 45th International Conference on Software Engineering (ICSE). IEEE, 2324-2335."},{"key":"e_1_3_2_1_90_1","doi-asserted-by":"publisher","DOI":"10.1145\/267580.267590"}],"event":{"name":"ISSTA '24: 33rd ACM SIGSOFT International Symposium on Software Testing and Analysis","location":"Vienna Austria","acronym":"ISSTA '24","sponsor":["SIGSOFT ACM Special Interest Group on Software Engineering","AITO"]},"container-title":["Proceedings of the 33rd ACM SIGSOFT International Symposium on Software Testing and Analysis"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3650212.3680342","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3650212.3680342","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T22:50:07Z","timestamp":1750287007000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3650212.3680342"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,9,11]]},"references-count":90,"alternative-id":["10.1145\/3650212.3680342","10.1145\/3650212"],"URL":"https:\/\/doi.org\/10.1145\/3650212.3680342","relation":{},"subject":[],"published":{"date-parts":[[2024,9,11]]},"assertion":[{"value":"2024-09-11","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}