{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T03:20:09Z","timestamp":1782876009752,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":83,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,7,12]],"date-time":"2023-07-12T00:00:00Z","timestamp":1689120000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/100000001","name":"NSF (National Science Foundation)","doi-asserted-by":"publisher","award":["CCF-1815494, CCF-210740, CCF-1845893, IIS-222194"],"award-info":[{"award-number":["CCF-1815494, CCF-210740, CCF-1845893, IIS-222194"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000185","name":"Defense Advanced Research Projects Agency","doi-asserted-by":"publisher","award":["Pacific N66001-21-C-4018"],"award-info":[{"award-number":["Pacific N66001-21-C-4018"]}],"id":[{"id":"10.13039\/100000185","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,7,12]]},"DOI":"10.1145\/3597926.3598035","type":"proceedings-article","created":{"date-parts":[[2023,7,13]],"date-time":"2023-07-13T20:12:53Z","timestamp":1689279173000},"page":"26-38","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":7,"title":["CONCORD: Clone-Aware Contrastive Learning for Source Code"],"prefix":"10.1145","author":[{"given":"Yangruibo","family":"Ding","sequence":"first","affiliation":[{"name":"Columbia University, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Saikat","family":"Chakraborty","sequence":"additional","affiliation":[{"name":"Microsoft Research, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Luca","family":"Buratti","sequence":"additional","affiliation":[{"name":"IBM Research, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Saurabh","family":"Pujar","sequence":"additional","affiliation":[{"name":"IBM Research, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Alessandro","family":"Morari","sequence":"additional","affiliation":[{"name":"IBM Research, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Gail","family":"Kaiser","sequence":"additional","affiliation":[{"name":"Columbia University, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Baishakhi","family":"Ray","sequence":"additional","affiliation":[{"name":"Columbia University, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2023,7,13]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.naacl-main.211"},{"key":"e_1_3_2_1_2_1","volume-title":"Multilingual training for Software Engineering. abs\/2112.02043","author":"Ahmed Toufique","year":"2022","unstructured":"Toufique Ahmed and Prem Devanbu . 2022. Multilingual training for Software Engineering. abs\/2112.02043 ( 2022 ). Toufique Ahmed and Prem Devanbu. 2022. Multilingual training for Software Engineering. abs\/2112.02043 (2022)."},{"key":"e_1_3_2_1_3_1","unstructured":"Miltiadis Allamanis Henry Jackson-Flux and Marc Brockschmidt. 2021. Self-Supervised Bug Detection and Repair. In NeurIPS. \t\t\t\t  Miltiadis Allamanis Henry Jackson-Flux and Marc Brockschmidt. 2021. Self-Supervised Bug Detection and Repair. In NeurIPS."},{"key":"e_1_3_2_1_4_1","volume-title":"Program Synthesis with Large Language Models. CoRR, abs\/2108.07732","author":"Austin Jacob","year":"2021","unstructured":"Jacob Austin , Augustus Odena , Maxwell Nye , Maarten Bosma , Henryk Michalewski , David Dohan , Ellen Jiang , Carrie J. Cai , Michael Terry , Quoc V. Le , and Charles Sutton . 2021. Program Synthesis with Large Language Models. CoRR, abs\/2108.07732 ( 2021 ), arXiv:2108.07732. arxiv:2108.07732 Jacob Austin, Augustus Odena, Maxwell Nye, Maarten Bosma, Henryk Michalewski, David Dohan, Ellen Jiang, Carrie J. Cai, Michael Terry, Quoc V. Le, and Charles Sutton. 2021. Program Synthesis with Large Language Models. CoRR, abs\/2108.07732 (2021), arXiv:2108.07732. arxiv:2108.07732"},{"key":"e_1_3_2_1_5_1","volume-title":"Proceedings of the Second Working Conference on Reverse Engineering (WCRE \u201995)","author":"Baker B. S.","year":"1995","unstructured":"B. S. Baker . 1995 . On Finding Duplication and Near-Duplication in Large Software Systems . In Proceedings of the Second Working Conference on Reverse Engineering (WCRE \u201995) . IEEE Computer Society, USA. 86. isbn:08 18671114 B. S. Baker. 1995. On Finding Duplication and Near-Duplication in Large Software Systems. In Proceedings of the Second Working Conference on Reverse Engineering (WCRE \u201995). IEEE Computer Society, USA. 86. isbn:0818671114"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3404835.3462840"},{"key":"e_1_3_2_1_7_1","unstructured":"Luca Buratti Saurabh Pujar Mihaela Bornea Scott McCarley Yunhui Zheng Gaetano Rossiello Alessandro Morari Jim Laredo Veronika Thost Yufan Zhuang and Giacomo Domeniconi. 2020. Exploring Software Naturalness through Neural Language Models. arxiv:2006.12641. \t\t\t\t  Luca Buratti Saurabh Pujar Mihaela Bornea Scott McCarley Yunhui Zheng Gaetano Rossiello Alessandro Morari Jim Laredo Veronika Thost Yufan Zhuang and Giacomo Domeniconi. 2020. Exploring Software Naturalness through Neural Language Models. arxiv:2006.12641."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/3377816.3381720"},{"key":"e_1_3_2_1_9_1","volume-title":"Do programmers prefer predictable expressions in code? Cognitive science, 44, 12","author":"Casalnuovo Casey","year":"2020","unstructured":"Casey Casalnuovo , Kevin Lee , Hulin Wang , Prem Devanbu , and Emily Morgan . 2020. Do programmers prefer predictable expressions in code? Cognitive science, 44, 12 ( 2020 ), e12921. Casey Casalnuovo, Kevin Lee, Hulin Wang, Prem Devanbu, and Emily Morgan. 2020. Do programmers prefer predictable expressions in code? Cognitive science, 44, 12 (2020), e12921."},{"key":"e_1_3_2_1_10_1","volume-title":"Proceedings of the 42nd Annual Meeting of the Cognitive Science Society.","author":"Casalnuovo Casey","year":"2020","unstructured":"Casey Casalnuovo , E Morgan , and P Devanbu . 2020 . Does surprisal predict code comprehension difficulty . In Proceedings of the 42nd Annual Meeting of the Cognitive Science Society. Casey Casalnuovo, E Morgan, and P Devanbu. 2020. Does surprisal predict code comprehension difficulty. In Proceedings of the 42nd Annual Meeting of the Cognitive Science Society."},{"key":"e_1_3_2_1_11_1","volume-title":"2022 The ACM Joint European Software Engineering Conference and Symposium on the Foundations of Software Engineering (ESEC\/FSE).","author":"Chakraborty Saikat","year":"2022","unstructured":"Saikat Chakraborty , Toufique Ahmed , Yangruibo Ding , Premkumar Devanbu , and Baishakhi Ray . 2022 . NatGen: Generative pre-training by\" Naturalizing\" source code . In 2022 The ACM Joint European Software Engineering Conference and Symposium on the Foundations of Software Engineering (ESEC\/FSE). Saikat Chakraborty, Toufique Ahmed, Yangruibo Ding, Premkumar Devanbu, and Baishakhi Ray. 2022. NatGen: Generative pre-training by\" Naturalizing\" source code. In 2022 The ACM Joint European Software Engineering Conference and Symposium on the Foundations of Software Engineering (ESEC\/FSE)."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/TSE.2021.3087402"},{"key":"e_1_3_2_1_13_1","volume-title":"VarCLR: Variable Semantic Representation Pre-training via Contrastive Learning. CoRR, abs\/2112.02650","author":"Chen Qibin","year":"2021","unstructured":"Qibin Chen , Jeremy Lacomis , Edward J. Schwartz , Graham Neubig , Bogdan Vasilescu , and Claire Le Goues . 2021. VarCLR: Variable Semantic Representation Pre-training via Contrastive Learning. CoRR, abs\/2112.02650 ( 2021 ), arXiv:2112.02650. arxiv:2112.02650 Qibin Chen, Jeremy Lacomis, Edward J. Schwartz, Graham Neubig, Bogdan Vasilescu, and Claire Le Goues. 2021. VarCLR: Variable Semantic Representation Pre-training via Contrastive Learning. CoRR, abs\/2112.02650 (2021), arXiv:2112.02650. arxiv:2112.02650"},{"key":"e_1_3_2_1_14_1","volume-title":"Proceedings of the 37th International Conference on Machine Learning. PMLR, 1597\u20131607","author":"Chen Ting","year":"2020","unstructured":"Ting Chen , Simon Kornblith , Mohammad Norouzi , and Geoffrey Hinton . 2020 . A Simple Framework for Contrastive Learning of Visual Representations . In Proceedings of the 37th International Conference on Machine Learning. PMLR, 1597\u20131607 . https:\/\/proceedings.mlr.press\/v119\/chen20j.html Ting Chen, Simon Kornblith, Mohammad Norouzi, and Geoffrey Hinton. 2020. A Simple Framework for Contrastive Learning of Visual Representations. In Proceedings of the 37th International Conference on Machine Learning. PMLR, 1597\u20131607. https:\/\/proceedings.mlr.press\/v119\/chen20j.html"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2005.202"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/502059.502042"},{"key":"e_1_3_2_1_17_1","volume-title":"Manning","author":"Clark Kevin","year":"2020","unstructured":"Kevin Clark , Minh-Thang Luong , Quoc V. Le , and Christopher D . Manning . 2020 . ELECTRA : Pre-training Text Encoders as Discriminators Rather Than Generators. In ICLR. https:\/\/openreview.net\/pdf?id=r1xMH1BtvB Kevin Clark, Minh-Thang Luong, Quoc V. Le, and Christopher D. Manning. 2020. ELECTRA: Pre-training Text Encoders as Discriminators Rather Than Generators. In ICLR. https:\/\/openreview.net\/pdf?id=r1xMH1BtvB"},{"key":"e_1_3_2_1_18_1","unstructured":"CVE-2022-23559. 2022. https:\/\/nvd.nist.gov\/vuln\/detail\/CVE-2022-23559 \t\t\t\t  CVE-2022-23559. 2022. https:\/\/nvd.nist.gov\/vuln\/detail\/CVE-2022-23559"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N19-1423"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.436"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3324884.3416587"},{"key":"e_1_3_2_1_22_1","volume-title":"Proceedings of the IEEE International Conference on Software Maintenance (ICSM \u201999)","author":"Ducasse St\u00e9phane","year":"1999","unstructured":"St\u00e9phane Ducasse , Matthias Rieger , and Serge Demeyer . 1999 . A Language Independent Approach for Detecting Duplicated Code . In Proceedings of the IEEE International Conference on Software Maintenance (ICSM \u201999) . IEEE Computer Society, USA. 109. isbn:0769500161 St\u00e9phane Ducasse, Matthias Rieger, and Serge Demeyer. 1999. A Language Independent Approach for Detecting Duplicated Code. In Proceedings of the IEEE International Conference on Software Maintenance (ICSM \u201999). IEEE Computer Society, USA. 109. isbn:0769500161"},{"key":"e_1_3_2_1_23_1","volume-title":"PyTorch: An Imperative Style","author":"Paszke Adam","unstructured":"Adam Paszke . 2019. PyTorch: An Imperative Style , High-Performance Deep Learning Library . Adam Paszke. 2019. PyTorch: An Imperative Style, High-Performance Deep Learning Library."},{"key":"e_1_3_2_1_24_1","volume-title":"Multi-lingual Evaluation of Code Generation Models. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=Bo7eeXm6An8","author":"Athiwaratkun Ben","year":"2023","unstructured":"Ben Athiwaratkun . 2023 . Multi-lingual Evaluation of Code Generation Models. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=Bo7eeXm6An8 Ben Athiwaratkun. 2023. Multi-lingual Evaluation of Code Generation Models. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=Bo7eeXm6An8"},{"key":"e_1_3_2_1_25_1","unstructured":"Mart\u00edn Abadi. 2015. TensorFlow: Large-Scale Machine Learning on Heterogeneous Systems. https:\/\/www.tensorflow.org\/ Software available from tensorflow.org \t\t\t\t  Mart\u00edn Abadi. 2015. TensorFlow: Large-Scale Machine Learning on Heterogeneous Systems. https:\/\/www.tensorflow.org\/ Software available from tensorflow.org"},{"key":"e_1_3_2_1_26_1","volume-title":"Evaluating Large Language Models Trained on Code. CoRR, abs\/2107.03374","author":"Chen Mark","year":"2021","unstructured":"Mark Chen . 2021. Evaluating Large Language Models Trained on Code. CoRR, abs\/2107.03374 ( 2021 ), arXiv:2107.03374. arxiv:2107.03374 Mark Chen. 2021. Evaluating Large Language Models Trained on Code. CoRR, abs\/2107.03374 (2021), arXiv:2107.03374. arxiv:2107.03374"},{"key":"e_1_3_2_1_27_1","volume-title":"Language Models are Few-Shot Learners. CoRR, abs\/2005.14165","author":"Brown Tom B.","year":"2020","unstructured":"Tom B. Brown . 2020. Language Models are Few-Shot Learners. CoRR, abs\/2005.14165 ( 2020 ), arXiv:2005.14165. arxiv:2005.14165 Tom B. Brown. 2020. Language Models are Few-Shot Learners. CoRR, abs\/2005.14165 (2020), arXiv:2005.14165. arxiv:2005.14165"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-demos.6"},{"key":"e_1_3_2_1_29_1","volume-title":"Competition-Level Code Generation with AlphaCode. ArXiv, abs\/2203.07814","author":"Yujia Li.","year":"2022","unstructured":"Yujia Li. 2022. Competition-Level Code Generation with AlphaCode. ArXiv, abs\/2203.07814 ( 2022 ). Yujia Li. 2022. Competition-Level Code Generation with AlphaCode. ArXiv, abs\/2203.07814 (2022)."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.findings-emnlp.139"},{"key":"e_1_3_2_1_31_1","unstructured":"Tianyu Gao Xingcheng Yao and Danqi Chen. 2021. SimCSE: Simple Contrastive Learning of Sentence Embeddings. In Empirical Methods in Natural Language Processing (EMNLP). \t\t\t\t  Tianyu Gao Xingcheng Yao and Danqi Chen. 2021. SimCSE: Simple Contrastive Learning of Sentence Embeddings. In Empirical Methods in Natural Language Processing (EMNLP)."},{"key":"e_1_3_2_1_32_1","unstructured":"GitHub. 2021. GitHub Copilot: Your AI Pair Programmer. https:\/\/copilot.github.com\/ \t\t\t\t  GitHub. 2021. GitHub Copilot: Your AI Pair Programmer. https:\/\/copilot.github.com\/"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3368089.3409714"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3106237.3106264"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/3196398.3196432"},{"key":"#cr-split#-e_1_3_2_1_36_1.1","unstructured":"Daya Guo Shuai Lu Nan Duan Yanlin Wang Ming Zhou and Jian Yin. 2022. UniXcoder: Unified Cross-Modal Pre-training for Code Representation. https:\/\/doi.org\/10.48550\/ARXIV.2203.03850 10.48550\/ARXIV.2203.03850"},{"key":"#cr-split#-e_1_3_2_1_36_1.2","doi-asserted-by":"crossref","unstructured":"Daya Guo Shuai Lu Nan Duan Yanlin Wang Ming Zhou and Jian Yin. 2022. UniXcoder: Unified Cross-Modal Pre-training for Code Representation. https:\/\/doi.org\/10.48550\/ARXIV.2203.03850","DOI":"10.18653\/v1\/2022.acl-long.499"},{"key":"e_1_3_2_1_37_1","volume-title":"International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=jLoC4ez43PZ","author":"Guo Daya","year":"2021","unstructured":"Daya Guo , Shuo Ren , Shuai Lu , Zhangyin Feng , Duyu Tang , Shujie LIU , Long Zhou , Nan Duan , Alexey Svyatkovskiy , Shengyu Fu , Michele Tufano , Shao Kun Deng , Colin Clement , Dawn Drain , Neel Sundaresan , Jian Yin , Daxin Jiang , and Ming Zhou . 2021 . GraphCode\\BERT Pre-training Code Representations with Data Flow . In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=jLoC4ez43PZ Daya Guo, Shuo Ren, Shuai Lu, Zhangyin Feng, Duyu Tang, Shujie LIU, Long Zhou, Nan Duan, Alexey Svyatkovskiy, Shengyu Fu, Michele Tufano, Shao Kun Deng, Colin Clement, Dawn Drain, Neel Sundaresan, Jian Yin, Daxin Jiang, and Ming Zhou. 2021. GraphCode\\BERT Pre-training Code Representations with Data Flow. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=jLoC4ez43PZ"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.5555\/2337223.2337322"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1093\/biomet\/28.3-4.321"},{"key":"e_1_3_2_1_40_1","volume-title":"CodeSearchNet Challenge: Evaluating the State of Semantic Code Search. CoRR, abs\/1909.09436","author":"Husain Hamel","year":"2019","unstructured":"Hamel Husain , Ho-Hsiang Wu , Tiferet Gazit , Miltiadis Allamanis , and Marc Brockschmidt . 2019. CodeSearchNet Challenge: Evaluating the State of Semantic Code Search. CoRR, abs\/1909.09436 ( 2019 ), arXiv:1909.09436. arxiv:1909.09436 Hamel Husain, Ho-Hsiang Wu, Tiferet Gazit, Miltiadis Allamanis, and Marc Brockschmidt. 2019. CodeSearchNet Challenge: Evaluating the State of Semantic Code Search. CoRR, abs\/1909.09436 (2019), arXiv:1909.09436. arxiv:1909.09436"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"crossref","unstructured":"Paras Jain Ajay Jain Tianjun Zhang Pieter Abbeel Joseph E. Gonzalez and Ion Stoica. 2020. Contrastive Code Representation Learning. arXiv preprint. \t\t\t\t  Paras Jain Ajay Jain Tianjun Zhang Pieter Abbeel Joseph E. Gonzalez and Ion Stoica. 2020. Contrastive Code Representation Learning. arXiv preprint.","DOI":"10.18653\/v1\/2021.emnlp-main.482"},{"key":"e_1_3_2_1_42_1","volume-title":"TreeBERT: A Tree-Based Pre-Trained Model for Programming Language. ArXiv, abs\/2105.12485","author":"Jiang Xue","year":"2021","unstructured":"Xue Jiang , Zhuoran Zheng , Chen Lyu , Liang Li , and Lei Lyu . 2021. TreeBERT: A Tree-Based Pre-Trained Model for Programming Language. ArXiv, abs\/2105.12485 ( 2021 ). Xue Jiang, Zhuoran Zheng, Chen Lyu, Liang Li, and Lei Lyu. 2021. TreeBERT: A Tree-Based Pre-Trained Model for Programming Language. ArXiv, abs\/2105.12485 (2021)."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICSE.2009.5070547"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/2610384.2628055"},{"key":"e_1_3_2_1_45_1","volume-title":"ICML","author":"Kanade Aditya","year":"2020","unstructured":"Aditya Kanade , Petros Maniatis , Gogul Balakrishnan , and Kensen Shi . 2020 . Learning and evaluating contextual embedding of source code . In ICML 2020. Aditya Kanade, Petros Maniatis, Gogul Balakrishnan, and Kensen Shi. 2020. Learning and evaluating contextual embedding of source code. In ICML 2020."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/3377811.3380342"},{"key":"e_1_3_2_1_47_1","volume-title":"Matthew Jin, Xiaoyu Liu, Xin Shi, Colin B. Clement, and Neel Sundaresan.","author":"Kharkar Anant","year":"2022","unstructured":"Anant Kharkar , Roshanak Zilouchian Moghaddam , Matthew Jin, Xiaoyu Liu, Xin Shi, Colin B. Clement, and Neel Sundaresan. 2022 . Learning to Reduce False Positives in Analytic Bug Detectors . Anant Kharkar, Roshanak Zilouchian Moghaddam, Matthew Jin, Xiaoyu Liu, Xin Shi, Colin B. Clement, and Neel Sundaresan. 2022. Learning to Reduce False Positives in Analytic Bug Detectors."},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/SP.2017.62"},{"key":"e_1_3_2_1_49_1","volume-title":"Kingma and Jimmy Ba","author":"Diederik","year":"2015","unstructured":"Diederik P. Kingma and Jimmy Ba . 2015 . Adam : A Method for Stochastic Optimization. CoRR , abs\/1412.6980 (2015). Diederik P. Kingma and Jimmy Ba. 2015. Adam: A Method for Stochastic Optimization. CoRR, abs\/1412.6980 (2015)."},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1093\/comjnl\/27.2.97"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-2012"},{"key":"e_1_3_2_1_52_1","volume-title":"Ernst","author":"Li Jingyue","year":"2012","unstructured":"Jingyue Li and Michael D . Ernst . 2012 . CBCD : Cloned Buggy Code Detector. In Proceedings of the 34th International Conference on Software Engineering (ICSE \u201912). IEEE Press , 310\u2013320. isbn:9781467310673 Jingyue Li and Michael D. Ernst. 2012. CBCD: Cloned Buggy Code Detector. In Proceedings of the 34th International Conference on Software Engineering (ICSE \u201912). IEEE Press, 310\u2013320. isbn:9781467310673"},{"key":"e_1_3_2_1_53_1","volume-title":"CP-Miner: A Tool for Finding Copy-paste and Related Bugs in Operating System Code. In 6th Symposium on Operating Systems Design & Implementation (OSDI 04)","author":"Li Zhenmin","year":"2004","unstructured":"Zhenmin Li , Shan Lu , Suvda Myagmar , and Yuanyuan Zhou . 2004 . CP-Miner: A Tool for Finding Copy-paste and Related Bugs in Operating System Code. In 6th Symposium on Operating Systems Design & Implementation (OSDI 04) . USENIX Association, San Francisco, CA. https:\/\/www.usenix.org\/conference\/osdi-04\/cp-miner-tool-finding-copy-paste-and-related-bugs-operating-system-code Zhenmin Li, Shan Lu, Suvda Myagmar, and Yuanyuan Zhou. 2004. CP-Miner: A Tool for Finding Copy-paste and Related Bugs in Operating System Code. In 6th Symposium on Operating Systems Design & Implementation (OSDI 04). USENIX Association, San Francisco, CA. https:\/\/www.usenix.org\/conference\/osdi-04\/cp-miner-tool-finding-copy-paste-and-related-bugs-operating-system-code"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1145\/2991079.2991102"},{"key":"e_1_3_2_1_55_1","volume-title":"RoBERTa: A Robustly Optimized BERT Pretraining Approach. CoRR, abs\/1907.11692","author":"Liu Yinhan","year":"2019","unstructured":"Yinhan Liu , Myle Ott , Naman Goyal , Jingfei Du , Mandar Joshi , Danqi Chen , Omer Levy , Mike Lewis , Luke Zettlemoyer , and Veselin Stoyanov . 2019. RoBERTa: A Robustly Optimized BERT Pretraining Approach. CoRR, abs\/1907.11692 ( 2019 ), arXiv:1907.11692. arxiv:1907.11692 Yinhan Liu, Myle Ott, Naman Goyal, Jingfei Du, Mandar Joshi, Danqi Chen, Omer Levy, Mike Lewis, Luke Zettlemoyer, and Veselin Stoyanov. 2019. RoBERTa: A Robustly Optimized BERT Pretraining Approach. CoRR, abs\/1907.11692 (2019), arXiv:1907.11692. arxiv:1907.11692"},{"key":"#cr-split#-e_1_3_2_1_56_1.1","unstructured":"Shuai Lu Nan Duan Hojae Han Daya Guo Seung-won Hwang and Alexey Svyatkovskiy. 2022. ReACC: A Retrieval-Augmented Code Completion Framework. https:\/\/doi.org\/10.48550\/ARXIV.2203.07722 10.48550\/ARXIV.2203.07722"},{"key":"#cr-split#-e_1_3_2_1_56_1.2","unstructured":"Shuai Lu Nan Duan Hojae Han Daya Guo Seung-won Hwang and Alexey Svyatkovskiy. 2022. ReACC: A Retrieval-Augmented Code Completion Framework. https:\/\/doi.org\/10.48550\/ARXIV.2203.07722"},{"key":"e_1_3_2_1_57_1","volume-title":"Shengyu Fu, and Shujie Liu.","author":"Lu Shuai","year":"2021","unstructured":"Shuai Lu , Daya Guo , Shuo Ren , Junjie Huang , Alexey Svyatkovskiy , Ambrosio Blanco , Colin B. Clement , Dawn Drain , Daxin Jiang , Duyu Tang , Ge Li , Lidong Zhou , Linjun Shou , Long Zhou , Michele Tufano , Ming Gong , Ming Zhou , Nan Duan , Neel Sundaresan , Shao Kun Deng , Shengyu Fu, and Shujie Liu. 2021 . CodeXGLUE: A Machine Learning Benchmark Dataset for Code Understanding and Generation. CoRR , abs\/2102.04664 (2021). Shuai Lu, Daya Guo, Shuo Ren, Junjie Huang, Alexey Svyatkovskiy, Ambrosio Blanco, Colin B. Clement, Dawn Drain, Daxin Jiang, Duyu Tang, Ge Li, Lidong Zhou, Linjun Shou, Long Zhou, Michele Tufano, Ming Gong, Ming Zhou, Nan Duan, Neel Sundaresan, Shao Kun Deng, Shengyu Fu, and Shujie Liu. 2021. CodeXGLUE: A Machine Learning Benchmark Dataset for Code Understanding and Generation. CoRR, abs\/2102.04664 (2021)."},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1145\/3360578"},{"key":"e_1_3_2_1_59_1","unstructured":"MITRE. 2020. Common Weakness Enumeration.  https:\/\/cwe.mitre.org\/data\/index.html \t\t\t\t  MITRE. 2020. Common Weakness Enumeration.  https:\/\/cwe.mitre.org\/data\/index.html"},{"key":"e_1_3_2_1_60_1","unstructured":"MITRE. 2022. Common Vulnerabilities and Exposures.  https:\/\/cve.mitre.org \t\t\t\t  MITRE. 2022. Common Vulnerabilities and Exposures.  https:\/\/cve.mitre.org"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v30i1.10139"},{"key":"e_1_3_2_1_62_1","volume-title":"SPT-Code: Sequence-to-Sequence Pre-Training for Learning Source Code Representations. CoRR, abs\/2201.01549","author":"Niu Changan","year":"2022","unstructured":"Changan Niu , Chuanyi Li , Vincent Ng , Jidong Ge , Liguo Huang , and Bin Luo . 2022. SPT-Code: Sequence-to-Sequence Pre-Training for Learning Source Code Representations. CoRR, abs\/2201.01549 ( 2022 ), arXiv:2201.01549. arxiv:2201.01549 Changan Niu, Chuanyi Li, Vincent Ng, Jidong Ge, Liguo Huang, and Bin Luo. 2022. SPT-Code: Sequence-to-Sequence Pre-Training for Learning Source Code Representations. CoRR, abs\/2201.01549 (2022), arXiv:2201.01549. arxiv:2201.01549"},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1145\/3276517"},{"key":"e_1_3_2_1_64_1","volume-title":"Project CodeNet: A Large-Scale AI for Code Dataset for Learning a Diversity of Coding Tasks. CoRR, abs\/2105.12655","author":"Puri Ruchir","year":"2021","unstructured":"Ruchir Puri , David S. Kung , Geert Janssen , Wei Zhang , Giacomo Domeniconi , Vladimir Zolotov , Julian Dolby , Jie Chen , Mihir R. Choudhury , Lindsey Decker , Veronika Thost , Luca Buratti , Saurabh Pujar , and Ulrich Finkler . 2021. Project CodeNet: A Large-Scale AI for Code Dataset for Learning a Diversity of Coding Tasks. CoRR, abs\/2105.12655 ( 2021 ), arXiv:2105.12655. arxiv:2105.12655 Ruchir Puri, David S. Kung, Geert Janssen, Wei Zhang, Giacomo Domeniconi, Vladimir Zolotov, Julian Dolby, Jie Chen, Mihir R. Choudhury, Lindsey Decker, Veronika Thost, Luca Buratti, Saurabh Pujar, and Ulrich Finkler. 2021. Project CodeNet: A Large-Scale AI for Code Dataset for Learning a Diversity of Coding Tasks. CoRR, abs\/2105.12655 (2021), arXiv:2105.12655. arxiv:2105.12655"},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.1145\/2884781.2884848"},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1109\/ASE.2013.6693095"},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICSM.2009.5306301"},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-981-16-1927-4_5"},{"key":"e_1_3_2_1_69_1","doi-asserted-by":"publisher","DOI":"10.1145\/2884781.2884877"},{"key":"#cr-split#-e_1_3_2_1_70_1.1","unstructured":"Victor Sanh Lysandre Debut Julien Chaumond and Thomas Wolf. 2019. DistilBERT a distilled version of BERT: smaller faster cheaper and lighter. https:\/\/doi.org\/10.48550\/ARXIV.1910.01108 10.48550\/ARXIV.1910.01108"},{"key":"#cr-split#-e_1_3_2_1_70_1.2","unstructured":"Victor Sanh Lysandre Debut Julien Chaumond and Thomas Wolf. 2019. DistilBERT a distilled version of BERT: smaller faster cheaper and lighter. https:\/\/doi.org\/10.48550\/ARXIV.1910.01108"},{"key":"e_1_3_2_1_71_1","doi-asserted-by":"publisher","DOI":"10.5555\/2627435.2670313"},{"key":"e_1_3_2_1_72_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICSME.2014.77"},{"key":"e_1_3_2_1_73_1","unstructured":"Tree-sitter. 2022. Tree-sitter.  https:\/\/github.com\/tree-sitter\/tree-sitter \t\t\t\t  Tree-sitter. 2022. Tree-sitter.  https:\/\/github.com\/tree-sitter\/tree-sitter"},{"key":"e_1_3_2_1_74_1","doi-asserted-by":"publisher","DOI":"10.1109\/WCRE.2011.12"},{"key":"#cr-split#-e_1_3_2_1_75_1.1","unstructured":"Xin Wang Yasheng Wang Fei Mi Pingyi Zhou Yao Wan Xiao Liu Li Li Hao Wu Jin Liu and Xin Jiang. 2021. SynCoBERT: Syntax-Guided Multi-Modal Contrastive Pre-Training for Code Representation. https:\/\/doi.org\/10.48550\/ARXIV.2108.04556 10.48550\/ARXIV.2108.04556"},{"key":"#cr-split#-e_1_3_2_1_75_1.2","unstructured":"Xin Wang Yasheng Wang Fei Mi Pingyi Zhou Yao Wan Xiao Liu Li Li Hao Wu Jin Liu and Xin Jiang. 2021. SynCoBERT: Syntax-Guided Multi-Modal Contrastive Pre-Training for Code Representation. https:\/\/doi.org\/10.48550\/ARXIV.2108.04556"},{"key":"e_1_3_2_1_76_1","volume-title":"Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing, EMNLP 2021.","author":"Wang Yue","unstructured":"Yue Wang , Weishi Wang , Shafiq Joty , and Steven C.H. Hoi . 2021. CodeT5: Identifier-aware Unified Pre-trained Encoder-Decoder Models for Code Understanding and Generation . In Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing, EMNLP 2021. Yue Wang, Weishi Wang, Shafiq Joty, and Steven C.H. Hoi. 2021. CodeT5: Identifier-aware Unified Pre-trained Encoder-Decoder Models for Code Understanding and Generation. In Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing, EMNLP 2021."},{"key":"e_1_3_2_1_77_1","unstructured":"Frank F Xu Uri Alon Graham Neubig and Vincent J Hellendoorn. 2022. A Systematic Evaluation of Large Language Models of Code. arXiv preprint arXiv:2202.13169. \t\t\t\t  Frank F Xu Uri Alon Graham Neubig and Vincent J Hellendoorn. 2022. A Systematic Evaluation of Large Language Models of Code. arXiv preprint arXiv:2202.13169."},{"key":"e_1_3_2_1_78_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICSE-SEIP52600.2021.00020"},{"key":"e_1_3_2_1_79_1","volume-title":"Devign: Effective vulnerability identification by learning comprehensive program semantics via graph neural networks. In Advances in Neural Information Processing Systems. 10197\u201310207.","author":"Zhou Yaqin","year":"2019","unstructured":"Yaqin Zhou , Shangqing Liu , Jingkai Siow , Xiaoning Du , and Yang Liu . 2019 . Devign: Effective vulnerability identification by learning comprehensive program semantics via graph neural networks. In Advances in Neural Information Processing Systems. 10197\u201310207. Yaqin Zhou, Shangqing Liu, Jingkai Siow, Xiaoning Du, and Yang Liu. 2019. Devign: Effective vulnerability identification by learning comprehensive program semantics via graph neural networks. In Advances in Neural Information Processing Systems. 10197\u201310207."}],"event":{"name":"ISSTA '23: 32nd ACM SIGSOFT International Symposium on Software Testing and Analysis","location":"Seattle WA USA","acronym":"ISSTA '23","sponsor":["SIGSOFT ACM Special Interest Group on Software Engineering","AITO"]},"container-title":["Proceedings of the 32nd ACM SIGSOFT International Symposium on Software Testing and Analysis"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3597926.3598035","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3597926.3598035","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T17:48:41Z","timestamp":1750182521000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3597926.3598035"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,7,12]]},"references-count":83,"alternative-id":["10.1145\/3597926.3598035","10.1145\/3597926"],"URL":"https:\/\/doi.org\/10.1145\/3597926.3598035","relation":{},"subject":[],"published":{"date-parts":[[2023,7,12]]},"assertion":[{"value":"2023-07-13","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}