{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T16:41:24Z","timestamp":1783096884486,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":26,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,4,15]],"date-time":"2024-04-15T00:00:00Z","timestamp":1713139200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,4,15]]},"DOI":"10.1145\/3643916.3645030","type":"proceedings-article","created":{"date-parts":[[2024,6,13]],"date-time":"2024-06-13T12:40:20Z","timestamp":1718282420000},"page":"161-165","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":19,"title":["Investigating the Efficacy of Large Language Models for Code Clone Detection"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2433-4182","authenticated-orcid":false,"given":"Mohamad","family":"Khajezade","sequence":"first","affiliation":[{"name":"University of British Columbia, Kelowna, British Columbia, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7895-2023","authenticated-orcid":false,"given":"Jie JW","family":"Wu","sequence":"additional","affiliation":[{"name":"University of British Columbia, Kelowna, British Columbia, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4505-6257","authenticated-orcid":false,"given":"Fatemeh Hendijani","family":"Fard","sequence":"additional","affiliation":[{"name":"University of British Columbia, Kelowna, British Columbia, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0062-8418","authenticated-orcid":false,"given":"Gema","family":"Rodriguez-Perez","sequence":"additional","affiliation":[{"name":"University of British Columbia, Kelowna, British Columbia, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8464-8650","authenticated-orcid":false,"given":"Mohamed Sami","family":"Shehata","sequence":"additional","affiliation":[{"name":"University of British Columbia, Kelowna, British Columbia, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,6,13]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2019.2918202"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jss.2021.111141"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"crossref","unstructured":"Chenyao Liu Zeqi Lin Jian-Guang Lou Lijie Wen and Dongmei Zhang \"Can neural clone detection generalize to unseen functionalitiesf \" in 2021 36th IEEE\/ACM International Conference on Automated Software Engineering (ASE) 2021 pp. 617--629.","DOI":"10.1109\/ASE51524.2021.9678907"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"crossref","unstructured":"Tim Sonnekalb Bernd Gruner Clemens-Alexander Brust and Patrick M\u00e4der \"Generalizability of code clone detection on codebert \" in Proceedings of the 37th IEEE\/ACM International Conference on Automated Software Engineering 2022 pp. 1--3.","DOI":"10.1145\/3551349.3561165"},{"key":"e_1_3_2_1_5_1","first-page":"413","volume-title":"Contrastive cross-language code clone detection,\" in Proceedings of the 30th IEEE\/ACM International Conference on Program Comprehension","author":"Tao Chenning","year":"2022","unstructured":"Chenning Tao, Qi Zhan, Xing Hu, and Xin Xia, \"C4: Contrastive cross-language code clone detection,\" in Proceedings of the 30th IEEE\/ACM International Conference on Program Comprehension, 2022, pp. 413--424."},{"key":"e_1_3_2_1_6_1","first-page":"1877","volume-title":"Language models are few-shot learners,\" Advances in neural information processing systems","author":"Brown Tom","year":"2020","unstructured":"Tom Brown, Benjamin Mann, Nick Ryder, Melanie Subbiah, Jared D Kaplan, Prafulla Dhariwal, Arvind Neelakantan, Pranav Shyam, Girish Sastry, Amanda Askell, et al., \"Language models are few-shot learners,\" Advances in neural information processing systems, vol. 33, pp. 1877--1901, 2020."},{"key":"e_1_3_2_1_7_1","unstructured":"Bonan Min Hayley Ross Elior Sulem Amir Pouran Ben Veyseh Thien Huu Nguyen Oscar Sainz Eneko Agirre Ilana Heinz and Dan Roth \"Recent advances in natural language processing via large pre-trained language models: A survey \" arXiv preprint arXiv:2111.01243 2021."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.3390\/app122111220"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"crossref","unstructured":"Frank F Xu Uri Alon Graham Neubig and Vincent Josua Hellendoorn \"A systematic evaluation of large language models of code \" in Proceedings of the 6th ACM SIGPLAN International Symposium on Machine Programming 2022 pp. 1--10.","DOI":"10.1145\/3520312.3534862"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.14778\/3551793.3551841"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"crossref","unstructured":"Noor Nashid Mifta Sintaha and Ali Mesbah \"Retrieval-based prompt selection for code-related few-shot learning \" in Proceedings of the 45th International Conference on Software Engineering (ICSE'23) 2023.","DOI":"10.1109\/ICSE48619.2023.00205"},{"key":"e_1_3_2_1_12_1","volume-title":"pre-trained language models on code,\" arXiv preprint arXiv:2206.01335","author":"Barei\u00df Patrick","year":"2022","unstructured":"Patrick Barei\u00df, Beatriz Souza, Marcelo d'Amorim, and Michael Pradel, \"Code generation tools (almost) for free? a study of few-shot, pre-trained language models on code,\" arXiv preprint arXiv:2206.01335, 2022."},{"key":"e_1_3_2_1_13_1","unstructured":"Palash R Roy Ajmain I Alam Farouq Al-omari Banani Roy Chanchal K Roy and Kevin A Schneider \"Unveiling the potential of large language models in generating semantic and cross-language clones \" arXiv preprint arXiv:2309.06424 2023."},{"key":"e_1_3_2_1_14_1","volume-title":"a survey,\" arXiv preprint arXiv:2308.01191","author":"Dou Shihan","year":"2023","unstructured":"Shihan Dou, Junjie Shan, Haoxiang Jia, Wenhao Deng, Zhiheng Xi, Wei He, Yueming Wu, Tao Gui, Yang Liu, and Xuanjing Huang, \"Towards understanding the capability of large language models on code clone detection: a survey,\" arXiv preprint arXiv:2308.01191, 2023."},{"key":"e_1_3_2_1_15_1","volume-title":"Project codenet: A large-scale ai for code dataset for learning a diversity of coding tasks,\" arXiv preprint arXiv:2105.12655","author":"Puri Ruchir","year":"2021","unstructured":"Ruchir Puri, David S Kung, Geert Janssen, Wei Zhang, Giacomo Domeniconi, Vladmir Zolotov, Julian Dolby, Jie Chen, Mihir Choudhury, Lindsey Decker, et al., \"Project codenet: A large-scale ai for code dataset for learning a diversity of coding tasks,\" arXiv preprint arXiv:2105.12655, vol. 1035, 2021."},{"key":"e_1_3_2_1_16_1","first-page":"261","volume-title":"f-score, and nlp evaluation,\" in Proceedings of the Tenth International Conference on Language Resources and Evaluation (LREC'16)","author":"Derczynski Leon","year":"2016","unstructured":"Leon Derczynski, \"Complementarity, f-score, and nlp evaluation,\" in Proceedings of the Tenth International Conference on Language Resources and Evaluation (LREC'16), 2016, pp. 261--266."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/MS.2016.147"},{"key":"e_1_3_2_1_18_1","first-page":"1040","volume-title":"IEEE","author":"Moreno-Le\u00f3n Jes\u00fas","year":"2016","unstructured":"Jes\u00fas Moreno-Le\u00f3n, Gregorio Robles, and Marcos Rom\u00e1n-Gonz\u00e1lez, \"Comparing computational thinking development assessment scores with software complexity metrics,\" in 2016 IEEE global engineering education conference (EDUCON). IEEE, 2016, pp. 1040--1045."},{"key":"e_1_3_2_1_19_1","first-page":"23","article-title":"Cluster analysis to estimate the difficulty of programming problems,\" in Proceedings of the 3rd International Conference on Applications","author":"Intisar Chowdhury Md","year":"2018","unstructured":"Chowdhury Md Intisar and Yutaka Watanobe, \"Cluster analysis to estimate the difficulty of programming problems,\" in Proceedings of the 3rd International Conference on Applications in Information Technology, 2018, pp. 23--28.","journal-title":"Information Technology"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.infsof.2022.107130"},{"key":"e_1_3_2_1_21_1","first-page":"1536","volume-title":"Online","author":"Feng Zhangyin","year":"2020","unstructured":"Zhangyin Feng, Daya Guo, Duyu Tang, Nan Duan, Xiaocheng Feng, Ming Gong, Linjun Shou, Bing Qin, Ting Liu, Daxin Jiang, and Ming Zhou, \"CodeBERT: A pre-trained model for programming and natural languages,\" in Findings of the Association for Computational Linguistics: EMNLP 2020, Online, Nov. 2020, pp. 1536--1547, Association for Computational Linguistics."},{"key":"e_1_3_2_1_22_1","volume-title":"A robustly optimized bert pretraining approach,\" arXiv preprint arXiv:1907.11692","author":"Liu Yinhan","year":"2019","unstructured":"Yinhan Liu, Myle Ott, Naman Goyal, Jingfei Du, Mandar Joshi, Danqi Chen, Omer Levy, Mike Lewis, Luke Zettlemoyer, and Veselin Stoyanov, \"Roberta: A robustly optimized bert pretraining approach,\" arXiv preprint arXiv:1907.11692, 2019."},{"key":"e_1_3_2_1_23_1","volume-title":"Syntax-guided multi-modal contrastive pre-training for code representation,\" arXiv preprint arXiv:2108.04556","author":"Wang Xin","year":"2021","unstructured":"Xin Wang, Yasheng Wang, Fei Mi, Pingyi Zhou, Yao Wan, Xiao Liu, Li Li, Hao Wu, Jin Liu, and Xin Jiang, \"Syncobert: Syntax-guided multi-modal contrastive pre-training for code representation,\" arXiv preprint arXiv:2108.04556, 2021."},{"key":"e_1_3_2_1_24_1","volume-title":"Codexglue: A machine learning benchmark dataset for code understanding and generation,\" arXiv preprint arXiv:2102.04664","author":"Lu Shuai","year":"2021","unstructured":"Shuai Lu, Daya Guo, Shuo Ren, Junjie Huang, Alexey Svyatkovskiy, Ambrosio Blanco, Colin Clement, Dawn Drain, Daxin Jiang, Duyu Tang, et al., \"Codexglue: A machine learning benchmark dataset for code understanding and generation,\" arXiv preprint arXiv:2102.04664, 2021."},{"key":"e_1_3_2_1_25_1","unstructured":"Daya Guo Shuo Ren Shuai Lu Zhangyin Feng Duyu Tang Shujie LIU Long Zhou Nan Duan Alexey Svyatkovskiy Shengyu Fu Michele Tufano Shao Kun Deng Colin Clement Dawn Drain Neel Sundaresan Jian Yin Daxin Jiang and Ming Zhou \"Graphcode{bert}: Pre-training code representations with data flow \" in International Conference on Learning Representations 2021."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/TSE.2023.3311796"}],"event":{"name":"ICPC '24: 32nd IEEE\/ACM International Conference on Program Comprehension","location":"Lisbon Portugal","acronym":"ICPC '24","sponsor":["SIGSOFT ACM Special Interest Group on Software Engineering","IEEE CS"]},"container-title":["Proceedings of the 32nd IEEE\/ACM International Conference on Program Comprehension"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3643916.3645030","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3643916.3645030","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T23:56:44Z","timestamp":1750291004000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3643916.3645030"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,4,15]]},"references-count":26,"alternative-id":["10.1145\/3643916.3645030","10.1145\/3643916"],"URL":"https:\/\/doi.org\/10.1145\/3643916.3645030","relation":{},"subject":[],"published":{"date-parts":[[2024,4,15]]},"assertion":[{"value":"2024-06-13","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}