{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T13:02:18Z","timestamp":1784552538887,"version":"3.55.0"},"reference-count":54,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T00:00:00Z","timestamp":1784505600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T00:00:00Z","timestamp":1784505600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Empir Software Eng"],"published-print":{"date-parts":[[2027,2]]},"DOI":"10.1007\/s10664-026-10928-x","type":"journal-article","created":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T12:08:13Z","timestamp":1784549293000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Balancing usefulness and naturalness: an LLM-based curation pipeline for code review comments"],"prefix":"10.1007","volume":"32","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2737-0952","authenticated-orcid":false,"given":"Oussama","family":"Ben Sghaier","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5987-850X","authenticated-orcid":false,"given":"Martin","family":"Weyssow","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6304-9926","authenticated-orcid":false,"given":"Houari","family":"Sahraoui","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,20]]},"reference":[{"issue":"3","key":"10928_CR1","doi-asserted-by":"publisher","first-page":"31","DOI":"10.1109\/52.28121","volume":"6","author":"AF Ackerman","year":"1989","unstructured":"Ackerman AF, Buchwald LS, Lewski FH (1989) Software inspections: an effective verification process. IEEE Softw 6(3):31\u201336","journal-title":"IEEE Softw"},{"key":"10928_CR2","doi-asserted-by":"publisher","unstructured":"Bacchelli A, Bird C (2013) Expectations, outcomes, and challenges of modern code review. In: 2013 35th International conference on software engineering (ICSE), pp 712\u2013721. https:\/\/doi.org\/10.1109\/ICSE.2013.6606617","DOI":"10.1109\/ICSE.2013.6606617"},{"key":"10928_CR3","doi-asserted-by":"crossref","unstructured":"Bavota G, Russo B (2015) Four eyes are better than two: on the impact of code reviews on software quality. In: 2015 IEEE international conference on software maintenance and evolution (ICSME). IEEE, pp 81\u201390","DOI":"10.1109\/ICSM.2015.7332454"},{"issue":"3","key":"10928_CR4","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3641289","volume":"15","author":"Y Chang","year":"2024","unstructured":"Chang Y, Wang X, Wang J, Wu Y, Yang L, Zhu K, Chen H, Yi X, Wang C, Wang Y et al (2024) A survey on evaluation of large language models. ACM Trans Intell Syst Technol 15(3):1\u201345","journal-title":"ACM Trans Intell Syst Technol"},{"key":"10928_CR5","doi-asserted-by":"publisher","unstructured":"Chen N, Lin J, Hoi S, Xiao X, Zhang B (2014) AR-miner: mining informative reviews for developers from mobile app marketplace. In Proceedings of the 36th International Conference on Software Engineering (ICSE 2014). Association for Computing Machinery, New York, NY, USA, 767\u2013778. https:\/\/doi.org\/10.1145\/2568225.2568263","DOI":"10.1145\/2568225.2568263"},{"key":"10928_CR6","unstructured":"CuREV - HF Dataset (2025) Curev hf dataset. https:\/\/huggingface.co\/datasets\/OussamaBS\/CuREV"},{"key":"10928_CR7","unstructured":"CuREV Repository (2025) Curev repository. https:\/\/github.com\/OussamaSghaier\/CuREV"},{"key":"10928_CR8","unstructured":"CuREV+ - Replication package (2025) Curev+ - replication package. https:\/\/zenodo.org\/records\/17337508"},{"key":"10928_CR9","unstructured":"CuREV+ Repository (2025) Curev+ repository. https:\/\/github.com\/OussamaSghaier\/CuREV-plus"},{"key":"10928_CR10","unstructured":"CuREV+ HF Dataset (2025) Curev+ hf dataset. https:\/\/huggingface.co\/datasets\/OussamaBS\/CuREV-plus"},{"key":"10928_CR11","unstructured":"Data and Models (2025) Curev replication package. https:\/\/zenodo.org\/records\/14812107"},{"key":"10928_CR12","doi-asserted-by":"crossref","unstructured":"Fagan M (2002) Design and code inspections to reduce errors in program development. Software pioneers. Springer, pp 575\u2013607","DOI":"10.1007\/978-3-642-59412-0_35"},{"key":"10928_CR13","unstructured":"Guo D, Zhu Q, Yang Dea (2024) Deepseek-coder: When the large language model meets programming \u2013 the rise of code intelligence. arxiv:2401.14196"},{"key":"10928_CR14","unstructured":"Gupta A, Sundaresan N (2018) Intelligent code reviews using deep learning. In: Proceedings of the 24th ACM SIGKDD international conference on knowledge discovery and data mining (KDD\u201918) deep learning day"},{"key":"10928_CR15","doi-asserted-by":"crossref","unstructured":"Haouari D, Sahraoui H, Langlais P (2011) How good is your comment? A study of comments in java programs. In: 2011 International symposium on empirical software engineering and measurement. IEEE, pp 137\u2013146","DOI":"10.1109\/ESEM.2011.22"},{"key":"10928_CR16","doi-asserted-by":"crossref","unstructured":"Hindle A, Barr E, Su Z, Gabel M, Devanbu P (2012) On the naturalness of software. In Proceedings of the 34th International Conference on Software Engineering (ICSE '12). IEEE Press, 837\u2013847.","DOI":"10.1109\/ICSE.2012.6227135"},{"key":"10928_CR17","doi-asserted-by":"crossref","unstructured":"Hong Y, Tantithamthavorn C, Thongtanunam P, Aleti A (2022) Commentfinder: a simpler, faster, more accurate code review comments recommendation. In: Proceedings of the 30th ACM joint European software engineering conference and symposium on the foundations of software engineering, pp 507\u2013519","DOI":"10.1145\/3540250.3549119"},{"key":"10928_CR18","doi-asserted-by":"crossref","unstructured":"Hou X, Zhao Y, Liu Y, Yang Z, Wang K, Li L, Luo X, Lo D, Grundy J, Wang H (2023) Large language models for software engineering: a systematic literature review. ACM Transactions on Software Engineering and Methodology","DOI":"10.1145\/3695988"},{"key":"10928_CR19","unstructured":"Hu EJ, Shen Y, Wallis P, Allen-Zhu Z, Li Y, Wang S, Wang L, Chen W (2021) Lora: low-rank adaptation of large language models. arXiv preprint arXiv:2106.09685"},{"key":"10928_CR20","doi-asserted-by":"crossref","unstructured":"Li J, Galley M, Brockett C, Gao J, Dolan B (2015) A diversity-promoting objective function for neural conversation models. arXiv:1510.03055 arXiv preprint","DOI":"10.18653\/v1\/N16-1014"},{"key":"10928_CR21","unstructured":"Li X, Zhang T, Dubois Y, Taori R, Gulrajani I, Guestrin C, Liang P, Hashimoto TB (2023) Alpacaeval: an automatic evaluator of instruction-following models"},{"key":"10928_CR22","doi-asserted-by":"crossref","unstructured":"Li Z, Lu S, Guo Dea (2022a) Automating code review activities by large-scale pre-training. In: Proceedings of the 30th ACM Joint European software engineering conference and symposium on the foundations of software engineering, pp 1035\u20131047","DOI":"10.1145\/3540250.3549081"},{"key":"10928_CR23","doi-asserted-by":"crossref","unstructured":"Li L, Yang L, Jiang H, Yan J, Luo T, Hua Z, Liang G, Zuo C (2022b) Auger: automatically generating review comments with pre-training models. In: Proceedings of the 30th ACM Joint European Software Engineering Conference and Symposium on the Foundations of Software Engineering, pp 1009\u20131021","DOI":"10.1145\/3540250.3549099"},{"key":"10928_CR24","doi-asserted-by":"crossref","unstructured":"Lu J, Yu L, Li X, Yang L, Zuo C (2023) Llama-reviewer: advancing code review automation with large language models through parameter-efficient fine-tuning. In: 2023 IEEE 34th international symposium on software reliability engineering (ISSRE), IEEE, pp 647\u2013658","DOI":"10.1109\/ISSRE59848.2023.00026"},{"key":"10928_CR25","doi-asserted-by":"crossref","unstructured":"Lu J, Li X, Hua Z, Yu L, Cheng S, Yang L, Zhang F, Zuo C (2025) Deepcrceval: Revisiting the evaluation of code review comment generation. In: International conference on fundamental approaches to software engineering. Springer Nature Switzerland Cham, pp 43\u201364","DOI":"10.1007\/978-3-031-90900-9_3"},{"issue":"3","key":"10928_CR26","doi-asserted-by":"publisher","first-page":"276","DOI":"10.11613\/BM.2012.031","volume":"22","author":"ML McHugh","year":"2012","unstructured":"McHugh ML (2012) Interrater reliability: the kappa statistic. Biochemia Medica 22(3):276\u2013282","journal-title":"Biochemia Medica"},{"key":"10928_CR27","doi-asserted-by":"crossref","unstructured":"McIntosh S, Kamei Y, Adams B, Hassan AE (2014) The impact of code review coverage and code review participation on software quality: a case study of the Qt, VTK, and ITK projects. In: 11th working conference on mining software repositories, pp 192\u2013201","DOI":"10.1145\/2597073.2597076"},{"issue":"5","key":"10928_CR28","doi-asserted-by":"publisher","first-page":"2146","DOI":"10.1007\/s10664-015-9381-9","volume":"21","author":"S McIntosh","year":"2016","unstructured":"McIntosh S, Kamei Y, Adams B, Hassan AE (2016) An empirical study of the impact of modern code review practices on software quality. Empir Softw Eng 21(5):2146\u20132189","journal-title":"Empir Softw Eng"},{"key":"10928_CR29","doi-asserted-by":"crossref","unstructured":"Morales R, McIntosh S, Khomh F (2015) Do code review practices impact design quality? a case study of the Qt, VTK, and ITK projects. In: 2015 IEEE 22nd international conference on software analysis, evolution, and reengineering (SANER), IEEE, pp 171\u2013180","DOI":"10.1109\/SANER.2015.7081827"},{"key":"10928_CR30","doi-asserted-by":"crossref","unstructured":"Papineni K, Roukos S, Ward T, Zhu WJ (2002) Bleu: a method for automatic evaluation of machine translation. In: Proceedings of the 40th annual meeting of the association for computational linguistics, pp 311\u2013318","DOI":"10.3115\/1073083.1073135"},{"issue":"FSE","key":"10928_CR31","doi-asserted-by":"publisher","first-page":"1632","DOI":"10.1145\/3660780","volume":"1","author":"MS Rahman","year":"2024","unstructured":"Rahman MS, Codabux Z, Roy CK (2024) Do words have power? Understanding and fostering civility in code review discussion. Proc ACM Softw Eng 1(FSE):1632\u20131655","journal-title":"Proc ACM Softw Eng"},{"key":"10928_CR32","doi-asserted-by":"publisher","first-page":"111515","DOI":"10.1016\/j.jss.2022.111515","volume":"195","author":"P Rani","year":"2023","unstructured":"Rani P, Blasi A, Stulova N, Panichella S, Gorla A, Nierstrasz O (2023) A decade of code comment quality assessment: a systematic literature review. J Syst Softw 195:111515","journal-title":"J Syst Softw"},{"key":"10928_CR33","unstructured":"Ren S, Guo D, Lu S, Zhou L, Liu S, Tang D, Sundaresan N, Zhou M, Blanco A, Ma S (2020) Codebleu: a method for automatic evaluation of code synthesis. arXiv preprint arXiv:2009.10297"},{"key":"10928_CR34","doi-asserted-by":"crossref","unstructured":"Sadowski C, S\u00f6derberg E, Church L, Sipko M, Bacchelli A (2018) Modern code review: a case study at google. In: Proceedings of the 40th international conference on software engineering: software engineering in practice, pp 181\u2013190","DOI":"10.1145\/3183519.3183525"},{"key":"10928_CR35","doi-asserted-by":"crossref","unstructured":"Sennrich R, Haddow B, Birch A (2015) Neural machine translation of rare words with subword units. arXiv:1508.07909 arXiv preprint","DOI":"10.18653\/v1\/P16-1162"},{"key":"10928_CR36","doi-asserted-by":"crossref","unstructured":"Sghaier OB, Sahraoui H (2023) A multi-step learning approach to assist code review. In: 2023 IEEE 23rd international conference on software analysis, evolution, and reengineering (SANER). IEEE","DOI":"10.1109\/SANER56733.2023.00049"},{"issue":"FSE","key":"10928_CR37","doi-asserted-by":"publisher","first-page":"1086","DOI":"10.1145\/3643775","volume":"1","author":"O Ben Sghaier","year":"2024","unstructured":"Ben Sghaier O, Sahraoui H (2024) Improving the learning of code review successive tasks with cross-task knowledge distillation. Proc ACM Softw Eng 1(FSE):1086\u20131106","journal-title":"Proc ACM Softw Eng"},{"key":"10928_CR38","unstructured":"Sghaier OB, Maes L, Sahraoui H (2023) Unity is strength: cross-task knowledge distillation to improve code review generation. arXiv:2309.03362 arXiv preprint"},{"key":"10928_CR39","doi-asserted-by":"crossref","unstructured":"Sghaier OB, Weyssow M, Sahraoui H (2025) Harnessing large language models for curated code reviews. In: 2025 IEEE\/ACM 22nd international conference on mining software repositories (MSR). IEEE, pp 187\u2013198","DOI":"10.1109\/MSR66628.2025.00039"},{"key":"10928_CR40","doi-asserted-by":"crossref","unstructured":"Shi L, Mu F, Chen X, Wang S, Wang J, Yang Y, Li G, Xia X, Wang Q (2022) Are we building on the rock? on the importance of data preprocessing for code summarization. In: Proceedings of the 30th ACM Joint European software engineering conference and symposium on the foundations of software engineering, pp 107\u2013119","DOI":"10.1145\/3540250.3549145"},{"key":"10928_CR41","unstructured":"Silva A, Fang S, Monperrus M (2023) Repairllama: efficient representations and fine-tuned adapters for program repair. arXiv:2312.15698 arXiv preprint"},{"key":"10928_CR42","doi-asserted-by":"crossref","unstructured":"Terryn AR, de\u00a0Lhoneux M (2024) Exploratory study on the impact of english bias of generative large language models in dutch and french. In: Proceedings of the fourth workshop on human evaluation of NLP systems (HumEval)@ LREC-COLING 2024, pp 12\u201327","DOI":"10.63317\/5ogbe6wmm74i"},{"key":"10928_CR43","doi-asserted-by":"crossref","unstructured":"Tufano R, Pascarella L, Tufano M, Poshyvanyk D, Bavota G (2021a) Towards automating code review activities. In: 2021 IEEE\/ACM 43rd international conference on software engineering (ICSE), IEEE, pp 163\u2013174","DOI":"10.1109\/ICSE43902.2021.00027"},{"key":"10928_CR44","doi-asserted-by":"crossref","unstructured":"Tufano R, Pascarella L, Tufanoy M, Poshyvanykz D, Bavota G (2021b) Towards automating code review activities. In: 2021 IEEE\/ACM 43rd international conference on software engineering (ICSE), pp 163\u2013174","DOI":"10.1109\/ICSE43902.2021.00027"},{"key":"10928_CR45","doi-asserted-by":"crossref","unstructured":"Tufano R, Masiero S, Mastropaolo A, Pascarella L, Poshyvanyk D, Bavota G (2022) Using pre-trained models to boost code review automation. arXiv:2201.06850 arXiv preprint","DOI":"10.1145\/3510003.3510621"},{"key":"10928_CR46","doi-asserted-by":"crossref","unstructured":"Tufano R, Dabi\u0107 O, Mastropaolo A, Ciniselli M, Bavota G (2024) Code review automation: strengths and weaknesses of the state of the art. IEEE Trans Software Eng","DOI":"10.1109\/TSE.2023.3348172"},{"key":"10928_CR47","unstructured":"Weyssow M, Zhou X, Kim K, Lo D, Sahraoui H (2023) Exploring parameter-efficient fine-tuning techniques for code generation with large language models. arXiv:2308.10462 arXiv preprint"},{"key":"10928_CR48","unstructured":"Weyssow M, Kamanda A, Sahraoui H (2024) Codeultrafeedback: an llm-as-a-judge dataset for aligning large language models to coding preferences. arXiv:2403.09032 arXiv preprint"},{"key":"10928_CR49","unstructured":"Zhang T, Kishore V, Wu F, Weinberger KQ, Artzi Y (2020) Bertscore: evaluating text generation with bert. International Conference on Learning Representations"},{"key":"10928_CR50","doi-asserted-by":"publisher","first-page":"46595","DOI":"10.52202\/075280-2020","volume":"36","author":"L Zheng","year":"2023","unstructured":"Zheng L, Chiang WL, Sheng Y, Zhuang S, Wu Z, Zhuang Y, Lin Z, Li Z, Li D, Xing E et al (2023) Judging llm-as-a-judge with mt-bench and chatbot arena. Adv Neural Inf Process Syst 36:46595\u201346623","journal-title":"Adv Neural Inf Process Syst"},{"key":"10928_CR51","doi-asserted-by":"crossref","unstructured":"Zheng L, Chiang WL, Sheng Y, Zhuang S, Wu Z, Zhuang Y, Lin Z, Li Z, Li D, Xing E et\u00a0al (2024) Judging llm-as-a-judge with mt-bench and chatbot arena. Advances in Neural Information Processing Systems 36","DOI":"10.52202\/075280-2020"},{"issue":"5","key":"10928_CR52","doi-asserted-by":"publisher","first-page":"593","DOI":"10.3390\/electronics10050593","volume":"10","author":"J Zhou","year":"2021","unstructured":"Zhou J, Gandomi AH, Chen F, Holzinger A (2021) Evaluating the quality of machine learning explanations: a survey on methods and metrics. Electronics 10(5):593","journal-title":"Electronics"},{"key":"10928_CR53","doi-asserted-by":"crossref","unstructured":"Zhu Y, Lu S, Zheng L, Guo J, Zhang W, Wang J, Yu Y (2018) Texygen: a benchmarking platform for text generation models. In: The 41st international ACM SIGIR conference on research and development in information retrieval (SIGIR \u201918. ACM, pp 1097\u20131100","DOI":"10.1145\/3209978.3210080"},{"key":"10928_CR54","doi-asserted-by":"crossref","unstructured":"Zhuo TY (2023) Ice-score: instructing large language models to evaluate code. arXiv:2304.14317 arXiv preprint","DOI":"10.18653\/v1\/2024.findings-eacl.148"}],"container-title":["Empirical Software Engineering"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10664-026-10928-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10664-026-10928-x","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10664-026-10928-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T12:08:23Z","timestamp":1784549303000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10664-026-10928-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,20]]},"references-count":54,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2027,2]]}},"alternative-id":["10928"],"URL":"https:\/\/doi.org\/10.1007\/s10664-026-10928-x","relation":{},"ISSN":["1382-3256","1573-7616"],"issn-type":[{"value":"1382-3256","type":"print"},{"value":"1573-7616","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,20]]},"assertion":[{"value":"13 October 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 July 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"20 July 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors declare that they have no known competing interests or personal relationships that could have appeared to influence the work reported in this article.","order":1,"name":"Ethics","label":"Conflicts of Interest","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"This study does not involve human participants or animals.","order":2,"name":"Ethics","label":"Ethical Approval","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Not applicable. No human subjects were involved in this study.","order":3,"name":"Ethics","label":"Informed Consent","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Not applicable.","order":4,"name":"Ethics","label":"Clinical Trial Number","group":{"name":"EthicsHeading","label":"Declarations"}}],"article-number":"7"}}