{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,22]],"date-time":"2026-07-22T04:47:01Z","timestamp":1784695621668,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":19,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,4,12]],"date-time":"2026-04-12T00:00:00Z","timestamp":1775952000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"name":"VINNOVA","award":["2023-00541"],"award-info":[{"award-number":["2023-00541"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,4,12]]},"DOI":"10.1145\/3793655.3793712","type":"proceedings-article","created":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T16:04:46Z","timestamp":1784649886000},"page":"213-217","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["COMPASS: A Psychometrics-Guided Multi-Dimensional Benchmark for Code Generation Evaluation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0561-7977","authenticated-orcid":false,"given":"James","family":"Meaden","sequence":"first","affiliation":[{"name":"Codility, London, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7879-4371","authenticated-orcid":false,"given":"Markus","family":"Borg","sequence":"additional","affiliation":[{"name":"CodeScene, Malm\u00f6, Sweden and Lund University, Lund, Sweden"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,21]]},"reference":[{"key":"e_1_3_3_1_2_2","volume-title":"arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2108.07732","author":"Austin Jacob","year":"2021","unstructured":"Jacob Austin, Augustus Odena, Maxwell Nye, Maarten Bosma, Henryk Michalewski, David Dohan, Ellen Jiang, Carrie Cai, Michael Terry, Quoc Le, and Charles Sutton. 2021. Program Synthesis With Large Language Models. In arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2108.07732."},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICSME58944.2024.00072"},{"key":"e_1_3_3_1_4_2","doi-asserted-by":"publisher","DOI":"10.1145\/3644384.3644471"},{"key":"e_1_3_3_1_5_2","doi-asserted-by":"publisher","unstructured":"Liguo Chen Qi Guo Hongrui Jia Zhengran Zeng Xin Wang Yijiang Xu Jian Wu Yidong Wang Qing Gao Jindong Wang Wei Ye and Shikun Zhang. 2025. A Survey on Evaluating Large Language Models in Code Generation Tasks. 10.48550\/arXiv.2408.16498","DOI":"10.48550\/arXiv.2408.16498"},{"key":"e_1_3_3_1_6_2","unstructured":"Mark Chen Jerry Tworek Heewoo Jun Qiming Yuan Henrique\u00a0Ponde de Oliveira\u00a0Pinto Jared Kaplan Harri Edwards Yuri Burda Nicholas Joseph Greg Brockman Alex Ray Raul Puri Gretchen Krueger Mikhail Petrov Heidy Khlaaf Girish Sastry Pamela Mishkin Brooke Chan Scott Gray Nick Ryder Michael Pavlov Alethea Power Lukas Kaiser Mohammad Bavarian Clemens Winter Phil Tillet Felipe\u00a0Petroski Such Dave Cummings Matthias Plappert Fotios Chantzis Elizabeth Barnes Ariel Herbert-Voss William\u00a0H Guss Alex Nichol Alex Paino Nikolas Tezak Jie Tang Igor Babuschkin Shantanu Balaji Suyog Jain William Saunders Christopher Hesse Andrew Carr Jan Leike Joshua Achiam Vedant Misra Eiji Morikawa Alec Radford Melanie Knight Miles Brundage Mira Murati Pamela Mayer Peter Welinder Bob McGrew Dario Amodei Paul\u00a0F Christiano John Schulman Jared Kaplan and Ilya Sutskever. 2021. Evaluating Large Language Models Trained on Code. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2107.03374 (2021)."},{"key":"e_1_3_3_1_7_2","doi-asserted-by":"publisher","unstructured":"Daniel Graziotin Per Lenberg Robert Feldt and Stefan Wagner. 2021. Psychometrics in Behavioral Software Engineering: A Methodological Introduction with Guidelines. ACM Trans. Softw. Eng. Methodol. 31 1 (2021) 7:1\u20137:36. 10.1145\/3469888","DOI":"10.1145\/3469888"},{"key":"e_1_3_3_1_8_2","volume-title":"NeurIPS Datasets and Benchmarks Track","author":"Hendrycks Dan","year":"2021","unstructured":"Dan Hendrycks, Steven Basart, Mantas Mazeika, Saurav Kadavath, Andy Arora, Ethan Guo, Collin Burns, Abhay Puranik, Eric He, and Dawn Song. 2021. Measuring Coding Challenge Competence With APPS. In NeurIPS Datasets and Benchmarks Track."},{"key":"e_1_3_3_1_9_2","unstructured":"ISO\/IEC 25010. 2023. Systems and Software Engineering \u2014 Systems and Software Quality Requirements and Evaluation (SQuaRE) \u2014 System and Software Quality Models. https:\/\/www.iso.org\/standard\/78176.html."},{"key":"e_1_3_3_1_10_2","unstructured":"ISO\/IEC 5055. 2021. Information Technology \u2014 Software Measurement \u2014 Automated Source Code Quality Measures. https:\/\/www.iso.org\/standard\/80623.html."},{"key":"e_1_3_3_1_11_2","doi-asserted-by":"publisher","unstructured":"Juyong Jiang Fan Wang Jiasi Shen Sungju Kim and Sunghun Kim. 2025. A Survey on Large Language Models for Code Generation. ACM Trans. Softw. Eng. Methodol. (2025). 10.1145\/3747588","DOI":"10.1145\/3747588"},{"key":"e_1_3_3_1_12_2","volume-title":"NeurIPS Datasets and Benchmarks Track","author":"Lai Yiyang","year":"2023","unstructured":"Yiyang Lai, Zhaofeng Zhong, Yanda Chen, Ying Sheng, Bowen Zheng, Hexiang Hu, Yu Su, and Tong Zhang. 2023. SWE-bench: Can Language Models Resolve Real-World GitHub Issues?. In NeurIPS Datasets and Benchmarks Track."},{"key":"e_1_3_3_1_13_2","volume-title":"arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2305.12556","author":"Li Haoran","year":"2023","unstructured":"Haoran Li, Raymond Li, Loubna\u00a0Ben Allal, Beno\u00eet Sagot, and Baptiste Rozi\u00e8re. 2023. EvalPlus: Towards reliable evaluation of code generation models. In arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2305.12556."},{"key":"e_1_3_3_1_14_2","volume-title":"Machine Learning Yearning - Technincal Strategy for AI Engineers in the Era of Deep Learning","author":"Ng Andrew","year":"2018","unstructured":"Andrew Ng. 2018. Machine Learning Yearning - Technincal Strategy for AI Engineers in the Era of Deep Learning. DeepLearning.AI."},{"key":"e_1_3_3_1_15_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICSE-SEIP66354.2025.00060"},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"publisher","unstructured":"Shanghaoran Quan Jiaxi Yang Bowen Yu Bo Zheng Dayiheng Liu An Yang Xuancheng Ren Bofei Gao Yibo Miao Yunlong Feng Zekun Wang Jian Yang Zeyu Cui Yang Fan Yichang Zhang Binyuan Hui and Junyang Lin. 2025. CodeElo: Benchmarking Competition-level Code Generation of LLMs with Human-comparable Elo Ratings. 10.48550\/arXiv.2501.01257","DOI":"10.48550\/arXiv.2501.01257"},{"key":"e_1_3_3_1_17_2","doi-asserted-by":"publisher","DOI":"10.1145\/3524843.3528091"},{"key":"e_1_3_3_1_18_2","doi-asserted-by":"publisher","unstructured":"Min Zhang Tracy Hall and Nathan Baddoo. 2011. Code Bad Smells: A Review of Current Knowledge. Journal of Software Maintenance and Evolution: Research and Practice 23 3 (2011) 179\u2013202. 10.1002\/smr.521","DOI":"10.1002\/smr.521"},{"key":"e_1_3_3_1_19_2","doi-asserted-by":"publisher","unstructured":"Zibin Zheng Kaiwen Ning Yanlin Wang Jingwen Zhang Dewu Zheng Mingxi Ye and Jiachi Chen. 2024. A Survey of Large Language Models for Code: Evolution Benchmarking and Future Trends. 10.48550\/arXiv.2311.10372","DOI":"10.48550\/arXiv.2311.10372"},{"key":"e_1_3_3_1_20_2","volume-title":"Proc. of the International Conference on Learning Representations","author":"Zhuo Terry\u00a0Yue","year":"2025","unstructured":"Terry\u00a0Yue Zhuo, Minh\u00a0Chien Vu, Jenny Chim, Han Hu, Wenhao Yu, Ratnadira Widyasari, Imam Nur\u00a0Bani Yusuf, Haolan Zhan, Junda He, Indraneil Paul, Simon Brunner, Chen Gong, Thong Hoang, Armel\u00a0Randy Zebaze, Xiaoheng Hong, Wen-Ding Li, Jean Kaddour, Ming Xu, Zhihan Zhang, Prateek Yadav, Naman Jain, Alex Gu, Zhoujun Cheng, Jiawei Liu, Qian Liu, Zijian Wang, Binyuan Hui, Niklas Muennighoff, David Lo, Daniel Fried, Xiaoning Du, Harm de Vries, and Leandro Von\u00a0Werra. 2025. BigCodeBench: Benchmarking Code Generation with Diverse Function Calls and Complex Instructions. In Proc. of the International Conference on Learning Representations."}],"event":{"name":"FORGE '26: IEEE\/ACM Third International Conference on AI Foundation Models and Software Engineering","location":"Rio de Janeiro , Brazil","acronym":"FORGE '26","sponsor":["SIGSOFT ACM Special Interest Group on Software Engineering"]},"container-title":["Proceedings of the 2026 IEEE\/ACM Third International Conference on AI Foundation Models and Software Engineering"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3793655.3793712","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T16:49:22Z","timestamp":1784652562000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3793655.3793712"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,12]]},"references-count":19,"alternative-id":["10.1145\/3793655.3793712","10.1145\/3793655"],"URL":"https:\/\/doi.org\/10.1145\/3793655.3793712","relation":{},"subject":[],"published":{"date-parts":[[2026,4,12]]},"assertion":[{"value":"2026-07-21","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}