{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T03:36:46Z","timestamp":1782877006356,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":34,"publisher":"ACM","funder":[{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["CCF-2211429"],"award-info":[{"award-number":["CCF-2211429"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,6,23]]},"DOI":"10.1145\/3696630.3728510","type":"proceedings-article","created":{"date-parts":[[2025,7,28]],"date-time":"2025-07-28T19:08:09Z","timestamp":1753729689000},"page":"616-620","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Enhancing Code LLM Training with Programmer Attention"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5719-772X","authenticated-orcid":false,"given":"Yifan","family":"Zhang","sequence":"first","affiliation":[{"name":"Department of Computer Science, Vanderbilt University, Nashville, TN, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3542-7085","authenticated-orcid":false,"given":"Chen","family":"Huang","sequence":"additional","affiliation":[{"name":"Department of Computer Science, Sichuan University, Chengdu, Sichuan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5721-8794","authenticated-orcid":false,"given":"Zachary","family":"Karas","sequence":"additional","affiliation":[{"name":"Department of Computer Science, Vanderbilt University, Nashville, TN, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3489-4255","authenticated-orcid":false,"given":"Thuy Dung","family":"Nguyen","sequence":"additional","affiliation":[{"name":"Department of Computer Science, Vanderbilt University, Nashville, TN, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4001-3442","authenticated-orcid":false,"given":"Kevin","family":"Leach","sequence":"additional","affiliation":[{"name":"Department of Computer Science, Vanderbilt University, Nashville, TN, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2730-5077","authenticated-orcid":false,"given":"Yu","family":"Huang","sequence":"additional","affiliation":[{"name":"Department of Computer Science, Vanderbilt University, Nashville, TN, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,7,28]]},"reference":[{"key":"e_1_3_2_1_1_1","first-page":"e1526","article-title":"The role of lifelong machine learning in bridging the gap between human and machine learning: A scientometric analysis","volume":"14","author":"Abulaish Muhammad","year":"2024","unstructured":"Muhammad Abulaish, Nesar Ahmad Wasi, and Shachi Sharma. 2024. The role of lifelong machine learning in bridging the gap between human and machine learning: A scientometric analysis. Wiley Interdisciplinary Reviews: Data Mining and Knowledge Discovery 14, 2 (2024), e1526.","journal-title":"Wiley Interdisciplinary Reviews: Data Mining and Knowledge Discovery"},{"key":"e_1_3_2_1_2_1","volume-title":"Optimizing Code Runtime Performance through Context-Aware Retrieval-Augmented Generation. arXiv preprint arXiv:2501.16692","author":"Acharya Manish","year":"2025","unstructured":"Manish Acharya, Yifan Zhang, Yu Huang, and Kevin Leach. 2025. Optimizing Code Runtime Performance through Context-Aware Retrieval-Augmented Generation. arXiv preprint arXiv:2501.16692 (2025)."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3212695","article-title":"A survey of machine learning for big code and naturalness","volume":"51","author":"Allamanis Miltiadis","year":"2018","unstructured":"Miltiadis Allamanis, Earl T Barr, Premkumar Devanbu, and Charles Sutton. 2018. A survey of machine learning for big code and naturalness. ACM Computing Surveys (CSUR) 51, 4 (2018), 1\u201337.","journal-title":"ACM Computing Surveys (CSUR)"},{"key":"e_1_3_2_1_4_1","volume-title":"Proceedings of the ACM on Human-Computer Interaction 7, ETRA","author":"Bansal Aakash","year":"2023","unstructured":"Aakash Bansal, Bonita Sharif, and Collin McMillan. 2023. Towards modeling human attention from eye movements for neural source code summarization. Proceedings of the ACM on Human-Computer Interaction 7, ETRA (2023), 1\u201319."},{"key":"e_1_3_2_1_5_1","volume-title":"2023 38th IEEE\/ACM International Conference on Automated Software Engineering (ASE). IEEE, 1732\u20131736","author":"Bansal Aakash","year":"2023","unstructured":"Aakash Bansal, Chia-Yi Su, Zachary Karas, Yifan Zhang, Yu Huang, Toby Jia-Jun Li, and Collin McMillan. 2023. Modeling programmer attention as scanpath prediction. In 2023 38th IEEE\/ACM International Conference on Automated Software Engineering (ASE). IEEE, 1732\u20131736."},{"key":"e_1_3_2_1_6_1","volume-title":"Revisiting File Context for Source Code Summarization. arXiv preprint arXiv:2309.02326","author":"Bansal Aakash","year":"2023","unstructured":"Aakash Bansal, Chia-Yi Su, and Collin McMillan. 2023. Revisiting File Context for Source Code Summarization. arXiv preprint arXiv:2309.02326 (2023)."},{"key":"e_1_3_2_1_7_1","volume-title":"Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, et al.","author":"Chen Mark","year":"2021","unstructured":"Mark Chen, Jerry Tworek, Heewoo Jun, Qiming Yuan, Henrique Ponde De Oliveira Pinto, Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, et al. 2021. Evaluating large language models trained on code. arXiv preprint arXiv:2107.03374 (2021)."},{"key":"e_1_3_2_1_8_1","volume-title":"Codebert: A pre-trained model for programming and natural languages. arXiv preprint arXiv:2002.08155","author":"Feng Zhangyin","year":"2020","unstructured":"Zhangyin Feng, Daya Guo, Duyu Tang, Nan Duan, Xiaocheng Feng, Ming Gong, Linjun Shou, Bing Qin, Ting Liu, Daxin Jiang, et al. 2020. Codebert: A pre-trained model for programming and natural languages. arXiv preprint arXiv:2002.08155 (2020)."},{"key":"e_1_3_2_1_9_1","volume-title":"Proceedings of the 7th ACM\/IEEE International Workshop on Software-intensive Business. 7\u201314","author":"Hamza Muhammad","year":"2024","unstructured":"Muhammad Hamza, Dominik Siemon, Muhammad Azeem Akbar, and Tahsinur Rahman. 2024. Human-ai collaboration in software engineering: Lessons learned from a hands-on workshop. In Proceedings of the 7th ACM\/IEEE International Workshop on Software-intensive Business. 7\u201314."},{"key":"e_1_3_2_1_10_1","volume-title":"Exploring Demonstration Retrievers in RAG for Coding Tasks: Yeas and Nays! arXiv preprint arXiv:2410.09662","author":"He Pengfei","year":"2024","unstructured":"Pengfei He, Shaowei Wang, Shaiful Chowdhury, and Tse-Hsun Chen. 2024. Exploring Demonstration Retrievers in RAG for Coding Tasks: Yeas and Nays! arXiv preprint arXiv:2410.09662 (2024)."},{"key":"e_1_3_2_1_11_1","volume-title":"Proceedings of the 55th ACM Technical Symposium on Computer Science Education V. 2. 1680\u20131681","author":"Hoq Muntasir","year":"2024","unstructured":"Muntasir Hoq, Jessica Vandenberg, Bradford Mott, James Lester, Narges Norouzi, and Bita Akram. 2024. Towards Attention-Based Automatic Misconception Identification in Introductory Programming Courses. In Proceedings of the 55th ACM Technical Symposium on Computer Science Education V. 2. 1680\u20131681."},{"key":"e_1_3_2_1_12_1","volume-title":"How to Enable Effective Cooperation Between Humans and NLP Models: A Survey of Principles, Formalizations, and Beyond. arXiv preprint arXiv:2501.05714","author":"Huang Chen","year":"2025","unstructured":"Chen Huang, Yang Deng, Wenqiang Lei, Jiancheng Lv, Tat-Seng Chua, and Jimmy Xiangji Huang. 2025. How to Enable Effective Cooperation Between Humans and NLP Models: A Survey of Principles, Formalizations, and Beyond. arXiv preprint arXiv:2501.05714 (2025)."},{"key":"e_1_3_2_1_13_1","volume-title":"Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing. 10876\u201310891","author":"Huang Chen","year":"2023","unstructured":"Chen Huang, Peixin Qin, Wenqiang Lei, and Jiancheng Lv. 2023. Reduce Human Labor On Evaluating Conversational Information Retrieval System: A Human-Machine Collaboration Approach. In Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing. 10876\u201310891."},{"key":"e_1_3_2_1_14_1","volume-title":"A Tale of Two Comprehensions? Analyzing Student Programmer Attention during Code Summarization. ACM Transactions on Software Engineering and Methodology","author":"Karas Zachary","year":"2024","unstructured":"Zachary Karas, Aakash Bansal, Yifan Zhang, Toby Li, Collin McMillan, and Yu Huang. 2024. A Tale of Two Comprehensions? Analyzing Student Programmer Attention during Code Summarization. ACM Transactions on Software Engineering and Methodology (2024)."},{"key":"e_1_3_2_1_15_1","volume-title":"Proceedings of the ACM on Software Engineering 1, FSE","author":"Kou Bonan","year":"2024","unstructured":"Bonan Kou, Shengmai Chen, Zhijie Wang, Lei Ma, and Tianyi Zhang. 2024. Do large language models pay similar attention like human programmers when generating code? Proceedings of the ACM on Software Engineering 1, FSE (2024), 2261\u20132284."},{"key":"e_1_3_2_1_16_1","first-page":"2086","article-title":"Understanding more about human and machine attention in deep neural networks","volume":"23","author":"Lai Qiuxia","year":"2020","unstructured":"Qiuxia Lai, Salman Khan, Yongwei Nie, Hanqiu Sun, Jianbing Shen, and Ling Shao. 2020. Understanding more about human and machine attention in deep neural networks. IEEE Transactions on Multimedia 23 (2020), 2086\u20132099.","journal-title":"IEEE Transactions on Multimedia"},{"key":"e_1_3_2_1_17_1","volume-title":"Malmixer: Few-shot malware classification with retrieval-augmented semi-supervised learning. arXiv preprint arXiv:2409.13213","author":"Li Jiliang","year":"2024","unstructured":"Jiliang Li, Yifan Zhang, Yu Huang, and Kevin Leach. 2024. Malmixer: Few-shot malware classification with retrieval-augmented semi-supervised learning. arXiv preprint arXiv:2409.13213 (2024)."},{"key":"e_1_3_2_1_18_1","volume-title":"Proceedings of the 32nd IEEE\/ACM International Conference on Program Comprehension. 47\u201351","author":"Li Jiliang","year":"2024","unstructured":"Jiliang Li, Yifan Zhang, Zachary Karas, Collin McMillan, Kevin Leach, and Yu Huang. 2024. Do Machines and Humans Focus on Similar Code? Exploring Explainability of Large Language Models in Code Summarization. In Proceedings of the 32nd IEEE\/ACM International Conference on Program Comprehension. 47\u201351."},{"key":"e_1_3_2_1_19_1","unstructured":"Yao Lu Song Bian Lequn Chen Yongjun He Yulong Hui Matthew Lentz Beibin Li Fei Liu Jialin Li Qi Liu et al. 2024. Computing in the Era of Large Generative Models: From Cloud-Native to AI-Native. arXiv preprint arXiv:2401.12230 (2024)."},{"key":"e_1_3_2_1_20_1","volume-title":"2015 IEEE 23rd international conference on program comprehension. IEEE, 25\u201335","author":"Minelli Roberto","year":"2015","unstructured":"Roberto Minelli, Andrea Mocci, and Michele Lanza. 2015. I know what you did last summer-an investigation of how developers spend their time. In 2015 IEEE 23rd international conference on program comprehension. IEEE, 25\u201335."},{"key":"e_1_3_2_1_21_1","volume-title":"Codegen2: Lessons for training llms on programming and natural languages. arXiv preprint arXiv:2305.02309","author":"Nijkamp Erik","year":"2023","unstructured":"Erik Nijkamp, Hiroaki Hayashi, Caiming Xiong, Silvio Savarese, and Yingbo Zhou. 2023. Codegen2: Lessons for training llms on programming and natural languages. arXiv preprint arXiv:2305.02309 (2023)."},{"key":"e_1_3_2_1_22_1","volume-title":"Proceedings of the ACM on Programming Languages 8, POPL","author":"Pailoor Shankara","year":"2024","unstructured":"Shankara Pailoor, Yuepeng Wang, and I\u015f\u0131l Dillig. 2024. Semantic code refactoring for abstract data types. Proceedings of the ACM on Programming Languages 8, POPL (2024), 816\u2013847."},{"key":"e_1_3_2_1_23_1","volume-title":"Codebleu: a method for automatic evaluation of code synthesis. arXiv preprint arXiv:2009.10297","author":"Ren Shuo","year":"2020","unstructured":"Shuo Ren, Daya Guo, Shuai Lu, Long Zhou, Shujie Liu, Duyu Tang, Neel Sundaresan, Ming Zhou, Ambrosio Blanco, and Shuai Ma. 2020. Codebleu: a method for automatic evaluation of code synthesis. arXiv preprint arXiv:2009.10297 (2020)."},{"key":"e_1_3_2_1_24_1","volume-title":"Proceedings of the 37th IEEE\/ACM International Conference on Automated Software Engineering. 1\u201312","author":"Richter Cedric","year":"2022","unstructured":"Cedric Richter, Jan Haltermann, Marie-Christine Jakobs, Felix Pauck, Stefan Schott, and Heike Wehrheim. 2022. Are Neural Bug Detectors Comparable to Software Developers on Variable Misuse Bugs?. In Proceedings of the 37th IEEE\/ACM International Conference on Automated Software Engineering. 1\u201312."},{"key":"e_1_3_2_1_25_1","volume-title":"Proceedings of the 2000 symposium on Eye tracking research & applications. 71\u201378","author":"Salvucci Dario D","year":"2000","unstructured":"Dario D Salvucci and Joseph H Goldberg. 2000. Identifying fixations and saccades in eye-tracking protocols. In Proceedings of the 2000 symposium on Eye tracking research & applications. 71\u201378."},{"key":"e_1_3_2_1_26_1","volume-title":"Proceedings of the 44th international conference on software engineering. 1597\u20131608","author":"Shi Ensheng","year":"2022","unstructured":"Ensheng Shi, Yanlin Wang, Lun Du, Junjie Chen, Shi Han, Hongyu Zhang, Dongmei Zhang, and Hongbin Sun. 2022. On the evaluation of neural code summarization. In Proceedings of the 44th international conference on software engineering. 1597\u20131608."},{"key":"e_1_3_2_1_27_1","volume-title":"2024 IEEE Symposium on Visual Languages and Human-Centric Computing (VL\/HCC). IEEE, 40\u201346","author":"Tang Ningzhi","year":"2024","unstructured":"Ningzhi Tang, Meng Chen, Zheng Ning, Aakash Bansal, Yu Huang, Collin McMillan, and Toby Jia-Jun Li. 2024. Developer behaviors in validating and repairing llm-generated code using ide and eye tracking. In 2024 IEEE Symposium on Visual Languages and Human-Centric Computing (VL\/HCC). IEEE, 40\u201346."},{"key":"e_1_3_2_1_28_1","volume-title":"Syncobert: Syntax-guided multi-modal contrastive pre-training for code representation. arXiv preprint arXiv:2108.04556","author":"Wang Xin","year":"2021","unstructured":"Xin Wang, Yasheng Wang, Fei Mi, Pingyi Zhou, Yao Wan, Xiao Liu, Li Li, Hao Wu, Jin Liu, and Xin Jiang. 2021. Syncobert: Syntax-guided multi-modal contrastive pre-training for code representation. arXiv preprint arXiv:2108.04556 (2021)."},{"key":"e_1_3_2_1_29_1","volume-title":"Nghi DQ Bui, Junnan Li, and Steven CH Hoi.","author":"Wang Yue","year":"2023","unstructured":"Yue Wang, Hung Le, Akhilesh Deepak Gotmare, Nghi DQ Bui, Junnan Li, and Steven CH Hoi. 2023. Codet5+: Open code large language models for code understanding and generation. arXiv preprint arXiv:2305.07922 (2023)."},{"key":"e_1_3_2_1_30_1","volume-title":"Codet5: Identifier-aware unified pre-trained encoder-decoder models for code understanding and generation. arXiv preprint arXiv:2109.00859","author":"Wang Yue","year":"2021","unstructured":"Yue Wang, Weishi Wang, Shafiq Joty, and Steven CH Hoi. 2021. Codet5: Identifier-aware unified pre-trained encoder-decoder models for code understanding and generation. arXiv preprint arXiv:2109.00859 (2021)."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"crossref","first-page":"111304","DOI":"10.1016\/j.jss.2022.111304","article-title":"Data augmentation by program transformation","volume":"190","author":"Yu Shiwen","year":"2022","unstructured":"Shiwen Yu, Ting Wang, and Ji Wang. 2022. Data augmentation by program transformation. Journal of Systems and Software 190 (2022), 111304.","journal-title":"Journal of Systems and Software"},{"key":"e_1_3_2_1_32_1","volume-title":"Huajie Shao, Kevin Leach, and Yu Huang.","author":"Zhang Yifan","year":"2022","unstructured":"Yifan Zhang, Chen Huang, Kevin Cao, Yueke Zhang, Scott Thomas Andersen, Huajie Shao, Kevin Leach, and Yu Huang. 2022. Pre-Training Representations of Binary Code Using Contrastive Learning. arXiv preprint arXiv:2210.05102 (2022)."},{"key":"e_1_3_2_1_33_1","volume-title":"Proceedings of the ACM on Software Engineering 1, FSE","author":"Zhang Yifan","year":"2024","unstructured":"Yifan Zhang, Jiliang Li, Zachary Karas, Aakash Bansal, Toby Jia-Jun Li, Collin McMillan, Kevin Leach, and Yu Huang. 2024. Eyetrans: Merging human and machine attention for neural code summarization. Proceedings of the ACM on Software Engineering 1, FSE (2024), 115\u2013136."},{"key":"e_1_3_2_1_34_1","volume-title":"Astro: An ast-assisted approach for generalizable neural clone detection. arXiv preprint arXiv:2208.08067","author":"Zhang Yifan","year":"2022","unstructured":"Yifan Zhang, Junwen Yang, Haoyu Dong, Qingchen Wang, Huajie Shao, Kevin Leach, and Yu Huang. 2022. Astro: An ast-assisted approach for generalizable neural clone detection. arXiv preprint arXiv:2208.08067 (2022)."}],"event":{"name":"FSE Companion '25: 33rd ACM International Conference on the Foundations of Software Engineering","location":"Clarion Hotel Trondheim Trondheim Norway","acronym":"FSE Companion '25","sponsor":["SIGSOFT ACM Special Interest Group on Software Engineering"]},"container-title":["Proceedings of the 33rd ACM International Conference on the Foundations of Software Engineering"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3696630.3728510","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,28]],"date-time":"2025-07-28T19:11:37Z","timestamp":1753729897000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3696630.3728510"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,23]]},"references-count":34,"alternative-id":["10.1145\/3696630.3728510","10.1145\/3696630"],"URL":"https:\/\/doi.org\/10.1145\/3696630.3728510","relation":{},"subject":[],"published":{"date-parts":[[2025,6,23]]},"assertion":[{"value":"2025-07-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}