{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T15:07:29Z","timestamp":1784300849064,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":33,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T00:00:00Z","timestamp":1783209600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,7,5]]},"DOI":"10.1145\/3803437.3805234","type":"proceedings-article","created":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T14:27:39Z","timestamp":1784298459000},"page":"609-619","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Does In-IDE Calibration of Large Language Models work at Scale?"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-1649-9596","authenticated-orcid":false,"given":"Roham","family":"Koohestani","sequence":"first","affiliation":[{"name":"Delft University of Technology, Delft, Netherlands"},{"name":"JetBrains Research, Amsterdam, Netherlands"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-1495-9824","authenticated-orcid":false,"given":"Agnia","family":"Sergeyuk","sequence":"additional","affiliation":[{"name":"JetBrains Research, Belgrade, Serbia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-4212-2204","authenticated-orcid":false,"given":"David","family":"Gros","sequence":"additional","affiliation":[{"name":"University of California, Davis, Davis, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-5932-7022","authenticated-orcid":false,"given":"Claudio","family":"Spiess","sequence":"additional","affiliation":[{"name":"University of California, Davis, Davis, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5941-7490","authenticated-orcid":false,"given":"Sergey","family":"Titov","sequence":"additional","affiliation":[{"name":"JetBrains Research, Amsterdam, Netherlands"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0818-2430","authenticated-orcid":false,"given":"Premkumar","family":"Devanbu","sequence":"additional","affiliation":[{"name":"University of California, Davis, Davis, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5093-5523","authenticated-orcid":false,"given":"Maliheh","family":"Izadi","sequence":"additional","affiliation":[{"name":"Delft University of Technology, Delft, Netherlands"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,17]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Proceedings of the 2019 chi conference on human factors in computing systems. 1\u201313","author":"Amershi Saleema","year":"2019","unstructured":"Saleema Amershi, Dan Weld, Mihaela Vorvoreanu, Adam Fourney, Besmira Nushi, Penny Collisson, Jina Suh, Shamsi Iqbal, Paul N Bennett, Kori Inkpen, et al. 2019. Guidelines for human-AI interaction. In Proceedings of the 2019 chi conference on human factors in computing systems. 1\u201313."},{"key":"e_1_3_2_1_2_1","volume-title":"Semi-structured qualitative studies","author":"Blandford Ann E","unstructured":"Ann E Blandford. 2013. Semi-structured qualitative studies. Interaction Design Foundation."},{"key":"e_1_3_2_1_3_1","unstructured":"Git Copilot. 2025. Github Copilot. https:\/\/github.com\/features\/copilot."},{"key":"e_1_3_2_1_4_1","volume-title":"Cursor: The AI-first Code Editor","year":"2023","unstructured":"Cursor. 2023. Cursor: The AI-first Code Editor. https:\/\/cursor.sh."},{"key":"e_1_3_2_1_5_1","volume-title":"Hantian Ding, Ming Tan, Nihal Jain, Murali Krishna Ramanathan, Ramesh Nallapati, Parminder Bhatia, Dan Roth, and Bing Xiang.","author":"Ding Yangruibo","year":"2023","unstructured":"Yangruibo Ding, Zijian Wang, Wasi Uddin Ahmad, Hantian Ding, Ming Tan, Nihal Jain, Murali Krishna Ramanathan, Ramesh Nallapati, Parminder Bhatia, Dan Roth, and Bing Xiang. 2023. CrossCodeEval: A Diverse and Multilingual Benchmark for Cross-File Code Completion. arXiv:2310.11248 [cs.LG] https:\/\/arxiv.org\/abs\/2310.11248"},{"key":"e_1_3_2_1_6_1","volume-title":"Sea change in software development: Economic and productivity analysis of the ai-powered developer lifecycle. arXiv preprint arXiv:2306.15033","author":"Dohmke Thomas","year":"2023","unstructured":"Thomas Dohmke, Marco Iansiti, and Greg Richards. 2023. Sea change in software development: Economic and productivity analysis of the ai-powered developer lifecycle. arXiv preprint arXiv:2306.15033 (2023)."},{"key":"e_1_3_2_1_7_1","unstructured":"Shihan Dou Haoxiang Jia Shenxi Wu Huiyuan Zheng Weikang Zhou Muling Wu Mingxu Chai Jessica Fan Caishuang Huang Yunbo Tao et al. 2024. What's wrong with your code generated by large language models? an extensive study. arXiv preprint arXiv:2407.06153 (2024)."},{"key":"e_1_3_2_1_8_1","volume-title":"HILDE: Intentional Code Generation via Human-in-the-Loop Decoding. arXiv preprint arXiv:2505.22906","author":"Gonz\u00e1lez Emmanuel Anaya","year":"2025","unstructured":"Emmanuel Anaya Gonz\u00e1lez, Raven Rothkopf, Sorin Lerner, and Nadia Polikarpova. 2025. HILDE: Intentional Code Generation via Human-in-the-Loop Decoding. arXiv preprint arXiv:2505.22906 (2025)."},{"key":"e_1_3_2_1_9_1","volume-title":"International conference on machine learning. PMLR, 1321\u20131330","author":"Guo Chuan","year":"2017","unstructured":"Chuan Guo, Geoff Pleiss, Yu Sun, and Kilian Q Weinberger. 2017. On calibration of modern neural networks. In International conference on machine learning. PMLR, 1321\u20131330."},{"key":"e_1_3_2_1_10_1","volume-title":"Look before you leap: An exploratory study of uncertainty measurement for large language models. arXiv preprint arXiv:2307.10236","author":"Huang Yuheng","year":"2023","unstructured":"Yuheng Huang, Jiayang Song, Zhijie Wang, Shengming Zhao, Huaming Chen, Felix Juefei-Xu, and Lei Ma. 2023. Look before you leap: An exploratory study of uncertainty measurement for large language models. arXiv preprint arXiv:2307.10236 (2023)."},{"key":"e_1_3_2_1_11_1","unstructured":"JetBrains. 2025. JetBrains AI Assistant. https:\/\/www.jetbrains.com\/junie\/."},{"key":"e_1_3_2_1_12_1","volume-title":"Proceedings of the 2025 CHI Conference on Human Factors in Computing Systems. 1\u201319","author":"Kim Sunnie SY","year":"2025","unstructured":"Sunnie SY Kim, Jennifer Wortman Vaughan, Q Vera Liao, Tania Lombrozo, and Olga Russakovsky. 2025. Fostering appropriate reliance on large language models: The role of explanations, sources, and inconsistencies. In Proceedings of the 2025 CHI Conference on Human Factors in Computing Systems. 1\u201319."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","unstructured":"Roham Koohestani Agnia Sergeyuk David Gros Claudio Spiess Sergey Titov Prem Devanbu and Maliheh Izadi. 2026. Replication Package for In-IDE Calibration. 10.5281\/zenodo.18231197","DOI":"10.5281\/zenodo.18231197"},{"key":"e_1_3_2_1_14_1","volume-title":"The Fools are Certain","author":"Kotti Zoe","year":"2025","unstructured":"Zoe Kotti, Konstantina Dritsa, Diomidis Spinellis, and Panos Louridas. 2025. The Fools are Certain; the Wise are Doubtful: Exploring LLM Confidence in Code Completion. arXiv preprint arXiv:2508.16131 (2025)."},{"key":"e_1_3_2_1_15_1","volume-title":"Trust in automation: Designing for appropriate reliance. Human factors 46, 1","author":"Lee John D","year":"2004","unstructured":"John D Lee and Katrina A See. 2004. Trust in automation: Designing for appropriate reliance. Human factors 46, 1 (2004), 50\u201380."},{"key":"e_1_3_2_1_16_1","volume-title":"Proceedings of the 2025 CHI Conference on Human Factors in Computing Systems. 1\u201316","author":"Li Jingshu","year":"2025","unstructured":"Jingshu Li, Yitian Yang, Q Vera Liao, Junti Zhang, and Yi-Chieh Lee. 2025. As Confidence Aligns: Understanding the Effect of AI Confidence on Human Self-confidence in Human-AI Decision Making. In Proceedings of the 2025 CHI Conference on Human Factors in Computing Systems. 1\u201316."},{"key":"e_1_3_2_1_17_1","first-page":"1","article-title":"The anatomy of prototypes: Prototypes as filters, prototypes as manifestations of design ideas","volume":"15","author":"Lim Youn-Kyung","year":"2008","unstructured":"Youn-Kyung Lim, Erik Stolterman, and Josh Tenenberg. 2008. The anatomy of prototypes: Prototypes as filters, prototypes as manifestations of design ideas. ACM Transactions on Computer-Human Interaction (TOCHI) 15, 2 (2008), 1\u201327.","journal-title":"ACM Transactions on Computer-Human Interaction (TOCHI)"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3696449","article-title":"A systematic review on fostering appropriate trust in Human-AI interaction: Trends, opportunities and challenges","volume":"1","author":"Mehrotra Siddharth","year":"2024","unstructured":"Siddharth Mehrotra, Chadha Degachi, Oleksandra Vereschak, Catholijn M Jonker, and Myrthe L Tielman. 2024. A systematic review on fostering appropriate trust in Human-AI interaction: Trends, opportunities and challenges. ACM Journal on Responsible Computing 1, 4 (2024), 1\u201345.","journal-title":"ACM Journal on Responsible Computing"},{"key":"e_1_3_2_1_19_1","volume-title":"Can we trust large language models generated code? a framework for in-context learning, security patterns, and code evaluations across diverse llms. arXiv preprint arXiv:2406.12513","author":"Mohsin Ahmad","year":"2024","unstructured":"Ahmad Mohsin, Helge Janicke, Adrian Wood, Iqbal H Sarker, Leandros Maglaras, and Naeem Janjua. 2024. Can we trust large language models generated code? a framework for in-context learning, security patterns, and code evaluations across diverse llms. arXiv preprint arXiv:2406.12513 (2024)."},{"key":"e_1_3_2_1_20_1","volume-title":"Proceedings of the AAAI conference on artificial intelligence","volume":"29","author":"Naeini Mahdi Pakdaman","year":"2015","unstructured":"Mahdi Pakdaman Naeini, Gregory Cooper, and Milos Hauskrecht. 2015. Obtaining well calibrated probabilities using bayesian binning. In Proceedings of the AAAI conference on artificial intelligence, Vol. 29."},{"key":"e_1_3_2_1_21_1","volume-title":"Proceedings of the 2023 ACM SIGSAC conference on computer and communications security. 2785\u20132799","author":"Perry Neil","year":"2023","unstructured":"Neil Perry, Megha Srivastava, Deepak Kumar, and Dan Boneh. 2023. Do users write more insecure code with ai assistants?. In Proceedings of the 2023 ACM SIGSAC conference on computer and communications security. 2785\u20132799."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"crossref","unstructured":"John Platt et al. 1999. Probabilistic outputs for support vector machines and comparisons to regularized likelihood methods. Advances in large margin classifiers 10 3 (1999) 61\u201374.","DOI":"10.7551\/mitpress\/1113.003.0008"},{"key":"e_1_3_2_1_23_1","volume-title":"The human-computer interaction handbook","author":"Rosson Mary Beth","unstructured":"Mary Beth Rosson and John M Carroll. 2007. Scenario-based design. In The human-computer interaction handbook. CRC Press, 1067\u20131086."},{"key":"e_1_3_2_1_24_1","first-page":"5","article-title":"Co-creation and the new landscapes of design","volume":"4","author":"Sanders Elizabeth B-N","year":"2008","unstructured":"Elizabeth B-N Sanders and Pieter Jan Stappers. 2008. Co-creation and the new landscapes of design. Co-design 4, 1 (2008), 5\u201318.","journal-title":"Co-design"},{"key":"e_1_3_2_1_25_1","volume-title":"Calibration and Correctness of Language Models for Code. In 2025 IEEE\/ACM 47th International Conference on Software Engineering (ICSE). IEEE Computer Society, 495\u2013507","author":"Spiess Claudio","year":"2024","unstructured":"Claudio Spiess, David Gros, Kunal Suresh Pai, Michael Pradel, Md Rafiqul Islam Rabin, Amin Alipour, Susmit Jha, Prem Devanbu, and Toufique Ahmed. 2024. Calibration and Correctness of Language Models for Code. In 2025 IEEE\/ACM 47th International Conference on Software Engineering (ICSE). IEEE Computer Society, 495\u2013507. 10.1109\/ICSE55347.2025.00040"},{"key":"e_1_3_2_1_26_1","volume-title":"2025 Stack Overflow Developer Survey. https:\/\/survey.stackoverflow.co\/2025 Accessed","author":"Overflow Stack","year":"2025","unstructured":"Stack Overflow. 2025. 2025 Stack Overflow Developer Survey. https:\/\/survey.stackoverflow.co\/2025 Accessed: Sep 10, 2025."},{"key":"e_1_3_2_1_27_1","volume-title":"Metacognition and uncertainty communication in humans and large language models. arXiv preprint arXiv:2504.14045","author":"Steyvers Mark","year":"2025","unstructured":"Mark Steyvers and Megan AK Peters. 2025. Metacognition and uncertainty communication in humans and large language models. arXiv preprint arXiv:2504.14045 (2025)."},{"key":"e_1_3_2_1_28_1","unstructured":"Fengfei Sun et al. 2025. Large Language Models are overconfident and amplify human bias. arXiv preprint arXiv:2505.02151 (2025)."},{"key":"e_1_3_2_1_29_1","volume-title":"Proceedings of the 27th international conference on intelligent user interfaces. 212\u2013228","author":"Sun Jiao","year":"2022","unstructured":"Jiao Sun, Q Vera Liao, Michael Muller, Mayank Agarwal, Stephanie Houde, Kartik Talamadupula, and Justin D Weisz. 2022. Investigating explainability of generative AI for code through scenario-based design. In Proceedings of the 27th international conference on intelligent user interfaces. 212\u2013228."},{"key":"e_1_3_2_1_30_1","first-page":"1","article-title":"Generation probabilities are not enough: Uncertainty highlighting in ai code completions","volume":"32","author":"Vasconcelos Helena","year":"2025","unstructured":"Helena Vasconcelos, Gagan Bansal, Adam Fourney, Q Vera Liao, and Jennifer Wortman Vaughan. 2025. Generation probabilities are not enough: Uncertainty highlighting in ai code completions. ACM Transactions on Computer-Human Interaction 32, 1 (2025), 1\u201330.","journal-title":"ACM Transactions on Computer-Human Interaction"},{"key":"e_1_3_2_1_31_1","volume-title":"Proceedings of the Extended Abstracts of the CHI Conference on Human Factors in Computing Systems. 1\u201313","author":"Weisz Justin D","year":"2025","unstructured":"Justin D Weisz, Shraddha Vijay Kumar, Michael Muller, Karen-Ellen Browne, Arielle Goldberg, Katrin Ellice Heintze, and Shagun Bajpai. 2025. Examining the use and impact of an ai code assistant on developer productivity and experience in the enterprise. In Proceedings of the Extended Abstracts of the CHI Conference on Human Factors in Computing Systems. 1\u201313."},{"key":"e_1_3_2_1_32_1","unstructured":"Windsurf. 2023. Windsurf AI Code Editor. https:\/\/windsurf.ai."},{"key":"e_1_3_2_1_33_1","volume-title":"Can LLMs express their uncertainty? An empirical evaluation of confidence elicitation in LLMs. arXiv preprint arXiv:2306.13063","author":"Xiong Miao","year":"2023","unstructured":"Miao Xiong, Zhiyuan Hu, Xinyang Lu, Yifei Li, Jie Fu, Junxian He, and Bryan Hooi. 2023. Can LLMs express their uncertainty? An empirical evaluation of confidence elicitation in LLMs. arXiv preprint arXiv:2306.13063 (2023)."}],"event":{"name":"FSE Companion '26: 34th ACM International Conference on the Foundations of Software Engineering","location":"Concordia University Montreal QC Canada","acronym":"FSE Companion '26","sponsor":["SIGSOFT ACM Special Interest Group on Software Engineering"]},"container-title":["Proceedings of the 34th ACM International Conference on the Foundations of Software Engineering"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3803437.3805234","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T14:45:28Z","timestamp":1784299528000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3803437.3805234"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,5]]},"references-count":33,"alternative-id":["10.1145\/3803437.3805234","10.1145\/3803437"],"URL":"https:\/\/doi.org\/10.1145\/3803437.3805234","relation":{},"subject":[],"published":{"date-parts":[[2026,7,5]]},"assertion":[{"value":"2026-07-17","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}