{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,12]],"date-time":"2026-06-12T10:09:39Z","timestamp":1781258979317,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":49,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,2,22]],"date-time":"2024-02-22T00:00:00Z","timestamp":1708560000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,2,22]]},"DOI":"10.1145\/3641399.3641408","type":"proceedings-article","created":{"date-parts":[[2024,2,20]],"date-time":"2024-02-20T18:15:26Z","timestamp":1708452926000},"page":"1-11","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["CodeQueries: A Dataset of Semantic Queries over Code"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-9943-5222","authenticated-orcid":false,"given":"Surya Prakash","family":"Sahu","sequence":"first","affiliation":[{"name":"Indian Institute of Science, Bengaluru, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3084-3757","authenticated-orcid":false,"given":"Madhurima","family":"Mandal","sequence":"additional","affiliation":[{"name":"Myntra, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-7202-0502","authenticated-orcid":false,"given":"Shikhar","family":"Bharadwaj","sequence":"additional","affiliation":[{"name":"Google Research, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8701-6977","authenticated-orcid":false,"given":"Aditya","family":"Kanade","sequence":"additional","affiliation":[{"name":"Microsoft Research, India, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3777-5291","authenticated-orcid":false,"given":"Petros","family":"Maniatis","sequence":"additional","affiliation":[{"name":"Google DeepMind, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-7202-6860","authenticated-orcid":false,"given":"Shirish","family":"Shevade","sequence":"additional","affiliation":[{"name":"Indian Institute of Science, Bengaluru, India"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,2,22]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"QL: Object-oriented Queries on Relational Data. In 30th European Conference on Object-Oriented Programming. Schloss Dagstuhl - Leibniz-Zentrum f\u00fcr Informatik.","author":"Avgustinov Pavel","year":"2016","unstructured":"Pavel Avgustinov, Oege de Moor, Michael\u00a0Peyton Jones, and Max Sch\u00e4fer. 2016. QL: Object-oriented Queries on Relational Data. In 30th European Conference on Object-Oriented Programming. Schloss Dagstuhl - Leibniz-Zentrum f\u00fcr Informatik."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/SANER50967.2021.00015"},{"key":"e_1_3_2_1_3_1","volume-title":"Active Learning of Points-to Specifications. SIGPLAN Not. 53, 4","author":"Bastani Osbert","year":"2018","unstructured":"Osbert Bastani, Rahul Sharma, Alex Aiken, and Percy Liang. 2018. Active Learning of Points-to Specifications. SIGPLAN Not. 53, 4 (2018)."},{"key":"e_1_3_2_1_4_1","volume-title":"Computer Aided Verification - 29th International Conference","author":"Bielik Pavol","unstructured":"Pavol Bielik, Veselin Raychev, and Martin\u00a0T. Vechev. 2017. Learning a Static Analyzer from Data. In Computer Aided Verification - 29th International Conference. Springer."},{"key":"e_1_3_2_1_5_1","volume-title":"Language models are few-shot learners. Advances in neural information processing systems 33","author":"Brown Tom","year":"2020","unstructured":"Tom Brown, Benjamin Mann, Nick Ryder, Melanie Subbiah, Jared\u00a0D Kaplan, Prafulla Dhariwal, Arvind Neelakantan, Pranav Shyam, Girish Sastry, Amanda Askell, 2020. Language models are few-shot learners. Advances in neural information processing systems 33 (2020), 1877\u20131901."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3338906.3340458"},{"key":"e_1_3_2_1_7_1","volume-title":"Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, and others.","author":"Chen Mark","year":"2021","unstructured":"Mark Chen, Jerry Tworek, Heewoo Jun, Qiming Yuan, Henrique Ponde de\u00a0Oliveira Pinto, Jared Kaplan, Harri Edwards, Yuri Burda, Nicholas Joseph, Greg Brockman, and others. 2021. Evaluating large language models trained on code. arXiv preprint arXiv:2107.03374 (2021)."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/3314221.3314648"},{"key":"e_1_3_2_1_9_1","volume-title":"Simple and effective multi-paragraph reading comprehension. arXiv preprint arXiv:1710.10723","author":"Clark Christopher","year":"2017","unstructured":"Christopher Clark and Matt Gardner. 2017. Simple and effective multi-paragraph reading comprehension. arXiv preprint arXiv:1710.10723 (2017)."},{"key":"e_1_3_2_1_10_1","volume-title":"Proceedings of the 38th International Conference on Machine Learning(Proceedings of Machine Learning Research). PMLR.","author":"Cummins Chris","year":"2021","unstructured":"Chris Cummins, Zacharias\u00a0V. Fisches, Tal Ben-Nun, Torsten Hoefler, Michael F.\u00a0P. O\u2019Boyle, and Hugh Leather. 2021. ProGraML: A Graph-based Program Representation for Data Flow Analysis and Compiler Optimizations. In Proceedings of the 38th International Conference on Machine Learning(Proceedings of Machine Learning Research). PMLR."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3428293"},{"key":"e_1_3_2_1_12_1","volume-title":"Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies. Association for Computational Linguistics.","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies. Association for Computational Linguistics."},{"key":"e_1_3_2_1_13_1","volume-title":"Findings of the Association for Computational Linguistics: EMNLP","author":"Feng Zhangyin","unstructured":"Zhangyin Feng, Daya Guo, Duyu Tang, Nan Duan, Xiaocheng Feng, Ming Gong, Linjun Shou, Bing Qin, Ting Liu, Daxin Jiang, and Ming Zhou. 2020. CodeBERT: A Pre-Trained Model for Programming and Natural Languages. In Findings of the Association for Computational Linguistics: EMNLP. Association for Computational Linguistics."},{"key":"e_1_3_2_1_14_1","volume-title":"Palm 2 technical report. arXiv preprint arXiv:2305.10403","year":"2023","unstructured":"Google. 2023. Palm 2 technical report. arXiv preprint arXiv:2305.10403 (2023)."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2021.04.019"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/3180155.3180167"},{"key":"e_1_3_2_1_17_1","volume-title":"Graphcodebert: Pre-training code representations with data flow. arXiv preprint arXiv:2009.08366","author":"Guo Daya","year":"2020","unstructured":"Daya Guo, Shuo Ren, Shuai Lu, Zhangyin Feng, Duyu Tang, Shujie Liu, Long Zhou, Nan Duan, Alexey Svyatkovskiy, Shengyu Fu, 2020. Graphcodebert: Pre-training code representations with data flow. arXiv preprint arXiv:2009.08366 (2020)."},{"key":"e_1_3_2_1_18_1","volume-title":"Deep Learning Type Inference. In ACM Joint European Software Engineering Conference and Symposium on the Foundations of Software Engineering. Association for Computing Machinery.","author":"Hellendoorn J.","year":"2018","unstructured":"Vincent\u00a0J. Hellendoorn, Christian Bird, Earl\u00a0T. Barr, and Miltiadis Allamanis. 2018. Deep Learning Type Inference. In ACM Joint European Software Engineering Conference and Symposium on the Foundations of Software Engineering. Association for Computing Machinery."},{"key":"e_1_3_2_1_19_1","volume-title":"Neural Code Search Revisited: Enhancing Code Snippet Retrieval through Natural Language Intent. CoRR abs\/2008.12193","author":"Heyman Geert","year":"2020","unstructured":"Geert Heyman and Tom\u00a0Van Cutsem. 2020. Neural Code Search Revisited: Enhancing Code Snippet Retrieval through Natural Language Intent. CoRR abs\/2008.12193 (2020). arXiv:2008.12193"},{"key":"e_1_3_2_1_20_1","volume-title":"Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing","author":"Huang Junjie","unstructured":"Junjie Huang, Duyu Tang, Linjun Shou, Ming Gong, Ke Xu, Daxin Jiang, Ming Zhou, and Nan Duan. 2021. CoSQA: 20, 000+ Web Queries for Code Search and Question Answering. In Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing, ACL\/IJCNLP. Association for Computational Linguistics."},{"key":"e_1_3_2_1_21_1","volume-title":"CodeSearchNet Challenge: Evaluating the State of Semantic Code Search. CoRR abs\/1909.09436","author":"Husain Hamel","year":"2019","unstructured":"Hamel Husain, Ho-Hsiang Wu, Tiferet Gazit, Miltiadis Allamanis, and Marc Brockschmidt. 2019. CodeSearchNet Challenge: Evaluating the State of Semantic Code Search. CoRR abs\/1909.09436 (2019). arXiv:1909.09436"},{"key":"e_1_3_2_1_22_1","unstructured":"Charles Jin and Martin Rinard. 2023. Evidence of Meaning in Language Models Trained on Programs. arxiv:2305.11169\u00a0[cs.LG]"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.5555\/3524938.3525412"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.naacl-main.148"},{"key":"e_1_3_2_1_25_1","volume-title":"StarCoder: may the source be with you!arXiv preprint arXiv:2305.06161","author":"Li Raymond","year":"2023","unstructured":"Raymond Li, Loubna\u00a0Ben Allal, Yangtian Zi, Niklas Muennighoff, Denis Kocetkov, Chenghao Mou, Marc Marone, Christopher Akiki, Jia Li, Jenny Chim, 2023. StarCoder: may the source be with you!arXiv preprint arXiv:2305.06161 (2023)."},{"key":"e_1_3_2_1_26_1","volume-title":"Findings of the Association for Computational Linguistics: EMNLP","author":"Liu Chenxiao","unstructured":"Chenxiao Liu and Xiaojun Wan. 2021. CodeQA: A Question Answering Dataset for Source Code Comprehension. In Findings of the Association for Computational Linguistics: EMNLP. Association for Computational Linguistics."},{"key":"e_1_3_2_1_27_1","volume-title":"Fixing Weight Decay Regularization in Adam. CoRR abs\/1711.05101","author":"Loshchilov Ilya","year":"2017","unstructured":"Ilya Loshchilov and Frank Hutter. 2017. Fixing Weight Decay Regularization in Adam. CoRR abs\/1711.05101 (2017). arXiv:1711.05101"},{"key":"e_1_3_2_1_28_1","volume-title":"Type4Py: Deep Similarity Learning-Based Type Inference for Python. CoRR","author":"Mir M.","year":"2021","unstructured":"Amir\u00a0M. Mir, Evaldas Latoskinas, Sebastian Proksch, and Georgios Gousios. 2021. Type4Py: Deep Similarity Learning-Based Type Inference for Python. CoRR (2021). arXiv:2101.04470"},{"key":"e_1_3_2_1_29_1","volume-title":"CodeGen2: Lessons for Training LLMs on Programming and Natural Languages. arXiv preprint arXiv:2305.02309","author":"Nijkamp Erik","year":"2023","unstructured":"Erik Nijkamp, Hiroaki Hayashi, Caiming Xiong, Silvio Savarese, and Yingbo Zhou. 2023. CodeGen2: Lessons for Training LLMs on Programming and Natural Languages. arXiv preprint arXiv:2305.02309 (2023)."},{"key":"e_1_3_2_1_30_1","unstructured":"Long Ouyang Jeff Wu Xu Jiang Diogo Almeida Carroll\u00a0L. Wainwright Pamela Mishkin Chong Zhang Sandhini Agarwal Katarina Slama Alex Ray John Schulman Jacob Hilton Fraser Kelton Luke Miller Maddie Simens Amanda Askell Peter Welinder Paul Christiano Jan Leike and Ryan Lowe. 2022. Training language models to follow instructions with human feedback. arxiv:2203.02155\u00a0[cs.CL]"},{"key":"e_1_3_2_1_31_1","volume-title":"OptTyper: Probabilistic Type Inference by Optimising Logical and Natural Constraints. CoRR abs\/2004.00348","author":"Pandi Irene\u00a0Vlassi","year":"2020","unstructured":"Irene\u00a0Vlassi Pandi, Earl\u00a0T. Barr, Andrew\u00a0D. Gordon, and Charles Sutton. 2020. OptTyper: Probabilistic Type Inference by Optimising Logical and Natural Constraints. CoRR abs\/2004.00348 (2020). arXiv:2004.00348"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.3115\/1073083.1073135"},{"key":"e_1_3_2_1_33_1","volume-title":"Deep Learning for Code Workshop.","author":"Pashakhanloo Pardis","year":"2022","unstructured":"Pardis Pashakhanloo, Aaditya Naik, Hanjun Dai, Petros Maniatis, and Mayur Naik. 2022. Learning to Walk over Relational Graphs of Source Code. In Deep Learning for Code Workshop."},{"key":"e_1_3_2_1_34_1","volume-title":"International Conference on Learning Representations.","author":"Pashakhanloo Pardis","year":"2021","unstructured":"Pardis Pashakhanloo, Aaditya Naik, Yuepeng Wang, Hanjun Dai, Petros Maniatis, and Mayur Naik. 2021. CodeTrek: Flexible Modeling of Code using an Extensible Relational Representation. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/3510003.3510038"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3368089.3409715"},{"key":"e_1_3_2_1_37_1","unstructured":"[37] Query Suite. 2022. https:\/\/github.com\/github\/codeql\/blob\/main\/python\/ql\/src\/codeql-suites\/python-lgtm.qls."},{"key":"e_1_3_2_1_38_1","volume-title":"Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics. Association for Computational Linguistics.","author":"Rajpurkar Pranav","year":"2018","unstructured":"Pranav Rajpurkar, Robin Jia, and Percy Liang. 2018. Know What You Don\u2019t Know: Unanswerable Questions for SQuAD. In Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics. Association for Computational Linguistics."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D16-1264"},{"key":"e_1_3_2_1_40_1","volume-title":"Third Workshop on Very Large Corpora.","author":"A.","unstructured":"Lance\u00a0A. Ramshaw and Mitch Marcus. 1995. Text Chunking using Transformation-Based Learning. In Third Workshop on Very Large Corpora."},{"key":"e_1_3_2_1_41_1","volume-title":"Probabilistic model for code with decision trees. ACM SIGPLAN Notices 51, 10","author":"Raychev Veselin","year":"2016","unstructured":"Veselin Raychev, Pavol Bielik, and Martin Vechev. 2016. Probabilistic model for code with decision trees. ACM SIGPLAN Notices 51, 10 (2016)."},{"key":"e_1_3_2_1_42_1","volume-title":"The probabilistic relevance framework: BM25 and beyond. Foundations and Trends\u00ae in Information Retrieval 3, 4","author":"Robertson Stephen","year":"2009","unstructured":"Stephen Robertson, Hugo Zaragoza, 2009. The probabilistic relevance framework: BM25 and beyond. Foundations and Trends\u00ae in Information Retrieval 3, 4 (2009), 333\u2013389."},{"key":"e_1_3_2_1_43_1","unstructured":"Xujie Si Hanjun Dai Mukund Raghothaman Mayur Naik and Le Song. 2018. Learning Loop Invariants for Program Verification. In Advances in Neural Information Processing Systems."},{"key":"e_1_3_2_1_44_1","volume-title":"Proceedings of the International Conference on Machine Learning.","author":"Sutton Charles","year":"2023","unstructured":"Charles Sutton, David Bieber, Kensen Shi, Kexin Pei, and Pengcheng Yin. 2023. Can Large Language Models Reason About Program Invariants?. In Proceedings of the International Conference on Machine Learning."},{"key":"e_1_3_2_1_45_1","volume-title":"Llama: Open and efficient foundation language models. arXiv preprint arXiv:2302.13971","author":"Touvron Hugo","year":"2023","unstructured":"Hugo Touvron, Thibaut Lavril, Gautier Izacard, Xavier Martinet, Marie-Anne Lachaux, Timoth\u00e9e Lacroix, Baptiste Rozi\u00e8re, Naman Goyal, Eric Hambro, Faisal Azhar, 2023. Llama: Open and efficient foundation language models. arXiv preprint arXiv:2302.13971 (2023)."},{"key":"e_1_3_2_1_46_1","volume-title":"International Conference on Learning Representations. OpenReview.net.","author":"Wei Jiayi","year":"2020","unstructured":"Jiayi Wei, Maruth Goyal, Greg Durrett, and Isil Dillig. 2020. LambdaNet: Probabilistic Type Inference using Graph Neural Networks. In International Conference on Learning Representations. OpenReview.net."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/K17-1028"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1259"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/3178876.3186081"}],"event":{"name":"ISEC 2024: 17th Innovations in Software Engineering Conference","location":"Bangalore India","acronym":"ISEC 2024"},"container-title":["Proceedings of the 17th Innovations in Software Engineering Conference"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3641399.3641408","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3641399.3641408","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,23]],"date-time":"2025-08-23T02:10:37Z","timestamp":1755915037000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3641399.3641408"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,2,22]]},"references-count":49,"alternative-id":["10.1145\/3641399.3641408","10.1145\/3641399"],"URL":"https:\/\/doi.org\/10.1145\/3641399.3641408","relation":{},"subject":[],"published":{"date-parts":[[2024,2,22]]},"assertion":[{"value":"2024-02-22","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}