{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,4]],"date-time":"2026-06-04T01:09:24Z","timestamp":1780535364829,"version":"3.54.1"},"reference-count":44,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2025,3,28]],"date-time":"2025-03-28T00:00:00Z","timestamp":1743120000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,3,28]],"date-time":"2025-03-28T00:00:00Z","timestamp":1743120000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"Natural Science Foundation of China","doi-asserted-by":"crossref","award":["62192731"],"award-info":[{"award-number":["62192731"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100001809","name":"Natural Science Foundation of China","doi-asserted-by":"crossref","award":["62192730"],"award-info":[{"award-number":["62192730"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"name":"National Key R& D Program","award":["2023YFB4503801"],"award-info":[{"award-number":["2023YFB4503801"]}]},{"DOI":"10.13039\/501100001809","name":"Natural Science Foundation of China","doi-asserted-by":"crossref","award":["62072007"],"award-info":[{"award-number":["62072007"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100001809","name":"Natural Science Foundation of China","doi-asserted-by":"crossref","award":["62192733"],"award-info":[{"award-number":["62192733"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100001809","name":"Natural Science Foundation of China","doi-asserted-by":"crossref","award":["61832009"],"award-info":[{"award-number":["61832009"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100001809","name":"Natural Science Foundation of China","doi-asserted-by":"crossref","award":["62192730"],"award-info":[{"award-number":["62192730"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"name":"Key Program of Hubei","award":["JD2023008"],"award-info":[{"award-number":["JD2023008"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Empir Software Eng"],"published-print":{"date-parts":[[2025,5]]},"DOI":"10.1007\/s10664-024-10603-z","type":"journal-article","created":{"date-parts":[[2025,3,30]],"date-time":"2025-03-30T22:54:30Z","timestamp":1743375270000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["SCodeSearcher: soft contrastive learning for code search"],"prefix":"10.1007","volume":"30","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9411-971X","authenticated-orcid":false,"given":"Jia","family":"Li","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zheng","family":"Fang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xianjie","family":"Shi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhi","family":"Jin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fang","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jia","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yunfei","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ge","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,3,28]]},"reference":[{"issue":"1","key":"10603_CR1","doi-asserted-by":"publisher","first-page":"45","DOI":"10.1016\/S0306-4573(02)00021-3","volume":"39","author":"A Aizawa","year":"2003","unstructured":"Aizawa A (2003) An information-theoretic perspective of tf-idf measures. Inf Process Manag 39(1):45\u201365","journal-title":"Inf Process Manag"},{"key":"10603_CR2","doi-asserted-by":"crossref","unstructured":"Bui ND, Yu Y, Jiang L (2021) Self-supervised contrastive learning for code retrieval and summarization via semantic-preserving transformations. In: Proceedings of the 44th international ACM SIGIR conference on research and development in information retrieval, pp 511\u2013521","DOI":"10.1145\/3404835.3462840"},{"key":"10603_CR3","doi-asserted-by":"crossref","unstructured":"Cambronero J, Li H, Kim S, Sen K, Chandra S (2019) When deep learning met code search. In: Proceedings of the 2019 27th ACM joint meeting on european software engineering conference and symposium on the foundations of software engineering, pp 964\u2013974","DOI":"10.1145\/3338906.3340458"},{"key":"10603_CR4","unstructured":"ChatGPT (2022) https:\/\/platform.openai.com\/docs\/models\/gpt-3-5"},{"key":"10603_CR5","unstructured":"Chen M, Tworek J, Jun H, Yuan Q, de\u00a0Oliveira\u00a0Pinto HP, Kaplan J, Edwards H, Burda Y, Joseph N, Brockman G, Ray A, Puri R, Krueger G, Petrov M, Khlaaf H, Sastry G, Mishkin P, Chan B, Gray S, Ryder N, Pavlov M, Power A, Kaiser L, Bavarian M, Winter C, Tillet P, Such FP, Cummings D, Plappert M, Chantzis F, Barnes E, Herbert-Voss A, Guss WH, Nichol A, Paino A, Tezak N, Tang J, Babuschkin I, Balaji S, Jain S, Saunders W, Hesse C, Carr AN, Leike J, Achiam J, Misra V, Morikawa E, Radford A, Knight M, Brundage M, Murati M, Mayer K, Welinder P, McGrew B, Amodei D, McCandlish S, Sutskever I, Zaremba W (2021) Evaluating large language models trained on code. arXiv:2107.03374"},{"key":"10603_CR6","unstructured":"Chen T, Kornblith S, Norouzi M, Hinton G (2020) A simple framework for contrastive learning of visual representations. In: International conference on machine learning, PMLR, pp 1597\u20131607"},{"key":"10603_CR7","unstructured":"Devlin J, Chang MW, Lee K, Toutanova K (2018) Bert: pre-training of deep bidirectional transformers for language understanding. arXiv:1810.04805"},{"key":"10603_CR8","unstructured":"Ding Y, Buratti L, Pujar S, Morari A, Ray B, Chakraborty S (2021) Contrastive learning for source code with structural and functional properties. arXiv:2110.03868"},{"key":"10603_CR9","doi-asserted-by":"crossref","unstructured":"Du L, Shi X, Wang Y, Shi E, Han S, Zhang D (2021) Is a single model enough? mucos: A multi-model ensemble learning approach for semantic code search. In: Proceedings of the 30th ACM international conference on information & knowledge management, pp 2994\u20132998","DOI":"10.1145\/3459637.3482127"},{"key":"10603_CR10","doi-asserted-by":"crossref","unstructured":"Feng Z, Guo D, Tang D, Duan N, Feng X, Gong M, Shou L, Qin B, Liu T, Jiang D, et al (2020) Codebert: a pre-trained model for programming and natural languages. arXiv:2002.08155","DOI":"10.18653\/v1\/2020.findings-emnlp.139"},{"key":"10603_CR11","doi-asserted-by":"crossref","unstructured":"Gao L, Ma X, Lin J, Callan J (2022) Precise zero-shot dense retrieval without relevance labels. arXiv:2212.10496","DOI":"10.18653\/v1\/2023.acl-long.99"},{"key":"10603_CR12","unstructured":"GPT-3.5 (2022). https:\/\/platform.openai.com\/docs\/deprecations"},{"key":"10603_CR13","unstructured":"GPT-4: (2023)"},{"key":"10603_CR14","doi-asserted-by":"crossref","unstructured":"Gu X, Zhang H, Kim S (2018) Deep code search. In: 2018 IEEE\/ACM 40th International Conference on Software Engineering (ICSE), IEEE, pp 933\u2013944","DOI":"10.1145\/3180155.3180167"},{"key":"10603_CR15","doi-asserted-by":"crossref","unstructured":"Guo D, Lu S, Duan N, Wang Y, Zhou M, Yin J (2022) Unixcoder: unified cross-modal pre-training for code representation. arXiv:2203.03850","DOI":"10.18653\/v1\/2022.acl-long.499"},{"key":"10603_CR16","unstructured":"Guo D, Ren S, Lu S, Feng Z, Tang D, Liu S, Zhou L, Duan N, Svyatkovskiy A, Fu S, et\u00a0al (2020) Graphcodebert: pre-training code representations with data flow. arXiv:2009.08366"},{"key":"10603_CR17","doi-asserted-by":"crossref","unstructured":"Huang J, Tang D, Shou L, Gong M, Xu K, Jiang D, Zhou M, Duan N (2021) Cosqa: 20,000+ web queries for code search and question answering. arXiv:2105.13239","DOI":"10.18653\/v1\/2021.acl-long.442"},{"key":"10603_CR18","unstructured":"Husain H, Wu HH, Gazit T, Allamanis M, Brockschmidt M (2019) Codesearchnet challenge: evaluating the state of semantic code search. arXiv:1909.09436"},{"issue":"1","key":"10603_CR19","doi-asserted-by":"publisher","first-page":"103539","DOI":"10.1016\/j.ipm.2023.103539","volume":"61","author":"Z Jian","year":"2024","unstructured":"Jian Z, Li J, Wu Q, Yao J (2024) Retrieval contrastive learning for aspect-level sentiment classification. Inf Process Manag 61(1):103539","journal-title":"Inf Process Manag"},{"key":"10603_CR20","unstructured":"Li D, Shen Y, Jin R, Mao Y, Wang K, Chen W (2022) Generation-augmented query expansion for code retrieval. arXiv:2212.10692"},{"key":"10603_CR21","doi-asserted-by":"crossref","unstructured":"Li H, Zhou X, Shen Z (2024) Rewriting the code: A simple method for large language model augmented code search. arXiv:2401.04514","DOI":"10.18653\/v1\/2024.acl-long.75"},{"key":"10603_CR22","doi-asserted-by":"crossref","unstructured":"Li J, Liu F, Li J, Zhao Y, Li G, Jin Z (2023) Mcodesearcher: Multi-view contrastive learning for code search. In: Proceedings of the 14th Asia-pacific symposium on internetware, pp 270\u2013280","DOI":"10.1145\/3609437.3609456"},{"key":"10603_CR23","unstructured":"Li X, Gong Y, Shen Y, Qiu X, Zhang H, Yao B, Qi W, Jiang D, Chen W, Duan N (2022) Coderetriever: unimodal and bimodal contrastive learning. arXiv:2201.10866"},{"issue":"6624","key":"10603_CR24","doi-asserted-by":"publisher","first-page":"1092","DOI":"10.1126\/science.abq1158","volume":"378","author":"Y Li","year":"2022","unstructured":"Li Y, Choi D, Chung J, Kushman N, Schrittwieser J, Leblond R, Eccles T, Keeling J, Gimeno F, Dal Lago A et al (2022) Competition-level code generation with alphacode. Science 378(6624):1092\u20131097","journal-title":"Science"},{"issue":"2","key":"10603_CR25","doi-asserted-by":"publisher","first-page":"300","DOI":"10.1007\/s10618-008-0118-x","volume":"18","author":"E Linstead","year":"2009","unstructured":"Linstead E, Bajracharya S, Ngo T, Rigor P, Lopes C, Baldi P (2009) Sourcerer: mining and searching internet-scale software repositories. Data Min Knowl Disc 18(2):300\u2013336","journal-title":"Data Min Knowl Disc"},{"key":"10603_CR26","unstructured":"Liu Y, Ott M, Goyal N, Du J, Joshi M, Chen D, Levy O, Lewis M, Zettlemoyer L, Stoyanov V (2019) Roberta: a robustly optimized BERT pretraining approach. arXiv:1907.11692."},{"key":"10603_CR27","unstructured":"Loshchilov I, Hutter F (2017) Decoupled weight decay regularization. arXiv:1711.05101"},{"key":"10603_CR28","doi-asserted-by":"crossref","unstructured":"Lu M, Sun X, Wang S, Lo D, Duan Y (2015) Query expansion via wordnet for effective code search. In: 2015 IEEE 22nd international conference on software analysis, evolution, and reengineering (SANER), pp 545\u2013549. IEEE","DOI":"10.1109\/SANER.2015.7081874"},{"key":"10603_CR29","unstructured":"Lu S, Guo D, Ren S, Huang J, Svyatkovskiy A, Blanco A, Clement C, Drain D, Jiang D, Tang D, et\u00a0al (2021) Codexglue: a machine learning benchmark dataset for code understanding and generation. arXiv:2102.04664"},{"key":"10603_CR30","doi-asserted-by":"crossref","unstructured":"Lv F, Zhang H, Lou Jg, Wang S, Zhang D, Zhao J (2015) Codehow: effective code search based on api understanding and extended boolean model (e). In: 2015 30th IEEE\/ACM international conference on automated software engineering (ASE), pp 260\u2013270. IEEE","DOI":"10.1109\/ASE.2015.42"},{"key":"10603_CR31","doi-asserted-by":"crossref","unstructured":"Mao Y, He P, Liu X, Shen Y, Gao J, Han J, Chen W (2020) Generation-augmented retrieval for open-domain question answering. arXiv:2009.08553","DOI":"10.18653\/v1\/2021.acl-long.316"},{"key":"10603_CR32","doi-asserted-by":"crossref","unstructured":"Niu C, Li C, Ng V, Ge J, Huang L, Luo B (2022) Spt-code: sequence-to-sequence pre-training for learning the representation of source code. arXiv:2201.01549","DOI":"10.1145\/3510003.3510096"},{"key":"10603_CR33","doi-asserted-by":"crossref","unstructured":"Qin Z, Jagerman R, Hui K, Zhuang H, Wu J, Shen J, Liu T, Liu J, Metzler D, Wang X, et\u00a0al (2023) Large language models are effective text rankers with pairwise ranking prompting. arXiv:2306.17563","DOI":"10.18653\/v1\/2024.findings-naacl.97"},{"key":"10603_CR34","doi-asserted-by":"crossref","unstructured":"Sennrich R, Haddow B, Birch A (2015) Neural machine translation of rare words with subword units. arXiv:1508.07909","DOI":"10.18653\/v1\/P16-1162"},{"key":"10603_CR35","unstructured":"Shi E, Gub W, Wang Y, Du L, Zhang H, Han S, Zhang D, Sun H (2022) Enhancing semantic code search with multimodal contrastive learning and soft data augmentation. arXiv:2204.03293"},{"key":"10603_CR36","doi-asserted-by":"crossref","unstructured":"Shuai J, Xu L, Liu C, Yan M, Xia X, Lei Y (2020) Improving code search with co-attentive representation learning. In: Proceedings of the 28th international conference on program comprehension, pp 196\u2013207","DOI":"10.1145\/3387904.3389269"},{"key":"10603_CR37","unstructured":"THUDM: Codegeex (2022) https:\/\/github.com\/THUDM\/CodeGeeX"},{"key":"10603_CR38","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser L, Polosukhin I (2017) Attention is all you need. In: I.\u00a0Guyon, U.\u00a0von Luxburg, S.\u00a0Bengio, H.M. Wallach, R.\u00a0Fergus, S.V.N. Vishwanathan, R.\u00a0Garnett (Eds.), Advances in neural information processing systems 30: annual conference on neural information processing systems 2017, December 4-9, 2017, Long Beach, CA, USA, pp 5998\u20136008. https:\/\/proceedings.neurips.cc\/paper\/2017\/hash\/3f5ee243547dee91fbd053c1c4a845aa-Abstract.html"},{"key":"10603_CR39","doi-asserted-by":"crossref","unstructured":"Wan Y, Shu J, Sui Y, Xu G, Zhao Z, Wu J, Yu P (2019) Multi-modal attention network learning for semantic source code retrieval. In: 2019 34th IEEE\/ACM international conference on automated software engineering (ASE), IEEE, pp 13\u201325.","DOI":"10.1109\/ASE.2019.00012"},{"key":"10603_CR40","doi-asserted-by":"crossref","unstructured":"Wang L, Yang N, Wei F (2023) Query2doc: query expansion with large language models. arXiv:2303.07678","DOI":"10.18653\/v1\/2023.emnlp-main.585"},{"key":"10603_CR41","unstructured":"Wang X, Wang Y Mi F, Zhou P, Wan Y, Liu X, Li L, Wu H, Liu J, Jiang X (2021) Syncobert: syntax-guided multi-modal contrastive pre-training for code representation. arXiv:2108.04556"},{"key":"10603_CR42","doi-asserted-by":"crossref","unstructured":"Wang Y, Wang W, Joty S, Hoi SC (2021) Codet5: identifier-aware unified pre-trained encoder-decoder models for code understanding and generation. arXiv:2109.00859","DOI":"10.18653\/v1\/2021.emnlp-main.685"},{"key":"10603_CR43","unstructured":"Wu B, Zhang Z, Wang J, Zhao H (2021) Sentence-aware contrastive learning for open-domain passage retrieval. arXiv:2110.07524"},{"key":"10603_CR44","doi-asserted-by":"crossref","unstructured":"Yao Z, Weld DS, Chen WP, Sun H (2018) Staqc: a systematically mined question-code dataset from stack overflow. In: Proceedings of the 2018 World Wide Web Conference, pp 1693\u20131703","DOI":"10.1145\/3178876.3186081"}],"container-title":["Empirical Software Engineering"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10664-024-10603-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10664-024-10603-z\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10664-024-10603-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,20]],"date-time":"2025-11-20T13:28:27Z","timestamp":1763645307000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10664-024-10603-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,3,28]]},"references-count":44,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2025,5]]}},"alternative-id":["10603"],"URL":"https:\/\/doi.org\/10.1007\/s10664-024-10603-z","relation":{},"ISSN":["1382-3256","1573-7616"],"issn-type":[{"value":"1382-3256","type":"print"},{"value":"1573-7616","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,3,28]]},"assertion":[{"value":"8 December 2024","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 March 2025","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"All authors declared that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflicts of Interest"}}],"article-number":"87"}}