{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T22:23:39Z","timestamp":1783635819272,"version":"3.55.0"},"reference-count":39,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100006579","name":"Ministry of Industry and Information Technology of the People's Republic of China","doi-asserted-by":"publisher","award":["CEIEC-2020-ZM02-0134"],"award-info":[{"award-number":["CEIEC-2020-ZM02-0134"]}],"id":[{"id":"10.13039\/501100006579","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Knowledge-Based Systems"],"published-print":{"date-parts":[[2026,8]]},"DOI":"10.1016\/j.knosys.2026.116311","type":"journal-article","created":{"date-parts":[[2026,5,22]],"date-time":"2026-05-22T23:36:13Z","timestamp":1779492973000},"page":"116311","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Stealing supervised fine tuning samples: A data extraction attack driven by token modulation and loss ratio"],"prefix":"10.1016","volume":"348","author":[{"given":"Jiawei","family":"Pi","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7729-5439","authenticated-orcid":false,"given":"Senlin","family":"Luo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Limin","family":"Pan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chengke","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ji","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.knosys.2026.116311_bib0001","article-title":"Language models are few-shot learners advances in neural information processing systems 33[J]","volume":"33","author":"Brown","year":"2020","journal-title":"Lang. Models Few-Shot Learn. Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.knosys.2026.116311_bib0002","unstructured":"Bai J, Bai S, Chu Y, et al. Qwen technical report[J]. arXiv preprint arXiv:2309.16609, 2023."},{"key":"10.1016\/j.knosys.2026.116311_bib0003","doi-asserted-by":"crossref","first-page":"786","DOI":"10.1145\/3720449","article-title":"Api-guided dataset synthesis to finetune large code models[J]","volume":"9","author":"Li","year":"2025","journal-title":"Proc. ACM Program. Lang."},{"key":"10.1016\/j.knosys.2026.116311_bib0004","series-title":"SemEval-2017 Task 1: Semantic Textual Similarity Multilingual and Cross-lingual Focused Evaluation. The 11th International Workshop on Semantic Evaluation (SemEval-2017)[C]","first-page":"1","author":"Cer","year":"2017"},{"key":"10.1016\/j.knosys.2026.116311_bib0005","unstructured":"Wei J, Bosma M, Zhao V Y, et al. Finetuned language models are zero-shot learners[EB\/OL]. (2021-09-03)[2025-12-17]. https:\/\/arxiv.org\/abs\/2109.01652."},{"key":"10.1016\/j.knosys.2026.116311_bib0006","unstructured":"Cirillo S, Desiato D, Scalera M, et al. A visual privacy tool to help users in preserving social network data[C]. IS-EUD Workshops. 2023."},{"issue":"1","key":"10.1016\/j.knosys.2026.116311_bib0007","doi-asserted-by":"crossref","first-page":"19","DOI":"10.1186\/s40537-022-00566-7","article-title":"Social network data analysis to highlight privacy threats in sharing data[J]","volume":"9","author":"Cerruto","year":"2022","journal-title":"J. Big Data"},{"key":"10.1016\/j.knosys.2026.116311_bib0008","series-title":"Proceedings of the 2020 ACM SIGSAC conference on computer and communications security[C]","first-page":"363","article-title":"Analyzing information leakage of updates to natural language models","author":"Zanella-B\u00e9guelin","year":"2020"},{"key":"10.1016\/j.knosys.2026.116311_bib0009","series-title":"USENIX security symposium (USENIX Security 21)","first-page":"2633","article-title":"Extracting training data from large language models[C]. 30th","author":"Carlini","year":"2021"},{"key":"10.1016\/j.knosys.2026.116311_bib0010","unstructured":"Nasr M, Carlini N, Hayase J, et al. Scalable extraction of training data from (production) language models[J]. arXiv preprint arXiv:2311.17035, 2023."},{"issue":"4","key":"10.1016\/j.knosys.2026.116311_bib0011","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3625097","article-title":"Malicious account identification in social network platforms[J]","volume":"23","author":"Caruccio","year":"2023","journal-title":"ACM Trans. Internet Technol."},{"key":"10.1016\/j.knosys.2026.116311_bib0012","series-title":"Proceedings of the 2019 conference of the North American chapter of the association for computational linguistics: human language technologies, volume 1 (long and short papers)[C]","first-page":"4171","article-title":"Bert: pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2019"},{"key":"10.1016\/j.knosys.2026.116311_bib0013","unstructured":"Lipton Z C, Berkowitz J, Elkan C. A critical review of recurrent neural networks for sequence learning[EB\/OL]. (2015-05-29)[2025-12-17]. https:\/\/arxiv.org\/abs\/1506.00019."},{"key":"10.1016\/j.knosys.2026.116311_bib0014","unstructured":"Nakamura Y, Hanaoka S, Nomura Y, et al. Kart: Privacy leakage framework of language models pre-trained with clinical records[EB\/OL]. (2020-10-10)[2025-12-17]. https:\/\/arxiv.org\/abs\/2101.00036."},{"key":"10.1016\/j.knosys.2026.116311_bib0015","series-title":"Proceedings of the 2020 ACM SIGSAC conference on computer and communications security[C]","first-page":"363","article-title":"Analyzing information leakage of updates to natural language models","author":"Zanella-B\u00e9guelin","year":"2020"},{"key":"10.1016\/j.knosys.2026.116311_bib0016","unstructured":"Inan HA, Ramadan O, Wutschitz L, et al. Training data leakage analysis in language models[EB\/OL]. (2021-01-14)[2025-12-17]. https:\/\/arxiv.org\/abs\/2101.05405."},{"key":"10.1016\/j.knosys.2026.116311_bib0017","author":"Chiang"},{"key":"10.1016\/j.knosys.2026.116311_bib0018","unstructured":"Li Z, Wang C, Ma P, et al. On the feasibility of specialized ability stealing for large language code models[EB\/OL]. (2023-03-03)[2025-12-17]. https:\/\/arxiv.org\/abs\/2303.03012."},{"key":"10.1016\/j.knosys.2026.116311_bib0019","series-title":"2023 IEEE Symposium on Security and Privacy (SP)[C]","first-page":"346","article-title":"Analyzing leakage of personally identifiable information in language models","author":"Lukas","year":"2023"},{"key":"10.1016\/j.knosys.2026.116311_bib0020","doi-asserted-by":"crossref","first-page":"20750","DOI":"10.52202\/075280-0911","article-title":"Propile: probing privacy leakage in large language models[J]","volume":"36","author":"Kim","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.knosys.2026.116311_bib0021","first-page":"1877","article-title":"Language models are few-shot learners[J]","volume":"33","author":"Brown","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.knosys.2026.116311_bib0022","article-title":"A survey of reinforcement learning from human feedback[J]","author":"Kaufmann","year":"2024","journal-title":"Trans. Mach. Learn. Res."},{"key":"10.1016\/j.knosys.2026.116311_bib0023","unstructured":"Nasr M, Rando J, Carlini N, et al. Scalable extraction of training data from aligned, production language models[EB\/OL]. (2023-11-17)[2025-12-17]. https:\/\/arxiv.org\/abs\/2311.17035."},{"key":"10.1016\/j.knosys.2026.116311_bib0024","unstructured":"Amodei D, Olah C, Steinhardt J, et al. Concrete problems in AI safety[EB\/OL]. (2016-06-21)[2025-12-17]. https:\/\/arxiv.org\/abs\/1606.06565."},{"key":"10.1016\/j.knosys.2026.116311_bib0025","first-page":"27730","article-title":"Training language models to follow instructions with human feedback[J]","volume":"35","author":"Ouyang","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.knosys.2026.116311_bib0026","unstructured":"Nakka K K, Frikha A, Mendes R, et al. PII-Compass: Guiding LLM training data extraction prompts towards the target PII via grounding[EB\/OL]. (2024-07-03)[2025-12-17]. https:\/\/arxiv.org\/abs\/2407.02943."},{"key":"10.1016\/j.knosys.2026.116311_bib0027","series-title":"Proceedings of the 2025 ACM SIGSAC Conference on Computer and Communications Security[C]","first-page":"3071","article-title":"Differentiation-based extraction of proprietary data from fine-tuned llms","author":"Li","year":"2025"},{"issue":"S1","key":"10.1016\/j.knosys.2026.116311_bib0028","doi-asserted-by":"crossref","DOI":"10.1121\/1.2016299","article-title":"Perplexity\u2014a measure of the difficulty of speech recognition tasks[J]","volume":"62","author":"Jelinek","year":"1977","journal-title":"J. Acoust. Soc. Am."},{"key":"10.1016\/j.knosys.2026.116311_bib0029","series-title":"The Tenth International Conference on Learning Representations. Virtual Event","first-page":"3","article-title":"Lora: low-rank adaptation of large language models[J]","volume":"1","author":"Hu","year":"2022"},{"key":"10.1016\/j.knosys.2026.116311_bib0030","doi-asserted-by":"crossref","first-page":"10088","DOI":"10.52202\/075280-0441","article-title":"Qlora: efficient finetuning of quantized llms[J]","volume":"36","author":"Dettmers","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.knosys.2026.116311_bib0031","first-page":"1137","article-title":"A neural probabilistic language model[J]","volume":"3","author":"Bengio","year":"2003","journal-title":"J. Mach. Learn. Res."},{"issue":"1","key":"10.1016\/j.knosys.2026.116311_bib0032","doi-asserted-by":"crossref","first-page":"2","DOI":"10.3390\/technologies9010002","article-title":"A survey on contrastive self-supervised learning[J]","volume":"9","author":"Jaiswal","year":"2020","journal-title":"Technologies"},{"key":"10.1016\/j.knosys.2026.116311_bib0033","unstructured":"Wei Y, Wang Z, Liu J, et al. Magicoder: Empowering code generation with oss-instruct[J]. arXiv preprint arXiv:2312.02120, 2023."},{"key":"10.1016\/j.knosys.2026.116311_bib0034","unstructured":"Yue X, Qu X, Zhang G, et al. Mammoth: Building math generalist models through hybrid instruction tuning[J]. arXiv preprint arXiv:2309.05653, 2023."},{"key":"10.1016\/j.knosys.2026.116311_bib0035","series-title":"Proceedings of the 2018 conference on empirical methods in natural language processing","first-page":"3911","article-title":"Spider: a large-scale human-labeled dataset for complex and cross-domain semantic parsing and text-to-sql task[C]","author":"Yu","year":"2018"},{"key":"10.1016\/j.knosys.2026.116311_bib0036","unstructured":"Grattafiori A, Dubey A, Jauhri A, et al. The llama 3 herd of models[J]. arXiv preprint arXiv:2407.21783, 2024."},{"key":"10.1016\/j.knosys.2026.116311_bib0037","article-title":"SmolLM3: smol, multilingual, long-context reasoner[J]","author":"Bakouch","year":"2025","journal-title":"Hugging Face Blog"},{"key":"10.1016\/j.knosys.2026.116311_bib0038","unstructured":"Ding L. AdaRubric: task-adaptive rubrics for LLM agent evaluation[J]. arXiv preprint arXiv:2603.21362, 2026."},{"key":"10.1016\/j.knosys.2026.116311_bib0039","series-title":"The Eleventh International Conference on Learning Representations","article-title":"Quantifying memorization across neural language models[C]","author":"Carlini","year":"2022"}],"container-title":["Knowledge-Based Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0950705126010373?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0950705126010373?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T22:07:18Z","timestamp":1783634838000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0950705126010373"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8]]},"references-count":39,"alternative-id":["S0950705126010373"],"URL":"https:\/\/doi.org\/10.1016\/j.knosys.2026.116311","relation":{},"ISSN":["0950-7051"],"issn-type":[{"value":"0950-7051","type":"print"}],"subject":[],"published":{"date-parts":[[2026,8]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Stealing supervised fine tuning samples: A data extraction attack driven by token modulation and loss ratio","name":"articletitle","label":"Article Title"},{"value":"Knowledge-Based Systems","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.knosys.2026.116311","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Published by Elsevier B.V.","name":"copyright","label":"Copyright"}],"article-number":"116311"}}