{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T20:17:42Z","timestamp":1783109862793,"version":"3.54.6"},"publisher-location":"Singapore","reference-count":38,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819543663","type":"print"},{"value":"9789819543670","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,11,22]],"date-time":"2025-11-22T00:00:00Z","timestamp":1763769600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,11,22]],"date-time":"2025-11-22T00:00:00Z","timestamp":1763769600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-981-95-4367-0_2","type":"book-chapter","created":{"date-parts":[[2025,11,21]],"date-time":"2025-11-21T12:37:17Z","timestamp":1763728637000},"page":"17-31","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["DFIR-Metric: A Benchmark Dataset for\u00a0Evaluating Large Language Models in\u00a0Digital Forensics and\u00a0Incident Response"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-0095-106X","authenticated-orcid":false,"given":"Bilel","family":"Cherif","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2626-3434","authenticated-orcid":false,"given":"Tamas","family":"Bisztray","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-3951-1932","authenticated-orcid":false,"given":"Richard A.","family":"Dubniczky","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-4075-5284","authenticated-orcid":false,"given":"Aaesha","family":"Aldahmani","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-3061-2400","authenticated-orcid":false,"given":"Saeed","family":"Alshehhi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9002-5935","authenticated-orcid":false,"given":"Norbert","family":"Tihanyi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,11,22]]},"reference":[{"key":"2_CR1","doi-asserted-by":"crossref","unstructured":"Alam, M.T., Bhusal, D., Nguyen, L., Rastogi, N.: Ctibench: a benchmark for evaluating LLMs in cyber threat intelligence. In: Advances in Neural Information Processing Systems 37 (NeurIPS 2024), Datasets and Benchmarks Track (2024)","DOI":"10.52202\/079017-1607"},{"key":"2_CR2","doi-asserted-by":"publisher","unstructured":"Barrington, S., Bohacek, M., Farid, H.: The DeepSpeak Dataset (2025). https:\/\/doi.org\/10.48550\/arXiv.2408.05366. arXiv:2408.05366 [cs] version: 3","DOI":"10.48550\/arXiv.2408.05366"},{"key":"2_CR3","doi-asserted-by":"crossref","unstructured":"Carrier, T., Victor, P., Tekeoglu, A., Lashkari, A.: Detecting obfuscated malware using memory feature engineering. In: Proceedings of the 8th International Conference on Information Systems Security and Privacy. SCITEPRESS - Science and Technology Publications (2022)","DOI":"10.5220\/0010908200003120"},{"issue":"1","key":"2_CR4","doi-asserted-by":"publisher","first-page":"3280","DOI":"10.1038\/s41467-025-56989-2","volume":"16","author":"Q Chen","year":"2025","unstructured":"Chen, Q., et al.: Benchmarking large language models for biomedical natural language processing applications and recommendations. Nat. Commun. 16(1), 3280 (2025)","journal-title":"Nat. Commun."},{"key":"2_CR5","doi-asserted-by":"publisher","unstructured":"Dang-Nguyen, D.T., Pasquini, C., Conotter, V., Boato, G.: RAISE: a raw images dataset for digital image forensics. In: Proceedings of the 6th ACM Multimedia Systems Conference, MMSys 2015, pp. 219\u2013224. Association for Computing Machinery, New York (2015). https:\/\/doi.org\/10.1145\/2713168.2713194","DOI":"10.1145\/2713168.2713194"},{"key":"2_CR6","doi-asserted-by":"publisher","unstructured":"Fei, Z., et al.: LawBench: benchmarking legal knowledge of large language models. In: Al-Onaizan, Y., Bansal, M., Chen, Y.N. (eds.) Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing, pp. 7933\u20137962. Association for Computational Linguistics, Miami, Florida, USA (2024). https:\/\/doi.org\/10.18653\/v1\/2024.emnlp-main.452","DOI":"10.18653\/v1\/2024.emnlp-main.452"},{"key":"2_CR7","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.iotcps.2025.01.001","volume":"5","author":"MA Ferrag","year":"2025","unstructured":"Ferrag, M.A., et al.: Generative AI in cybersecurity: a comprehensive review of LLM applications and vulnerabilities. Internet Things Cyber-Phys. Syst. 5, 1\u201346 (2025). https:\/\/doi.org\/10.1016\/j.iotcps.2025.01.001","journal-title":"Internet Things Cyber-Phys. Syst."},{"key":"2_CR8","doi-asserted-by":"publisher","first-page":"23733","DOI":"10.1109\/ACCESS.2024.3363469","volume":"12","author":"MA Ferrag","year":"2024","unstructured":"Ferrag, M.A., et al.: Revolutionizing cyber threat detection with large language models: a privacy-preserving BERT-based lightweight model for IoT\/IIoT devices. IEEE Access 12, 23733\u201323750 (2024). https:\/\/doi.org\/10.1109\/ACCESS.2024.3363469","journal-title":"IEEE Access"},{"key":"2_CR9","doi-asserted-by":"publisher","unstructured":"Glazer, E., et al.: FrontierMath: A Benchmark for Evaluating Advanced Mathematical Reasoning in AI (2024). https:\/\/doi.org\/10.48550\/arXiv.2411.04872. http:\/\/arxiv.org\/abs\/2411.04872. arXiv:2411.04872","DOI":"10.48550\/arXiv.2411.04872"},{"key":"2_CR10","doi-asserted-by":"publisher","DOI":"10.1016\/j.fsidi.2021.301264","volume":"38","author":"G Horsman","year":"2021","unstructured":"Horsman, G., Lyle, J.R.: Dataset construction challenges for digital forensics. Forensic Sci. Int. Digit. Invest. 38, 301264 (2021). https:\/\/doi.org\/10.1016\/j.fsidi.2021.301264","journal-title":"Forensic Sci. Int. Digit. Invest."},{"key":"2_CR11","unstructured":"Johansen, G.: Digital Forensics and Incident Response, 2nd edn. Packt Publishing, Birmingham (2020)"},{"key":"2_CR12","doi-asserted-by":"publisher","unstructured":"Joyce, R.J., Patel, T., Nicholas, C., Raff, E.: AVScan2Vec: feature learning on antivirus scan data for production-scale malware corpora. In: Proceedings of the 16th ACM Workshop on Artificial Intelligence and Security, AISec 2023, pp. 185\u2013196. Association for Computing Machinery, New York (2023). https:\/\/doi.org\/10.1145\/3605764.3623907","DOI":"10.1145\/3605764.3623907"},{"key":"2_CR13","doi-asserted-by":"publisher","unstructured":"Kent, K., Chevalier, S., Grance, T., Dang, H.: Guide to Integrating Forensic Techniques into Incident Response. Technical report, NIST Special Publication (SP) 800-86, National Institute of Standards and Technology (2006). https:\/\/doi.org\/10.6028\/NIST.SP.800-86","DOI":"10.6028\/NIST.SP.800-86"},{"issue":"3","key":"2_CR14","doi-asserted-by":"publisher","first-page":"162","DOI":"10.1109\/LNET.2022.3185553","volume":"4","author":"J Liu","year":"2022","unstructured":"Liu, J., et al.: A new realistic benchmark for advanced persistent threats in network traffic. IEEE Network. Lett. 4(3), 162\u2013166 (2022). https:\/\/doi.org\/10.1109\/LNET.2022.3185553","journal-title":"IEEE Network. Lett."},{"key":"2_CR15","doi-asserted-by":"publisher","unstructured":"Liu, J., Simsek, M., Kantarci, B., Bagheri, M., Djukic, P.: Collaborative feature maps of networks and hosts for AI-driven intrusion detection. In: GLOBECOM 2022 - 2022 IEEE Global Communications Conference, pp. 2662\u20132667 (2022). https:\/\/doi.org\/10.1109\/GLOBECOM48099.2022.10000985. ISSN: 2576-6813","DOI":"10.1109\/GLOBECOM48099.2022.10000985"},{"key":"2_CR16","doi-asserted-by":"publisher","unstructured":"Loumachi, F.Y., Ghanem, M.C., Ferrag, M.A.: Advancing cyber incident timeline analysis through retrieval-augmented generation and large language models. Computers 14(2), 67 (2025). https:\/\/doi.org\/10.3390\/computers14020067","DOI":"10.3390\/computers14020067"},{"key":"2_CR17","doi-asserted-by":"publisher","unstructured":"Michelet, G., Breitinger, F.: Chatgpt, llama, can you write my report? An experiment on assisted digital forensics reports written using (local) large language models. Forensic Sci. Int. Digit. Invest. 48, 301683 (2024). https:\/\/doi.org\/10.1016\/j.fsidi.2023.301683. dFRWS EU 2024 - Selected Papers from the 11th Annual Digital Forensics Research Conference Europe","DOI":"10.1016\/j.fsidi.2023.301683"},{"key":"2_CR18","doi-asserted-by":"publisher","DOI":"10.1016\/j.fsidi.2023.301683","volume":"48","author":"G Michelet","year":"2024","unstructured":"Michelet, G., Breitinger, F.: ChatGPT, Llama, can you write my report? An experiment on assisted digital forensics reports written using (local) large language models. Forensic Sci. Int. Digit. Invest. 48, 301683 (2024). https:\/\/doi.org\/10.1016\/j.fsidi.2023.301683","journal-title":"Forensic Sci. Int. Digit. Invest."},{"key":"2_CR19","doi-asserted-by":"publisher","DOI":"10.1016\/j.adhoc.2025.103840","volume":"174","author":"H Mohamed","year":"2025","unstructured":"Mohamed, H., Koroniotis, N., Schiliro, F., Moustafa, N.: IoT-CAD: a comprehensive digital forensics dataset for AI-based cyberattack attribution detection methods in IoT environments. Ad Hoc Netw. 174, 103840 (2025). https:\/\/doi.org\/10.1016\/j.adhoc.2025.103840","journal-title":"Ad Hoc Netw."},{"key":"2_CR20","doi-asserted-by":"publisher","DOI":"10.1016\/j.comnet.2023.109688","volume":"227","author":"S Myneni","year":"2023","unstructured":"Myneni, S., et al.: Unraveled - a semi-synthetic dataset for advanced persistent threats. Comput. Netw. 227, 109688 (2023). https:\/\/doi.org\/10.1016\/j.comnet.2023.109688","journal-title":"Comput. Netw."},{"key":"2_CR21","doi-asserted-by":"publisher","unstructured":"Nikolakopoulos, A., et al.: Large language models in modern forensic investigations: harnessing the power of generative artificial intelligence in crime resolution and suspect identification. In: 2024 5th International Conference in Electronic Engineering, Information Technology & Education (EEITE), pp.\u00a01\u20135 (2024). https:\/\/doi.org\/10.1109\/EEITE61750.2024.10654427","DOI":"10.1109\/EEITE61750.2024.10654427"},{"issue":"3","key":"2_CR22","doi-asserted-by":"publisher","first-page":"4","DOI":"10.1109\/MS.2023.3248401","volume":"40","author":"I Ozkaya","year":"2023","unstructured":"Ozkaya, I.: Application of large language models to software engineering tasks: opportunities, risks, and implications. IEEE Softw. 40(3), 4\u20138 (2023). https:\/\/doi.org\/10.1109\/MS.2023.3248401","journal-title":"IEEE Softw."},{"key":"2_CR23","doi-asserted-by":"crossref","unstructured":"Rajpurkar, P., Zhang, J., Lopyrev, K., Liang, P.: SQuAD: 100,000+ questions for machine comprehension of text. In: Proceedings of the 2016 Conference on Empirical Methods in Natural Language Processing. Association for Computational Linguistics, Stroudsburg, PA, USA (2016)","DOI":"10.18653\/v1\/D16-1264"},{"key":"2_CR24","doi-asserted-by":"publisher","DOI":"10.1016\/j.fsidi.2023.301609","volume":"46","author":"M Scanlon","year":"2023","unstructured":"Scanlon, M., Breitinger, F., Hargreaves, C., Hilgert, J.N., Sheppard, J.: ChatGPT for digital forensic investigation: the good, the bad, and the unknown. Forensic Sci. Int. Digit. Invest. 46, 301609 (2023). https:\/\/doi.org\/10.1016\/j.fsidi.2023.301609","journal-title":"Forensic Sci. Int. Digit. Invest."},{"key":"2_CR25","doi-asserted-by":"publisher","unstructured":"Sharma, B., Ghawaly, J., McCleary, K., Webb, A.M., Baggili, I.: Forensicllm: a local large language model for digital forensics. Forensic Sci. Int. Digit. Invest. 52, 301872 (2025). https:\/\/doi.org\/10.1016\/j.fsidi.2025.301872. dFRWS EU 2025 - Selected Papers from the 12th Annual Digital Forensics Research Conference Europe","DOI":"10.1016\/j.fsidi.2025.301872"},{"key":"2_CR26","doi-asserted-by":"publisher","DOI":"10.1016\/j.fsidi.2025.301872","volume":"52","author":"B Sharma","year":"2025","unstructured":"Sharma, B., Ghawaly, J., McCleary, K., Webb, A.M., Baggili, I.: ForensicLLM: a local large language model for digital forensics. Forensic Sci. Int. Digit. Invest. 52, 301872 (2025). https:\/\/doi.org\/10.1016\/j.fsidi.2025.301872","journal-title":"Forensic Sci. Int. Digit. Invest."},{"issue":"1","key":"2_CR27","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1186\/s13635-017-0067-2","volume":"2017","author":"D Shullani","year":"2017","unstructured":"Shullani, D., Fontani, M., Iuliani, M., Shaya, O.A., Piva, A.: VISION: a video and image dataset for source identification. EURASIP J. Inf. Secur. 2017(1), 1\u201316 (2017). https:\/\/doi.org\/10.1186\/s13635-017-0067-2","journal-title":"EURASIP J. Inf. Secur."},{"key":"2_CR28","doi-asserted-by":"publisher","DOI":"10.1016\/j.compeleceng.2025.110307","volume":"124","author":"AK Sood","year":"2025","unstructured":"Sood, A.K., Zeadally, S., Hong, E.: The paradigm of hallucinations in AI-driven cybersecurity systems: understanding taxonomy, classification outcomes, and mitigations. Comput. Electr. Eng. 124, 110307 (2025). https:\/\/doi.org\/10.1016\/j.compeleceng.2025.110307","journal-title":"Comput. Electr. Eng."},{"key":"2_CR29","doi-asserted-by":"publisher","unstructured":"Studiawan, H., Breitinger, F., Scanlon, M.: Towards a standardized methodology and dataset for evaluating LLM-based digital forensic timeline analysis (2025). https:\/\/doi.org\/10.48550\/arXiv.2505.03100","DOI":"10.48550\/arXiv.2505.03100"},{"key":"2_CR30","doi-asserted-by":"publisher","unstructured":"Tihanyi, N., et al.: Dynamic intelligence assessment: benchmarking LLMs on the road to AGI with a focus on model confidence. In: 2024 IEEE International Conference on Big Data (BigData), pp. 3313\u20133321 (2024). https:\/\/doi.org\/10.1109\/BigData62323.2024.10825051. ISSN: 2573-2978","DOI":"10.1109\/BigData62323.2024.10825051"},{"key":"2_CR31","doi-asserted-by":"publisher","unstructured":"Tihanyi, N., Ferrag, M.A., Jain, R., Bisztray, T., Debbah, M.: CyberMetric: a benchmark dataset based on retrieval-augmented generation for evaluating LLMs in cybersecurity knowledge. In: 2024 IEEE International Conference on Cyber Security and Resilience (CSR), pp. 296\u2013302 (2024). https:\/\/doi.org\/10.1109\/CSR61664.2024.10679494","DOI":"10.1109\/CSR61664.2024.10679494"},{"key":"2_CR32","doi-asserted-by":"crossref","unstructured":"Turing, A.M.: Computing machinery and intelligence (1950). In: Ideas That Created the Future, pp. 147\u2013164. The MIT Press (2021)","DOI":"10.7551\/mitpress\/12274.003.0016"},{"key":"2_CR33","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Proceedings of the 31st International Conference on Neural Information Processing Systems, NIPS 2017, pp. 6000\u20136010. Curran Associates Inc., Red Hook, NY, USA (2017)"},{"key":"2_CR34","doi-asserted-by":"publisher","unstructured":"Wang, A., Singh, A., Michael, J., Hill, F., Levy, O., Bowman, S.: GLUE: a multi-task benchmark and analysis platform for natural language understanding. In: Linzen, T., Chrupa\u0142a, G., Alishahi, A. (eds.) Proceedings of the 2018 EMNLP Workshop BlackboxNLP: Analyzing and Interpreting Neural Networks for NLP, pp. 353\u2013355. Association for Computational Linguistics, Brussels, Belgium (2018). https:\/\/doi.org\/10.18653\/v1\/W18-5446","DOI":"10.18653\/v1\/W18-5446"},{"key":"2_CR35","unstructured":"Wang, S., Long, Z., Fan, Z., Huang, X., Wei, Z.: Benchmark self-evolving: a multi-agent framework for dynamic LLM evaluation. In: Proceedings of the 31st International Conference on Computational Linguistics, pp. 3310\u20133328. Association for Computational Linguistics, Abu Dhabi, UAE (2025)"},{"key":"2_CR36","doi-asserted-by":"publisher","unstructured":"Wickramasekara, A., Densmore, A., Breitinger, F., Studiawan, H., Scanlon, M.: AutoDFBench: a framework for AI generated digital forensic code and tool testing and evaluation. In: Proceedings of the Digital Forensics Doctoral Symposium, pp.\u00a01\u20137. ACM, Brno Czech Republic (2025). https:\/\/doi.org\/10.1145\/3712716.3712718","DOI":"10.1145\/3712716.3712718"},{"key":"2_CR37","doi-asserted-by":"publisher","unstructured":"Wickramasekara, A., Scanlon, M.: A framework for integrated digital forensic investigation employing AutoGen AI agents. In: 2024 12th International Symposium on Digital Forensics and Security (ISDFS), pp. 1\u20136. IEEE, San Antonio, TX, USA (2024). https:\/\/doi.org\/10.1109\/ISDFS60797.2024.10527235","DOI":"10.1109\/ISDFS60797.2024.10527235"},{"key":"2_CR38","doi-asserted-by":"crossref","unstructured":"Yin, Z., et al.: Digital forensics in the age of large language models (2025). https:\/\/arxiv.org\/abs\/2504.02963","DOI":"10.1007\/978-3-031-98036-7_3"}],"container-title":["Lecture Notes in Computer Science","Neural Information Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-95-4367-0_2","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T19:32:48Z","timestamp":1783107168000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-95-4367-0_2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,22]]},"ISBN":["9789819543663","9789819543670"],"references-count":38,"URL":"https:\/\/doi.org\/10.1007\/978-981-95-4367-0_2","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,11,22]]},"assertion":[{"value":"22 November 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"The authors declare no competing interests relevant to the content of this article.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing Interests"}},{"value":"ICONIP","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Neural Information Processing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Okinawa","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Japan","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20 November 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"24 November 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"32","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"iconip2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/iconip2025.apnns.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}