{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T15:00:14Z","timestamp":1784300414131,"version":"3.55.0"},"reference-count":111,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2026,3,27]],"date-time":"2026-03-27T00:00:00Z","timestamp":1774569600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0"},{"start":{"date-parts":[[2026,3,27]],"date-time":"2026-03-27T00:00:00Z","timestamp":1774569600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0"}],"funder":[{"DOI":"10.13039\/501100005713","name":"Technische Universit\u00e4t M\u00fcnchen","doi-asserted-by":"crossref","id":[{"id":"10.13039\/501100005713","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Empir Software Eng"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1007\/s10664-025-10792-1","type":"journal-article","created":{"date-parts":[[2026,3,27]],"date-time":"2026-03-27T14:18:44Z","timestamp":1774621124000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["XMutant: XAI-based fuzzing for deep learning systems"],"prefix":"10.1007","volume":"31","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-0861-4093","authenticated-orcid":false,"given":"Xingcheng","family":"Chen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7825-3409","authenticated-orcid":false,"given":"Matteo","family":"Biagiola","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6229-8231","authenticated-orcid":false,"given":"Vincenzo","family":"Riccio","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1323-8769","authenticated-orcid":false,"given":"Marcelo","family":"d\u2019Amorim","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8956-3894","authenticated-orcid":false,"given":"Andrea","family":"Stocco","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,3,27]]},"reference":[{"key":"10792_CR1","doi-asserted-by":"publisher","unstructured":"Abdessalem RB, Panichella A, Nejati S, Briand LC, Stifter T (2018) Testing autonomous cars for feature interaction failures using many-objective search. In: Proceedings of the 33rd ACM\/IEEE International Conference on Automated Software Engineering, ASE 2018, pp. 143\u2013154. ACM, New York, NY, USA. https:\/\/doi.org\/10.1145\/3238147.3238192","DOI":"10.1145\/3238147.3238192"},{"key":"10792_CR2","unstructured":"Adebayo J, Gilmer J, Muelly M, Goodfellow I, Hardt M, Kim B (2018) Sanity checks for saliency maps. Advances in Neural Information Processing Systems 31"},{"issue":"2","key":"10792_CR3","doi-asserted-by":"publisher","first-page":"47","DOI":"10.1007\/s10664-023-10433-5","volume":"29","author":"MH Amini","year":"2024","unstructured":"Amini MH, Naseri S, Nejati S (2024) Evaluating the impact of flaky simulators on testing autonomous driving systems. Empirical Softw Engg 29(2):47. https:\/\/doi.org\/10.1007\/s10664-023-10433-5","journal-title":"Empirical Softw Engg"},{"key":"10792_CR4","doi-asserted-by":"crossref","unstructured":"Baresi L, Hu DYX, Stocco A, Tonella P (2025) Efficient domain augmentation for autonomous driving testing using diffusion models. In: Proceedings of 47th International Conference on Software Engineering, ICSE \u201925. IEEE","DOI":"10.1109\/ICSE55347.2025.00206"},{"key":"10792_CR5","doi-asserted-by":"crossref","unstructured":"Ben Abdessalem R, Nejati S, Briand LC, Stifter T (2016) Testing advanced driver assistance systems using multi-objective search and neural networks. In: 2016 31st IEEE\/ACM International Conference on Automated Software Engineering (ASE), pp. 63\u201374","DOI":"10.1145\/2970276.2970311"},{"key":"10792_CR6","doi-asserted-by":"publisher","unstructured":"Ben Abdessalem R, Nejati S, C Briand L, Stifter T (2018) Testing vision-based control systems using learnable evolutionary algorithms. In: 2018 IEEE\/ACM 40th International Conference on Software Engineering (ICSE), pp. 1016\u20131026. https:\/\/doi.org\/10.1145\/3180155.3180160","DOI":"10.1145\/3180155.3180160"},{"issue":"4","key":"10792_CR7","doi-asserted-by":"publisher","first-page":"72","DOI":"10.1007\/s10664-024-10458-4","volume":"29","author":"M Biagiola","year":"2024","unstructured":"Biagiola M, Stocco A, Riccio V, Tonella P (2024) Two is better than one: Digital siblings to improve autonomous driving testing. Empir Softw Eng 29(4):72","journal-title":"Empir Softw Eng"},{"key":"10792_CR8","doi-asserted-by":"publisher","unstructured":"Bojarski M, Choromanska A, Choromanski K, Firner B, Jackel L, Muller U, Zieba K (2016) Visualbackprop: efficient visualization of cnns. https:\/\/doi.org\/10.48550\/ARXIV.1611.05418. https:\/\/arxiv.org\/abs\/1611.05418","DOI":"10.48550\/ARXIV.1611.05418"},{"key":"10792_CR9","unstructured":"Bojarski M, Testa DD, Dworakowski D, Firner B, Flepp B, Goyal P, Jackel LD, Monfort M, Muller U, Zhang J, Zhang X, Zhao J, Zieba K (2016) End to end learning for self-driving cars. CoRR abs\/1604.07316. http:\/\/arxiv.org\/abs\/1604.07316"},{"key":"10792_CR10","doi-asserted-by":"crossref","unstructured":"Catmull E, Rom R (1974) A class of local interpolating splines. In: Computer aided geometric design, pp. 317\u2013326. Elsevier","DOI":"10.1016\/B978-0-12-079050-0.50020-5"},{"key":"10792_CR11","doi-asserted-by":"publisher","unstructured":"Chattopadhay A, Sarkar A, Howlader P, Balasubramanian VN (2018) Grad-cam++: Generalized gradient-based visual explanations for deep convolutional networks. In: 2018 IEEE Winter Conference on Applications of Computer Vision (WACV). IEEE. https:\/\/doi.org\/10.1109\/wacv.2018.00097","DOI":"10.1109\/wacv.2018.00097"},{"key":"10792_CR12","doi-asserted-by":"publisher","unstructured":"Cheng M, Zhou Y, Xie X (2023) Behavexplor: Behavior diversity guided testing for autonomous driving systems. In: Proceedings of the 32nd ACM SIGSOFT International Symposium on Software Testing and Analysis, ISSTA 2023, p. 488\u2013500. Association for Computing Machinery, New York, NY, USA. https:\/\/doi.org\/10.1145\/3597926.3598072","DOI":"10.1145\/3597926.3598072"},{"issue":"7","key":"10792_CR13","doi-asserted-by":"publisher","first-page":"437","DOI":"10.1109\/32.605761","volume":"23","author":"D Cohen","year":"1997","unstructured":"Cohen D, Dalal S, Fredman M, Patton G (1997) The aetg system: an approach to testing based on combinatorial design. IEEE Trans Software Eng 23(7):437\u2013444. https:\/\/doi.org\/10.1109\/32.605761","journal-title":"IEEE Trans Software Eng"},{"key":"10792_CR14","volume-title":"Statistical power analysis for the behavioral sciences","author":"J Cohen","year":"1988","unstructured":"Cohen J (1988) Statistical power analysis for the behavioral sciences. L. Erlbaum Associates, Hillsdale, N.J"},{"issue":"6","key":"10792_CR15","doi-asserted-by":"publisher","first-page":"141","DOI":"10.1109\/MSP.2012.2211477","volume":"29","author":"L Deng","year":"2012","unstructured":"Deng L (2012) The mnist database of handwritten digit images for machine learning research. IEEE Signal Process Mag 29(6):141\u2013142","journal-title":"IEEE Signal Process Mag"},{"key":"10792_CR16","unstructured":"Doersch C (2016) Tutorial on variational autoencoders. arXiv preprint arXiv:1606.05908"},{"key":"10792_CR17","doi-asserted-by":"crossref","unstructured":"Dola S, McDaniel R, Dwyer MB, Soffa ML (2024) Cit4dnn: Generating diverse and rare inputs for neural networks using latent space combinatorial testing. In: Proceedings of the IEEE\/ACM 46th International Conference on Software Engineering, pp. 1\u201313","DOI":"10.1145\/3597503.3639106"},{"key":"10792_CR18","doi-asserted-by":"crossref","unstructured":"Do\u0161ilovi\u0107 FK, Br\u010di\u0107 M, Hlupi\u0107 N (2018) Explainable artificial intelligence: A survey. In: 2018 41st International convention on information and communication technology, electronics and microelectronics (MIPRO), pp. 0210\u20130215. IEEE","DOI":"10.23919\/MIPRO.2018.8400040"},{"issue":"1","key":"10792_CR19","doi-asserted-by":"publisher","first-page":"68","DOI":"10.1145\/3359786","volume":"63","author":"M Du","year":"2019","unstructured":"Du M, Liu N, Hu X (2019) Techniques for interpretable machine learning. Commun ACM 63(1):68\u201377","journal-title":"Commun ACM"},{"key":"10792_CR20","unstructured":"Dunn I, Melham T, Kroening D (2020) Semantic adversarial perturbations using learnt representations. CoRR abs\/2001.11055. https:\/\/arxiv.org\/abs\/2001.11055"},{"key":"10792_CR21","doi-asserted-by":"crossref","unstructured":"Dunn I, Pouget H, Kroening D, Melham T (2021) Exposing previously undetectable faults in deep neural networks. In: Proceedings of the 30th ACM SIGSOFT International Symposium on Software Testing and Analysis, pp. 56\u201366","DOI":"10.1145\/3460319.3464801"},{"key":"10792_CR22","doi-asserted-by":"publisher","unstructured":"Fahmy H, Pastore F, Briand L, Stifter T (2023) Simulator-based explanation and debugging of hazard-triggering events in dnn-based safety-critical systems. ACM Transactions on Software Engineering and Methodology 32(4):1\u201347. https:\/\/doi.org\/10.48550\/ARXIV.2204.00480. https:\/\/arxiv.org\/abs\/2204.00480","DOI":"10.48550\/ARXIV.2204.00480"},{"key":"10792_CR23","unstructured":"Fahmy HM, Bagherzadeh M, Pastore F, Briand LC (2020) Supporting DNN safety analysis and retraining through heatmap-based unsupervised learning. CoRR abs\/2002.00863. https:\/\/arxiv.org\/abs\/2002.00863"},{"issue":"2","key":"10792_CR24","doi-asserted-by":"publisher","first-page":"73","DOI":"10.1016\/0010-4485(83)90171-9","volume":"15","author":"G Farin","year":"1983","unstructured":"Farin G (1983) Algorithms for rational b\u00e9zier curves. Comput Aided Des 15(2):73\u201377","journal-title":"Comput Aided Des"},{"issue":"5","key":"10792_CR25","doi-asserted-by":"publisher","first-page":"378","DOI":"10.1037\/h0031619","volume":"76","author":"JL Fleiss","year":"1971","unstructured":"Fleiss JL (1971) Measuring nominal scale agreement among many raters. Psychol Bull 76(5):378","journal-title":"Psychol Bull"},{"key":"10792_CR26","doi-asserted-by":"publisher","unstructured":"Fong RC, Vedaldi A (2017) Interpretable explanations of black boxes by meaningful perturbation. In: 2017 IEEE International Conference on Computer Vision (ICCV), pp. 3449\u20133457. https:\/\/doi.org\/10.1109\/ICCV.2017.371","DOI":"10.1109\/ICCV.2017.371"},{"key":"10792_CR27","doi-asserted-by":"publisher","unstructured":"Gambi A, Mueller M, Fraser G (2019) Automatically testing self-driving cars with search-based procedural content generation. In: Proceedings of the 28th ACM SIGSOFT International Symposium on Software Testing and Analysis, ISSTA 2019, pp. 318\u2013328. ACM, New York, NY, USA. https:\/\/doi.org\/10.1145\/3293882.3330566","DOI":"10.1145\/3293882.3330566"},{"key":"10792_CR28","doi-asserted-by":"crossref","unstructured":"Gao Y, Piccinini M, Zhang Y, Wang D, Moller K, Brusnicki R, Zarrouki B, Gambi A, Totz JF, Storms K, Peters S, Stocco A, Alrifaee B, Pavone M, Betz J (2025) Foundation models in autonomous driving: A survey on scenario generation and scenario analysis. https:\/\/arxiv.org\/abs\/2506.11526","DOI":"10.1109\/OJITS.2026.3660686"},{"key":"10792_CR29","unstructured":"Goodfellow I, Pouget-Abadie J, Mirza M, Xu B, Warde-Farley D, Ozair S, Courville A, Bengio Y (2014) Generative adversarial nets. Advances in Neural Information Processing Systems 27:2672\u20132680. http:\/\/papers.nips.cc\/paper\/5423-generative-adversarial-nets.pdf"},{"issue":"11","key":"10792_CR30","doi-asserted-by":"publisher","first-page":"139","DOI":"10.1145\/3422622","volume":"63","author":"I Goodfellow","year":"2020","unstructured":"Goodfellow I, Pouget-Abadie J, Mirza M, Xu B, Warde-Farley D, Ozair S, Courville A, Bengio Y (2020) Generative adversarial networks. Commun ACM 63(11):139\u2013144","journal-title":"Commun ACM"},{"key":"10792_CR31","unstructured":"Goodfellow IJ, Shlens J, Szegedy C (2014) Explaining and harnessing adversarial examples. CoRR abs\/1412.6572 https:\/\/api.semanticscholar.org\/CorpusID:6706414"},{"key":"10792_CR32","doi-asserted-by":"crossref","unstructured":"Gunning D, Stefik M, Choi J, Miller T, Stumpf S, Yang GZ (2019) Xai\u2013explainable artificial intelligence. Science Robotics 4(37):eaay7120","DOI":"10.1126\/scirobotics.aay7120"},{"key":"10792_CR33","doi-asserted-by":"crossref","unstructured":"Guo J, Jiang Y, Zhao Y, Chen Q, Sun J (2018) Dlfuzz: Differential fuzzing testing of deep learning systems. In: Proceedings of the 2018 26th ACM Joint Meeting on European Software Engineering Conference and Symposium on the Foundations of Software Engineering, pp. 739\u2013743","DOI":"10.1145\/3236024.3264835"},{"key":"10792_CR34","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735","volume-title":"Long short-term memory","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter S (1997) Long short-term memory. Neural Computation MIT-Press, Cambridge"},{"key":"10792_CR35","doi-asserted-by":"publisher","unstructured":"Humbatova N, Jahangirova G, Bavota G, Riccio V, Stocco A, Tonella P (2020) Taxonomy of real faults in deep learning systems. In: Proceedings of 42nd International Conference on Software Engineering, ICSE \u201920, p. 12 pages. ACM, New York, NY, USA. https:\/\/doi.org\/10.1145\/3377811.3380395","DOI":"10.1145\/3377811.3380395"},{"issue":"1","key":"10792_CR36","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s10515-021-00310-0","volume":"29","author":"M Hussain","year":"2022","unstructured":"Hussain M, Ali N, Hong JE (2022) Deepguard: A framework for safeguarding autonomous driving systems from inconsistent behaviour. Automated Software Engg 29(1):1. https:\/\/doi.org\/10.1007\/s10515-021-00310-0","journal-title":"Automated Software Engg"},{"key":"10792_CR37","unstructured":"ISO (2011) Road vehicles \u2013 Functional safety"},{"key":"10792_CR38","doi-asserted-by":"crossref","unstructured":"Jahangirova G, Stocco A, Tonella P (2021) Quality metrics and oracles for autonomous vehicles testing. In: Proceedings of 14th IEEE International Conference on Software Testing, Verification and Validation, ICST \u201921. IEEE","DOI":"10.1109\/ICST49551.2021.00030"},{"key":"10792_CR39","unstructured":"Jetley S, Lord NA, Lee N, Torr PH (2018) Learn to pay attention. arXiv preprint arXiv:1804.02391"},{"key":"10792_CR40","doi-asserted-by":"crossref","unstructured":"Jha S, Banerjee SS, Tsai T, Hari SKS, Sullivan MB, Kalbarczyk ZT, Keckler SW, Iyer RK (2019) Ml-based fault injection for autonomous vehicles: A case for bayesian fault injection. 2019 49th Annual IEEE\/IFIP International Conference on Dependable Systems and Networks (DSN) pp. 112\u2013124. https:\/\/api.semanticscholar.org\/CorpusID:195776612","DOI":"10.1109\/DSN.2019.00025"},{"key":"10792_CR41","doi-asserted-by":"publisher","DOI":"10.1002\/stvr.1894","volume":"34","author":"Z Jiang","year":"2024","unstructured":"Jiang Z, Li H, Wang R, Tian X, Liang C, Yan F, Zhang J, Liu Z (2024) Validity matters: Uncertainty-guided testing of deep neural networks. Software Testing Verification and Reliability 34:e1894","journal-title":"Software Testing Verification and Reliability"},{"key":"10792_CR42","unstructured":"Julian, KD, Kochenderfer MJ, Owen MP (2018) Deep neural network compression for aircraft collision avoidance systems. CoRR abs\/1810.04240"},{"key":"10792_CR43","doi-asserted-by":"crossref","unstructured":"Kang S, Feldt R, Yoo S (2020) Sinvad: Search-based image space navigation for dnn image classifier test input generation. In: Proceedings of the IEEE\/ACM 42nd International Conference on Software Engineering Workshops, pp. 521\u2013528","DOI":"10.1145\/3387940.3391456"},{"issue":"103","key":"10792_CR44","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3635706","volume":"33","author":"S Kang","year":"2024","unstructured":"Kang S, Feldt R, Yoo S (2024) Deceiving humans and machines alike: Search-based test input generation for dnns using variational autoencoders. ACM Transactions on Software Engineering Methodologies 33(103):1\u201324. https:\/\/doi.org\/10.1145\/3635706","journal-title":"ACM Transactions on Software Engineering Methodologies"},{"key":"10792_CR45","unstructured":"Kao V (2018) Sentimental analysis on imdb by lstm. https:\/\/www.kaggle.com\/code\/vincentman0403\/sentimental-analysis-on-imdb-by-lstm\/notebook. Accessed: 2024-08-18"},{"key":"10792_CR46","doi-asserted-by":"publisher","unstructured":"Kim J, Canny J (2017) Interpretable learning for self-driving cars by visualizing causal attention. https:\/\/doi.org\/10.48550\/ARXIV.1703.10631. https:\/\/arxiv.org\/abs\/1703.10631","DOI":"10.48550\/ARXIV.1703.10631"},{"key":"10792_CR47","doi-asserted-by":"publisher","unstructured":"Kim J, Feldt R, Yoo S (2019) Guiding deep learning system testing using surprise adequacy. In: 2019 IEEE\/ACM 41st International Conference on Software Engineering (ICSE), ICSE \u201919, pp. 1039\u20131049. IEEE, IEEE Press, Piscataway, NJ, USA. https:\/\/doi.org\/10.1109\/ICSE.2019.00108","DOI":"10.1109\/ICSE.2019.00108"},{"key":"10792_CR48","doi-asserted-by":"publisher","unstructured":"Kim S, Liu M, Rhee JJ, Jeon Y, Kwon Y, Kim CH (2022) DriveFuzz. In: Proceedings of the 2022 ACM SIGSAC Conference on Computer and Communications Security. ACM https:\/\/doi.org\/10.1145\/3548606.3560558. https:\/\/doi.org\/10.1145%2F3548606.3560558","DOI":"10.1145\/3548606.3560558"},{"key":"10792_CR49","unstructured":"Kingma DP, Welling M (2013) Auto-encoding variational bayes. arXiv:1312.6114"},{"key":"10792_CR50","unstructured":"Kothari V (2020) Image classification of mnist using VGG16. Kaggle Notebook. https:\/\/www.kaggle.com\/code\/viratkothari\/image-classification-of-mnist-using-vgg16"},{"key":"10792_CR51","doi-asserted-by":"crossref","unstructured":"Lambertenghi SC, Leonhard H, Stocco A (2025) Benchmarking image perturbations for testing automated driving assistance systems. In: Proc. of 18th IEEE International Conference on Software Testing, Verification and Validation, ICST \u201925. IEEE","DOI":"10.1109\/ICST62969.2025.10988980"},{"key":"10792_CR52","doi-asserted-by":"crossref","unstructured":"Landis JR, Koch GG (1977) The measurement of observer agreement for categorical data. Biometrics pp. 159\u2013174","DOI":"10.2307\/2529310"},{"key":"10792_CR53","unstructured":"Larman C et\u00a0al (1998) Applying UML and patterns, vol.\u00a02. Prentice Hall Upper Saddle River"},{"issue":"11","key":"10792_CR54","doi-asserted-by":"publisher","first-page":"2278","DOI":"10.1109\/5.726791","volume":"86","author":"Y LeCun","year":"1998","unstructured":"LeCun Y, Bottou L, Bengio Y, Haffner P (1998) Gradient-based learning applied to document recognition. Proc IEEE 86(11):2278\u20132324","journal-title":"Proc IEEE"},{"key":"10792_CR55","doi-asserted-by":"publisher","unstructured":"Li G, Li Y, Jha S, Tsai T, Sullivan M, Hari SKS, Kalbarczyk Z, Iyer R (2020) Av-fuzzer: Finding safety violations in autonomous driving systems. In: 2020 IEEE 31st International Symposium on Software Reliability Engineering (ISSRE), pp. 25\u201336. https:\/\/doi.org\/10.1109\/ISSRE5003.2020.00012","DOI":"10.1109\/ISSRE5003.2020.00012"},{"key":"10792_CR56","doi-asserted-by":"crossref","unstructured":"Li XH, Shi Y, Li H, Bai W, Cao CC, Chen L (2021) An experimental study of quantitative evaluations on saliency methods. In: Proceedings of the 27th ACM sigkdd conference on knowledge discovery & data mining, pp. 3200\u20133208","DOI":"10.1145\/3447548.3467148"},{"key":"10792_CR57","doi-asserted-by":"publisher","unstructured":"Lou G, Deng Y, Zheng X, Zhang M, Zhang T (2021) Testing of autonomous driving systems: Where are we and where should we go?. https:\/\/doi.org\/10.48550\/ARXIV.2106.12233. https:\/\/arxiv.org\/abs\/2106.12233","DOI":"10.48550\/ARXIV.2106.12233"},{"issue":"1","key":"10792_CR58","doi-asserted-by":"publisher","first-page":"384","DOI":"10.1109\/TSE.2022.3150788","volume":"49","author":"C Lu","year":"2023","unstructured":"Lu C, Shi Y, Zhang H, Zhang M, Wang T, Yue T, Ali S (2023) Learning configurations of operating environment of autonomous vehicles to maximize their collisions. IEEE Trans Software Eng 49(1):384\u2013402. https:\/\/doi.org\/10.1109\/TSE.2022.3150788","journal-title":"IEEE Trans Software Eng"},{"key":"10792_CR59","doi-asserted-by":"publisher","unstructured":"Ma L, Juefei-Xu F, Zhang F, Sun J, Xue M, Li B, Chen C, Su T, Li L, Liu Y, Zhao J, Wang Y (2018) Deepgauge: Multi-granularity testing criteria for deep learning systems. In: Proceedings of the 33rd ACM\/IEEE International Conference on Automated Software Engineering, ASE 2018, pp. 120\u2013131. ACM, New York, NY, USA. https:\/\/doi.org\/10.1145\/3238147.3238202","DOI":"10.1145\/3238147.3238202"},{"key":"10792_CR60","unstructured":"Maas AL, Daly RE, Pham PT, Huang D, Ng AY, Potts C (2011) Learning word vectors for sentiment analysis. In: Proceedings of the 49th Annual Meeting of the Association for Computational Linguistics: Human Language Technologies - Volume 1, HLT \u201911, pp. 142\u2013150. Association for Computational Linguistics, Stroudsburg, PA, USA. http:\/\/dl.acm.org\/citation.cfm?id=2002472.2002491"},{"key":"10792_CR61","doi-asserted-by":"crossref","unstructured":"Maryam, Biagiola M, Stocco A, Riccio V (2025) Benchmarking Generative AI Models for Deep Learning Test Input Generation. In: Proceedings of 18th IEEE International Conference on Software Testing, Verification and Validation, ICST \u201925, p. 12 pages. IEEE","DOI":"10.1109\/ICST62969.2025.10989043"},{"issue":"11","key":"10792_CR62","doi-asserted-by":"publisher","first-page":"39","DOI":"10.1145\/219717.219748","volume":"38","author":"GA Miller","year":"1995","unstructured":"Miller GA (1995) Wordnet: a lexical database for english. Commun ACM 38(11):39\u201341","journal-title":"Commun ACM"},{"key":"10792_CR63","unstructured":"Mouret J, Clune J (2015) Illuminating search spaces by mapping elites. CoRR abs\/1504.04909. http:\/\/arxiv.org\/abs\/1504.04909"},{"key":"10792_CR64","doi-asserted-by":"publisher","unstructured":"Mullins GE, Stankiewicz PG, Hawthorne RC, Gupta SK (2018) Adaptive generation of challenging scenarios for testing and evaluation of autonomous vehicles. J Syst Softw 137:197\u2013215. https:\/\/doi.org\/10.1016\/j.jss.2017.10.031. http:\/\/www.sciencedirect.com\/science\/article\/pii\/S0164121217302546","DOI":"10.1016\/j.jss.2017.10.031"},{"key":"10792_CR65","unstructured":"Naeem MF, Oh SJ, Uh Y, Choi Y, Yoo J (2020) Reliable fidelity and diversity metrics for generative models"},{"issue":"4","key":"10792_CR66","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3640335","volume":"33","author":"N Neelofar","year":"2024","unstructured":"Neelofar N, Aleti A (2024) Identifying and explaining safety-critical scenarios for autonomous vehicles via key features. ACM Transactions on Software Engineering and Methodology 33(4):1\u201332","journal-title":"ACM Transactions on Software Engineering and Methodology"},{"key":"10792_CR67","unstructured":"OpenAI (2024) Gpt-4o mini: advancing cost-efficient intelligence. https:\/\/openai.com\/index\/gpt-4o-mini-advancing-cost-efficient-intelligence\/. Accessed: 2024-09-08"},{"key":"10792_CR68","unstructured":"OpenAI (2024) The most powerful platform for building ai products. https:\/\/openai.com\/api\/. Accessed: 2024-09-08"},{"key":"10792_CR69","doi-asserted-by":"publisher","unstructured":"Pei K, Cao Y, Yang J, Jana S (2017) Deepxplore: Automated whitebox testing of deep learning systems. In: proceedings of the 26th Symposium on Operating Systems Principles, SOSP \u201917, vol.\u00a062, pp. 1\u201318. ACM, New York, NY, USA. https:\/\/doi.org\/10.1145\/3132747.3132785","DOI":"10.1145\/3132747.3132785"},{"key":"10792_CR70","unstructured":"Replication package (2025) https:\/\/github.com\/ast-fortiss-tum\/XMutant"},{"key":"10792_CR71","doi-asserted-by":"crossref","unstructured":"Ribeiro MT, Singh S, Guestrin C (2016) \"why should I trust you?\": Explaining the predictions of any classifier. In: Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining, San Francisco, CA, USA, August 13-17, 2016, pp. 1135\u20131144","DOI":"10.1145\/2939672.2939778"},{"key":"10792_CR72","doi-asserted-by":"crossref","unstructured":"Riccio V, Humbatova N, Jahangirova G, Tonella P (2021) Deepmetis: Augmenting a deep learning test set to increase its mutation score. In: Proceedings of the 36th IEEE\/ACM International Conference on Automated Software Engineering, ASE \u201921. IEEE\/ACM","DOI":"10.1109\/ASE51524.2021.9678764"},{"key":"10792_CR73","doi-asserted-by":"crossref","unstructured":"Riccio V, Jahangirova G, Stocco A, Humbatova N, Weiss M, Tonella P (2020) Testing Machine Learning based Systems: A Systematic Mapping. Empirical Software Engineering","DOI":"10.1007\/s10664-020-09881-0"},{"key":"10792_CR74","doi-asserted-by":"crossref","unstructured":"Riccio V, Tonella P (2020) Model-Based Exploration of the Frontier of Behaviours for Deep Learning System Testing. In: Proceedings of ACM Joint European Software Engineering Conference and Symposium on the Foundations of Software Engineering, ESEC\/FSE \u201920","DOI":"10.1145\/3368089.3409730"},{"key":"10792_CR75","doi-asserted-by":"crossref","unstructured":"Riccio V, Tonella P (2023) When and why test generators for deep learning produce invalid inputs: an empirical study. In: Proceedings of 45th International Conference on Software Engineering, ICSE \u201923, p. 12 pages. ACM","DOI":"10.1109\/ICSE48619.2023.00104"},{"key":"10792_CR76","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-28954-6","volume-title":"Explainable AI: interpreting, explaining and visualizing deep learning","author":"W Samek","year":"2019","unstructured":"Samek W, Montavon G, Vedaldi A, Hansen LK, M\u00fcller KR (2019) Explainable AI: interpreting, explaining and visualizing deep learning, vol 11700. Springer Nature, Berlin"},{"key":"10792_CR77","unstructured":"Samek W, Wiegand T, M\u00fcller KR (2017) Explainable artificial intelligence: Understanding, visualizing and interpreting deep learning models. arXiv preprint arXiv:1708.08296"},{"key":"10792_CR78","doi-asserted-by":"crossref","unstructured":"Sch\u00fctze H, Manning CD, Raghavan P (2008) Introduction to information retrieval, vol.\u00a039. Cambridge University Press Cambridge","DOI":"10.1017\/CBO9780511809071"},{"key":"10792_CR79","doi-asserted-by":"crossref","unstructured":"Selvaraju RR, Cogswell M, Das A, Vedantam R, Parikh D, Batra D (2017) Grad-cam: Visual explanations from deep networks via gradient-based localization. In: Proceedings of the IEEE international conference on computer vision, pp. 618\u2013626","DOI":"10.1109\/ICCV.2017.74"},{"issue":"3","key":"10792_CR80","doi-asserted-by":"publisher","first-page":"379","DOI":"10.1002\/j.1538-7305.1948.tb01338.x","volume":"27","author":"CE Shannon","year":"1948","unstructured":"Shannon CE (1948) A mathematical theory of communication. The Bell system technical journal 27(3):379\u2013423","journal-title":"The Bell system technical journal"},{"key":"10792_CR81","unstructured":"Shrikumar A, Greenside P, Kundaje A (2017) Learning important features through propagating activation differences. In: International conference on machine learning, pp. 3145\u20133153. PMLR"},{"key":"10792_CR82","unstructured":"Simonyan K, Vedaldi A, Zisserman A (2013) Deep inside convolutional networks: Visualising image classification models and saliency maps. arXiv preprint arXiv:1312.6034"},{"key":"10792_CR83","unstructured":"Smilkov D, Thorat N, Kim B, Vi\u00e9gas F, Wattenberg M (2017) Smoothgrad: removing noise by adding noise. arXiv preprint arXiv:1706.03825"},{"key":"10792_CR84","doi-asserted-by":"publisher","unstructured":"Stocco A, Nunes PJ, d\u2019Amorim M, Tonella P (2022) ThirdEye: Attention Maps for Safe Autonomous Driving Systems. In: Proceedings of 37th IEEE\/ACM International Conference on Automated Software Engineering, ASE \u201922. IEEE\/ACM. https:\/\/doi.org\/10.1145\/3551349.3556968. Accepted","DOI":"10.1145\/3551349.3556968"},{"issue":"04","key":"10792_CR85","doi-asserted-by":"publisher","first-page":"1928","DOI":"10.1109\/TSE.2022.3202311","volume":"49","author":"A Stocco","year":"2023","unstructured":"Stocco A, Pulfer B, Tonella P (2023) Mind the Gap! A Study on the Transferability of Virtual Versus Physical-World Testing of Autonomous Driving Systems. IEEE Trans Software Eng 49(04):1928\u20131940. https:\/\/doi.org\/10.1109\/TSE.2022.3202311","journal-title":"IEEE Trans Software Eng"},{"issue":"3","key":"10792_CR86","doi-asserted-by":"publisher","first-page":"73","DOI":"10.1007\/s10664-023-10306-x","volume":"28","author":"A Stocco","year":"2023","unstructured":"Stocco A, Pulfer B, Tonella P (2023) Model vs system level testing of autonomous driving systems: A replication and extension study. Empirical Softw Engg 28(3):73. https:\/\/doi.org\/10.1007\/s10664-023-10306-x","journal-title":"Empirical Softw Engg"},{"key":"10792_CR87","doi-asserted-by":"crossref","unstructured":"Stocco A, Weiss M, Calzana, M., Tonella, P (2020) Misbehaviour prediction for autonomous driving systems. In: Proceedings of 42nd International Conference on Software Engineering, ICSE \u201920, p. 12 pages. ACM","DOI":"10.1145\/3377811.3380353"},{"key":"10792_CR88","unstructured":"Sundararajan M, Taly A, Yan Q (2017) Axiomatic attribution for deep networks"},{"key":"10792_CR89","doi-asserted-by":"publisher","unstructured":"Tang S, Zhang Z, Zhang Y, Zhou J, Guo Y, Liu S, Guo S, Li Y, Ma L, Xue Y, Liu Y (2022) A survey on automated driving system testing: Landscapes and trends. CoRR abs\/2206.05961. https:\/\/doi.org\/10.48550\/arXiv.2206.05961. https:\/\/arxiv.org\/abs\/2206.05961","DOI":"10.48550\/arXiv.2206.05961"},{"key":"10792_CR90","doi-asserted-by":"publisher","unstructured":"Tian Y, Pei K, Jana S, Ray B (2018) Deeptest: Automated testing of deep-neural-network-driven autonomous cars. In: Proceedings of the 40th International Conference on Software Engineering, ICSE \u201918, pp. 303\u2013314. ACM, New York, NY, USA. https:\/\/doi.org\/10.1145\/3180155.3180220","DOI":"10.1145\/3180155.3180220"},{"key":"10792_CR91","unstructured":"Tjoa E, Khok HJ, Chouhan T, Guan C (2022) Improving deep neural network classification confidence using heatmap-based explainable AI. CoRR abs\/2201.00009. https:\/\/arxiv.org\/abs\/2201.00009"},{"key":"10792_CR92","unstructured":"Udacity (2017) A self-driving car simulator built with Unity. https:\/\/github.com\/udacity\/self-driving-car-sim. Online; accessed 18 August 2019"},{"key":"10792_CR93","unstructured":"Unity3d (2021) https:\/\/unity.com"},{"key":"10792_CR94","unstructured":"Vilone G, Longo L (2020) Explainable artificial intelligence: a systematic review. arXiv preprint arXiv:2006.00093"},{"key":"10792_CR95","doi-asserted-by":"crossref","unstructured":"Wang R, Guo J, Gao C, Fan G, Chong CY, Xia X (2025) Can llms replace human evaluators? an empirical study of llm-as-a-judge in software engineering https:\/\/arxiv.org\/abs\/2502.06193","DOI":"10.1145\/3728963"},{"key":"10792_CR96","doi-asserted-by":"publisher","DOI":"10.1145\/3771557","author":"O Wei\u00dfl","year":"2025","unstructured":"Wei\u00dfl O, Abdellatif A, Chen X, Merabishvili G, Riccio V, Kacianka S, Stocco A (2025) Targeted deep learning system boundary testing. ACM Trans Softw Eng Methodol. https:\/\/doi.org\/10.1145\/3771557","journal-title":"ACM Trans Softw Eng Methodol"},{"issue":"6","key":"10792_CR97","doi-asserted-by":"publisher","first-page":"80","DOI":"10.2307\/3001968","volume":"1","author":"F Wilcoxon","year":"1945","unstructured":"Wilcoxon F (1945) Individual comparisons by ranking methods. Biometrics Bulletin 1(6):80. https:\/\/doi.org\/10.2307\/3001968","journal-title":"Biometrics Bulletin"},{"key":"10792_CR98","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-29807-3","volume-title":"Advances in K-means clustering: a data mining thinking","author":"J Wu","year":"2012","unstructured":"Wu J (2012) Advances in K-means clustering: a data mining thinking. Springer Science & Business Media, Berlin"},{"key":"10792_CR99","doi-asserted-by":"publisher","unstructured":"Xie X, Ma L, Juefei-Xu F, Xue M, Chen H, Liu Y, Zhao J, Li B, Yin J, See S (2019) Deephunter: A coverage-guided fuzz testing framework for deep neural networks. In: Proceedings of the 28th ACM SIGSOFT International Symposium on Software Testing and Analysis, ISSTA \u201919, pp. 146\u2013157. ACM, New York, NY, USA. https:\/\/doi.org\/10.1145\/3293882.3330579","DOI":"10.1145\/3293882.3330579"},{"key":"10792_CR100","doi-asserted-by":"publisher","unstructured":"Xu Y, Yang X, Gong L, Lin HC, Wu TY, Li Y, Vasconcelos N (2020) Explainable object-induced action decision for autonomous vehicles. https:\/\/doi.org\/10.48550\/ARXIV.2003.09405. https:\/\/arxiv.org\/abs\/2003.09405","DOI":"10.48550\/ARXIV.2003.09405"},{"key":"10792_CR101","unstructured":"Yuan Y, Pang Q, Wang S (2021) You can\u2019t see the forest for its trees: Assessing deep neural network testing via neural coverage. CoRR abs\/2112.01955. https:\/\/arxiv.org\/abs\/2112.01955"},{"key":"10792_CR102","unstructured":"Zhang C, Almpanidis G, Fan G, Deng B, Zhang Y, Liu J, Kamel A, Soda P, Gama J (2024) A systematic review on long-tailed learning. https:\/\/arxiv.org\/abs\/2408.00483"},{"key":"10792_CR103","doi-asserted-by":"crossref","unstructured":"Zhang J, Keung J, Ma X, Li X, Xiao Y, Li Y, Chan WK (2024) Enhancing valid test input generation with distribution awareness for deep neural networks. In: 2024 IEEE 48th Annual Computers, Software, and Applications Conference (COMPSAC), pp. 1095\u20131100. IEEE","DOI":"10.1109\/COMPSAC61105.2024.00148"},{"key":"10792_CR104","doi-asserted-by":"publisher","unstructured":"Zhang M, Zhang Y, Zhang L, Liu C, Khurshid S (2018) Deeproad: Gan-based metamorphic testing and input validation framework for autonomous driving systems. In: Proceedings of the 33rd ACM\/IEEE International Conference on Automated Software Engineering, ASE 2018, pp. 132\u2013142. ACM, New York, NY, USA. https:\/\/doi.org\/10.1145\/3238147.3238187","DOI":"10.1145\/3238147.3238187"},{"key":"10792_CR105","doi-asserted-by":"crossref","unstructured":"Zhang Q, Wang H, Lu H, Won D, Yoon SW (2018) Medical image synthesis with generative adversarial networks for tissue recognition. In: 2018 IEEE International Conference on Healthcare Informatics (ICHI). IEEE","DOI":"10.1109\/ICHI.2018.00030"},{"issue":"1","key":"10792_CR106","doi-asserted-by":"publisher","first-page":"717","DOI":"10.1038\/s41598-018-36745-x","volume":"9","author":"J Zhao","year":"2019","unstructured":"Zhao J, Feng Q, Wu P, Lupu RA, Wilke RA, Wells QS, Denny JC, Wei WQ (2019) Learning from longitudinal data in electronic health record and genetic data to improve cardiovascular event prediction. Sci Rep 9(1):717. https:\/\/doi.org\/10.1038\/s41598-018-36745-x","journal-title":"Sci Rep"},{"key":"10792_CR107","doi-asserted-by":"crossref","unstructured":"Zheng L, Chiang WL, Sheng Y, Zhuang S, Wu Z, Zhuang Y, Lin Z, Li Z, Li D, Xing EP, Zhang H, Gonzalez JE, Stoica I (2023) Judging llm-as-a-judge with mt-bench and chatbot arena. https:\/\/arxiv.org\/abs\/2306.05685","DOI":"10.52202\/075280-2020"},{"key":"10792_CR108","unstructured":"Zhong Z, Kaiser G, Ray B (2021) Neural network guided evolutionary fuzzing for finding traffic violations of autonomous vehicles"},{"key":"10792_CR109","doi-asserted-by":"crossref","unstructured":"Zohdinasab T, Riccio V, Gambi A, Tonella P (2021) Deephyperion: exploring the feature space of deep learning-based systems through illumination search. In: Proceedings of the 30th ACM SIGSOFT International Symposium on Software Testing and Analysis, ISSTA \u201921, pp. 79\u201390. Association for Computing Machinery","DOI":"10.1145\/3460319.3464811"},{"key":"10792_CR110","doi-asserted-by":"publisher","unstructured":"Zohdinasab T, Riccio V, Tonella P (2023) Deepatash: Focused test generation for deep learning systems. In: R.\u00a0Just, G.\u00a0Fraser (eds.) Proceedings of the 32nd ACM SIGSOFT International Symposium on Software Testing and Analysis, ISSTA 2023, Seattle, WA, USA, July 17-21, 2023, pp. 954\u2013966. ACM. https:\/\/doi.org\/10.1145\/3597926.3598109","DOI":"10.1145\/3597926.3598109"},{"key":"10792_CR111","doi-asserted-by":"publisher","unstructured":"Zohdinasab T, Riccio V, Tonella P (2023) An empirical study on low- and high-level explanations of deep learning misbehaviours. In: 2023 ACM\/IEEE International Symposium on Empirical Software Engineering and Measurement (ESEM), pp. 1\u201311. https:\/\/doi.org\/10.1109\/ESEM56168.2023.10304866","DOI":"10.1109\/ESEM56168.2023.10304866"}],"container-title":["Empirical Software Engineering"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10664-025-10792-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10664-025-10792-1","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10664-025-10792-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T21:01:04Z","timestamp":1781643664000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10664-025-10792-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,27]]},"references-count":111,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2026,7]]}},"alternative-id":["10792"],"URL":"https:\/\/doi.org\/10.1007\/s10664-025-10792-1","relation":{},"ISSN":["1382-3256","1573-7616"],"issn-type":[{"value":"1382-3256","type":"print"},{"value":"1573-7616","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,3,27]]},"assertion":[{"value":"4 March 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 December 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 March 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Not applicable.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical Approval"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Informed consent"}},{"value":"The authors declared that they have no conflict of interest.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflicts of interests"}},{"value":"Not applicable.","order":5,"name":"Ethics","group":{"name":"EthicsHeading","label":"Clinical Trial Number"}}],"article-number":"102"}}