{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T05:14:20Z","timestamp":1784351660118,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":56,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,6,14]],"date-time":"2023-06-14T00:00:00Z","timestamp":1686700800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,6,14]]},"DOI":"10.1145\/3593434.3593445","type":"proceedings-article","created":{"date-parts":[[2023,5,30]],"date-time":"2023-05-30T12:54:01Z","timestamp":1685451241000},"page":"32-41","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":10,"title":["DQSOps: Data Quality Scoring Operations Framework for Data-Driven Applications"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0683-2783","authenticated-orcid":false,"given":"Firas","family":"Bayram","sequence":"first","affiliation":[{"name":"Karlstad university, Sweden"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9051-7609","authenticated-orcid":false,"given":"Bestoun S.","family":"Ahmed","sequence":"additional","affiliation":[{"name":"Karlstad university, Sweden"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-8052-6577","authenticated-orcid":false,"given":"Erik","family":"Hallin","sequence":"additional","affiliation":[{"name":"Uddeholms AB, Sweden"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-5403-0108","authenticated-orcid":false,"given":"Anton","family":"Engman","sequence":"additional","affiliation":[{"name":"Uddeholms AB, Sweden"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2023,6,14]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Principal component analysis","author":"Abdi Herv\u00e9","year":"2010","unstructured":"Herv\u00e9 Abdi and Lynne\u00a0J Williams. 2010. Principal component analysis. Wiley interdisciplinary reviews: computational statistics 2, 4 (2010), 433\u2013459."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1007\/s40747-020-00214-8"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1080\/02640414.2011.574719"},{"key":"e_1_3_2_1_4_1","volume-title":"The oracle problem in software testing: A survey","author":"Barr T","year":"2014","unstructured":"Earl\u00a0T Barr, Mark Harman, Phil McMinn, Muzammil Shahbaz, and Shin Yoo. 2014. The oracle problem in software testing: A survey. IEEE transactions on software engineering 41, 5 (2014), 507\u2013525."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.4018\/JDM.2015010103"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/1891879.1891881"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.4018\/978-1-7998-5101-1.ch001"},{"key":"e_1_3_2_1_8_1","volume-title":"Random forests. Machine learning 45, 1","author":"Breiman Leo","year":"2001","unstructured":"Leo Breiman. 2001. Random forests. Machine learning 45, 1 (2001), 5\u201332."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/JSEN.2022.3203853"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3141248"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3318464.3384707"},{"key":"e_1_3_2_1_12_1","volume-title":"Deep learning for anomaly detection: A survey. arXiv preprint arXiv:1901.03407","author":"Chalapathy Raghavendra","year":"2019","unstructured":"Raghavendra Chalapathy and Sanjay Chawla. 2019. Deep learning for anomaly detection: A survey. arXiv preprint arXiv:1901.03407 (2019)."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/DSN53405.2022.00027"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/2939672.2939785"},{"key":"e_1_3_2_1_15_1","volume-title":"Statistical Learning to Operationalize a Domain Agnostic Data Quality Scoring. arXiv preprint arXiv:2108.08905","author":"Chug Sezal","year":"2021","unstructured":"Sezal Chug, Priya Kaushal, Ponnurangam Kumaraguru, and Tavpritesh Sethi. 2021. Statistical Learning to Operationalize a Domain Agnostic Data Quality Scoring. arXiv preprint arXiv:2108.08905 (2021)."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2019.2899751"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"crossref","unstructured":"RalphB D\u2019Agostino. 2017. Goodness-of-fit-techniques. Routledge.","DOI":"10.1201\/9780203753064"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1111\/j.1468-0335.2012.00929.x"},{"key":"e_1_3_2_1_19_1","unstructured":"Adenekan Dedeke. 2000. A Conceptual Framework for Developing Quality Measures for Information Systems.. In IQ. 126\u2013128."},{"key":"e_1_3_2_1_20_1","volume-title":"International Workshop on Data Quality and Trust in Big Data. Springer, 1\u201315","author":"Ehrlinger Lisa","year":"2018","unstructured":"Lisa Ehrlinger and Wolfram W\u00f6\u00df. 2018. A novel data quality metric for minimality. In International Workshop on Data Quality and Trust in Big Data. Springer, 1\u201315."},{"key":"e_1_3_2_1_21_1","volume-title":"Cramer\u2013von Mises, and Anderson\u2013Darling test statistics for exponential populations with estimated parameters. Communications in Statistics\u2014Simulation and Computation\u00ae 37, 7","author":"Evans L","year":"2008","unstructured":"Diane\u00a0L Evans, John\u00a0H Drew, and Lawrence\u00a0M Leemis. 2008. The distribution of the Kolmogorov\u2013Smirnov, Cramer\u2013von Mises, and Anderson\u2013Darling test statistics for exponential populations with estimated parameters. Communications in Statistics\u2014Simulation and Computation\u00ae 37, 7 (2008), 1396\u20131421."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/2854006.2854008"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3340482.3342743"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3148238"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.dss.2018.03.011"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394486.3406477"},{"key":"e_1_3_2_1_27_1","volume-title":"An analysis and survey of the development of mutation testing","author":"Jia Yue","year":"2010","unstructured":"Yue Jia and Mark Harman. 2010. An analysis and survey of the development of mutation testing. IEEE transactions on software engineering 37, 5 (2010), 649\u2013678."},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/SEAA53835.2021.00050"},{"key":"e_1_3_2_1_29_1","volume-title":"Application of an ontology for characterizing data quality for a secondary use of EHR data. Applied clinical informatics 7, 01","author":"Johnson G","year":"2016","unstructured":"Steven\u00a0G Johnson, Stuart Speedie, Gyorgy Simon, Vipin Kumar, and Bonnie\u00a0L Westra. 2016. Application of an ontology for characterizing data quality for a secondary use of EHR data. Applied clinical informatics 7, 01 (2016), 69\u201388."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.mlwa.2021.100024"},{"key":"e_1_3_2_1_31_1","volume-title":"International journal of information quality 2, 3","year":"2011","unstructured":"Shirlee-ann Knight. 2011. The combined conceptual life-cycle model of information quality: part 1, an investigative framework. International journal of information quality 2, 3 (2011), 205\u2013230."},{"key":"e_1_3_2_1_32_1","volume-title":"Nonparametric Statistics with Applications to Science and Engineering with R","author":"Kvam Paul","unstructured":"Paul Kvam, Brani Vidakovic, and Seong-joon Kim. 2022. Nonparametric Statistics with Applications to Science and Engineering with R. John Wiley & Sons."},{"key":"e_1_3_2_1_33_1","volume-title":"Proceedings of the AutoML Workshop at ICML, Vol.\u00a02020","author":"LeDell Erin","year":"2020","unstructured":"Erin LeDell and Sebastien Poirier. 2020. H2o automl: Scalable automatic machine learning. In Proceedings of the AutoML Workshop at ICML, Vol.\u00a02020."},{"key":"e_1_3_2_1_34_1","volume-title":"Measures of distributional similarity. arXiv preprint cs\/0001012","author":"Lee Lillian","year":"2000","unstructured":"Lillian Lee. 2000. Measures of distributional similarity. arXiv preprint cs\/0001012 (2000)."},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/TSE.2016.2597136"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/18.61115"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.3390\/technologies9020026"},{"key":"e_1_3_2_1_38_1","volume-title":"The practitioner\u2019s guide to data quality improvement","author":"Loshin David","unstructured":"David Loshin. 2010. The practitioner\u2019s guide to data quality improvement. Elsevier."},{"key":"e_1_3_2_1_39_1","first-page":"2346","article-title":"Learning under concept drift: A review","volume":"31","author":"Lu Jie","year":"2018","unstructured":"Jie Lu, Anjin Liu, Fan Dong, Feng Gu, Joao Gama, and Guangquan Zhang. 2018. Learning under concept drift: A review. IEEE Transactions on Knowledge and Data Engineering 31, 12 (2018), 2346\u20132363.","journal-title":"IEEE Transactions on Knowledge and Data Engineering"},{"key":"e_1_3_2_1_40_1","volume-title":"Multivariate analysis of quality: an introduction","author":"Martens Harald","unstructured":"Harald Martens and Magni Martens. 2001. Multivariate analysis of quality: an introduction. John Wiley & Sons."},{"key":"e_1_3_2_1_41_1","first-page":"146","article-title":"Big data management: concepts, techniques and challenges","volume":"50","author":"Meng Xiaofeng","year":"2013","unstructured":"Xiaofeng Meng and Xiang Ci. 2013. Big data management: concepts, techniques and challenges. Journal of computer research and development 50, 1 (2013), 146\u2013169.","journal-title":"Journal of computer research and development"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.im.2012.10.001"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300356"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-23525-7_11"},{"key":"e_1_3_2_1_45_1","volume-title":"Data quality: the accuracy dimension","author":"Olson E","unstructured":"Jack\u00a0E Olson. 2003. Data quality: the accuracy dimension. Elsevier."},{"key":"e_1_3_2_1_46_1","first-page":"412","article-title":"A review of missing data treatment methods","volume":"1","author":"Peng Liu","year":"2005","unstructured":"Liu Peng and Lei Lei. 2005. A review of missing data treatment methods. Intell. Inf. Manag. Syst. Technol 1 (2005), 412\u2013419.","journal-title":"Intell. Inf. Manag. Syst. Technol"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/505248.506010"},{"key":"e_1_3_2_1_48_1","volume-title":"Concepts of nonparametric theory","author":"Pratt John\u00a0Winsor","unstructured":"John\u00a0Winsor Pratt and Jean\u00a0Dickinson Gibbons. 2012. Concepts of nonparametric theory. Springer Science & Business Media."},{"key":"e_1_3_2_1_49_1","volume-title":"Applied data science","author":"Rettig Laura","unstructured":"Laura Rettig, Mourad Khayati, Philippe Cudr\u00e9-Mauroux, and Micha\u0142 Pi\u00f3rkowski. 2019. Online anomaly detection over big data streams. In Applied data science. Springer, 289\u2013312."},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1002\/stvr.1473"},{"key":"e_1_3_2_1_51_1","volume-title":"The utility of multivariate outlier detection techniques for data quality evaluation in large studies: an application within the ONDRI project. BMC medical research methodology 19, 1","author":"Sunderland M","year":"2019","unstructured":"Kelly\u00a0M Sunderland, Derek Beaton, Julia Fraser, Donna Kwan, Paula\u00a0M McLaughlin, Manuel Montero-Odasso, Alicia\u00a0J Peltsch, Frederico Pieruccini-Faria, Demetrios\u00a0J Sahlas, Richard\u00a0H Swartz, 2019. The utility of multivariate outlier detection techniques for data quality evaluation in large studies: an application within the ONDRI project. BMC medical research methodology 19, 1 (2019), 1\u201316."},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1186\/s40537-021-00468-0"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1186\/s40537-020-0285-1"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1080\/14783363.2017.1332954"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1145\/240455.240479"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1109\/69.404034"}],"event":{"name":"EASE '23: The International Conference on Evaluation and Assessment in Software Engineering","location":"Oulu Finland","acronym":"EASE '23"},"container-title":["Proceedings of the 27th International Conference on Evaluation and Assessment in Software Engineering"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3593434.3593445","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3593434.3593445","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T07:08:09Z","timestamp":1755846489000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3593434.3593445"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,6,14]]},"references-count":56,"alternative-id":["10.1145\/3593434.3593445","10.1145\/3593434"],"URL":"https:\/\/doi.org\/10.1145\/3593434.3593445","relation":{},"subject":[],"published":{"date-parts":[[2023,6,14]]},"assertion":[{"value":"2023-06-14","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}