{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,17]],"date-time":"2026-08-17T15:01:40Z","timestamp":1786978900856,"version":"3.56.0"},"publisher-location":"Cham","reference-count":45,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031834714","type":"print"},{"value":"9783031834721","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0"},{"start":{"date-parts":[[2025,3,16]],"date-time":"2025-03-16T00:00:00Z","timestamp":1742083200000},"content-version":"vor","delay-in-days":74,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"abstract":"<jats:title>Abstract<\/jats:title>\n                  <jats:p>\n                    This work introduces , an anonymization pipeline that can be used for generalization and suppression-based anonymization of nominal textual tabular data. It automatically generates value generalization hierarchies (VGHs) that, in turn, can be used to generalize attributes in quasi-identifiers. The pipeline leverages embeddings to generate semantically close value generalizations through iterative clustering. We applied KMeans and Hierarchical Agglomerative Clustering on 13 different predefined text embeddings (both open and closed-source (via APIs)). Our approach is experimentally tested on a well-known benchmark dataset for anonymization: The UCI Machine Learning Repository\u2019s Adult dataset.  supports anonymization procedures by offering more possibilities compared to using arbitrarily chosen VGHs. Experiments demonstrate that these VGHs can outperform manually constructed ones in terms of downstream efficacy (especially for small\n                    <jats:italic>k<\/jats:italic>\n                    -anonymity) and therefore can foster the quality of anonymized datasets. Our implementation is made public.\n                  <\/jats:p>","DOI":"10.1007\/978-3-031-83472-1_9","type":"book-chapter","created":{"date-parts":[[2025,3,15]],"date-time":"2025-03-15T12:13:47Z","timestamp":1742040827000},"page":"122-137","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["ClustEm4Ano: Clustering Text Embeddings of\u00a0Nominal Textual Attributes for\u00a0Microdata Anonymization"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-0986-3504","authenticated-orcid":false,"given":"Robert","family":"Aufschl\u00e4ger","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4370-9234","authenticated-orcid":false,"given":"Sebastian","family":"Wilhelm","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7303-113X","authenticated-orcid":false,"given":"Michael","family":"Heigl","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6206-2969","authenticated-orcid":false,"given":"Martin","family":"Schramm","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,3,16]]},"reference":[{"issue":"7","key":"9_CR1","doi-asserted-by":"publisher","DOI":"10.2196\/18055","volume":"22","author":"M Abdalla","year":"2020","unstructured":"Abdalla, M., Abdalla, M., Hirst, G., Rudzicz, F.: Exploring the privacy-preserving properties of word embeddings: algorithmic validation study. J. Med. Internet Res. 22(7), e18055 (2020)","journal-title":"J. Med. Internet Res."},{"issue":"6","key":"9_CR2","doi-asserted-by":"publisher","first-page":"901","DOI":"10.1093\/jamia\/ocaa038","volume":"27","author":"M Abdalla","year":"2020","unstructured":"Abdalla, M., Abdalla, M., Rudzicz, F., Hirst, G.: Using word embeddings to improve the privacy of clinical notes. J. Am. Med. Inform. Assoc. 27(6), 901\u2013907 (2020)","journal-title":"J. Am. Med. Inform. Assoc."},{"key":"9_CR3","doi-asserted-by":"crossref","unstructured":"Aufschl\u00e3ger, R., et al.: Anonymization procedures for tabular data: an explanatory technical and legal synthesis. Information 14(9) (2023)","DOI":"10.3390\/info14090487"},{"key":"9_CR4","unstructured":"Ayala-Rivera, V., et al.: Enhancing the utility of anonymized data by improving the quality of generalization hierarchies. Trans. Data Priv. 10(1) (2017)"},{"key":"9_CR5","doi-asserted-by":"crossref","unstructured":"Ayala-Rivera, V., Murphy, L., Thorpe, C.: Automatic construction of generalization hierarchies for publishing anonymized data. In: Proceedings of the 9th International Conference Knowledge Science, Engineering & Management (KSEM), pp. 262\u2013274 (2016)","DOI":"10.1007\/978-3-319-47650-6_21"},{"key":"9_CR6","doi-asserted-by":"crossref","unstructured":"Bayardo, R., Agrawal, R.: Data privacy through optimal $$k$$-anonymization. In: Proceedings of the 21st International Conference on Data Engineering (ICDE), pp. 217\u2013228 (2005)","DOI":"10.1109\/ICDE.2005.42"},{"key":"9_CR7","unstructured":"Becker, B., Kohavi, R.: Adult. UCI Machine Learning Repository (1996)"},{"issue":"2","key":"9_CR8","doi-asserted-by":"publisher","first-page":"151","DOI":"10.1007\/s41060-021-00285-x","volume":"13","author":"D Biesner","year":"2022","unstructured":"Biesner, D., et al.: Anonymization of German financial documents using neural network-based language models with contextual word representations. Int. J. Data Sci. Anal. 13(2), 151\u2013161 (2022)","journal-title":"Int. J. Data Sci. Anal."},{"key":"9_CR9","doi-asserted-by":"publisher","first-page":"123","DOI":"10.1023\/A:1018054314350","volume":"24","author":"L Breiman","year":"1996","unstructured":"Breiman, L.: Bagging predictors. Mach. Learn. 24, 123\u2013140 (1996)","journal-title":"Mach. Learn."},{"key":"9_CR10","doi-asserted-by":"publisher","first-page":"5","DOI":"10.1023\/A:1010933404324","volume":"45","author":"L Breiman","year":"2001","unstructured":"Breiman, L.: Random forests. Mach. Learn. 45, 5\u201332 (2001)","journal-title":"Mach. Learn."},{"key":"9_CR11","doi-asserted-by":"crossref","unstructured":"Byun, J.W., Kamra, A., Bertino, E., Li, N.: Efficient $$k$$-anonymization using clustering techniques. In: Proceedings of the 12th International Conference on Database Systems for Advanced Applications (DASFAA), pp. 188\u2013200 (2007)","DOI":"10.1007\/978-3-540-71703-4_18"},{"key":"9_CR12","unstructured":"Cai, X., Huang, J., Bian, Y., Church, K.: Isotropy in the contextual embedding space: clusters and manifolds. In: Proceedings of the 9th International Conference on Learning Representations (ICLR) (2021)"},{"key":"9_CR13","doi-asserted-by":"crossref","unstructured":"Campan, A., Cooper, N., Truta, T.M.: On-the-fly generalization hierarchies for numerical attributes revisited. In: Proceedings of the 8th VLDB Workshop on Secure Data Management (SDM), pp. 18\u201332 (2011)","DOI":"10.1007\/978-3-642-23556-6_2"},{"key":"9_CR14","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: BERT: pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the Conference of the N.American Chapter of the Association for Computational Linguistics: Human Language Technologies, vol. 1 (Long and Short Papers), pp. 4171\u20134186 (2019)"},{"key":"9_CR15","doi-asserted-by":"crossref","unstructured":"D\u00edaz, J.S.P., Garc\u00eda, A.L.: Comparison of machine learning models applied on anonymized data with different techniques. In: Proceedings of the IEEE International Conference on Cyber Security & Resilience (CSR), pp. 618\u2013623 (2023)","DOI":"10.1109\/CSR57506.2023.10224917"},{"issue":"5","key":"9_CR16","doi-asserted-by":"publisher","first-page":"670","DOI":"10.1197\/jamia.M3144","volume":"16","author":"K Emam","year":"2009","unstructured":"Emam, K., et al.: A globally optimal $$k$$-anonymity method for the de-identification of health data. J. Am. Med. Inform. Assoc. 16(5), 670\u201382 (2009)","journal-title":"J. Am. Med. Inform. Assoc."},{"issue":"1","key":"9_CR17","doi-asserted-by":"publisher","first-page":"119","DOI":"10.1006\/jcss.1997.1504","volume":"55","author":"Y Freund","year":"1997","unstructured":"Freund, Y., Schapire, R.E.: A decision-theoretic generalization of on-line learning and an application to boosting. J. Comput. Syst. Sci. 55(1), 119\u2013139 (1997)","journal-title":"J. Comput. Syst. Sci."},{"key":"9_CR18","doi-asserted-by":"crossref","unstructured":"Garat, D., Wonsever, D.: Automatic curation of court documents: anonymizing personal data. Information 13(1) (2022)","DOI":"10.3390\/info13010027"},{"key":"9_CR19","unstructured":"Grave, E., et al.: Learning word vectors for 157 languages. In: Proceedings of the 11th International Conference on Language Resources & Evaluation (LREC) (2018)"},{"key":"9_CR20","unstructured":"G\u00fcnther, M., et al.: Jina embeddings 2: 8192-token general-purpose text embeddings for long documents. arxiv:2310.19923 (2023)"},{"key":"9_CR21","doi-asserted-by":"crossref","unstructured":"Hassan, F., S\u00e1nchez, D., Soria-Comas, J., Domingo-Ferrer, J.: Automatic anonymization of textual documents: detecting sensitive information via word embeddings. In: Proceedings of the 18th IEEE International Conference on Trust, Security and Privacy in Computing (TrustCom) & Communications\/13th IEEE International Conference on Big Data Science & Engineering (BigDataSE), pp. 358\u2013365 (2019)","DOI":"10.1109\/TrustCom\/BigDataSE.2019.00055"},{"key":"9_CR22","doi-asserted-by":"crossref","unstructured":"Kohlmayer, F., et al.: Flash: efficient, stable and optimal $$k$$-anonymity. In: Proceedings of the International Conference on Privacy, Security, Risk (PASSAT) and Trust and International Conference on Social Computing (SocialCom), pp. 708\u2013717 (2012)","DOI":"10.1109\/SocialCom-PASSAT.2012.52"},{"key":"9_CR23","doi-asserted-by":"crossref","unstructured":"Li, N., Li, T., Venkatasubramanian, S.: t-closeness: privacy beyond $$k$$-anonymity and $$l$$-diversity. In: Proceedings of the 23rd IEEE International Conference on Data Engineering (ICDE), pp. 106\u2013115 (2006)","DOI":"10.1109\/ICDE.2007.367856"},{"key":"9_CR24","doi-asserted-by":"crossref","unstructured":"Lin, J.L., Wei, M.C.: An efficient clustering method for $$k$$-anonymization. In: Proceedings of the International Workshop on Privacy and Anonymity in Information Society (PAIS), pp. 46\u201350 (2008)","DOI":"10.1145\/1379287.1379297"},{"key":"9_CR25","doi-asserted-by":"crossref","unstructured":"Lison, P., et al.: Anonymisation models for text data: state of the art, challenges and future directions. In: Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing, vol. 1: Long Papers, pp. 4188\u20134203 (2021)","DOI":"10.18653\/v1\/2021.acl-long.323"},{"issue":"2","key":"9_CR26","doi-asserted-by":"publisher","first-page":"129","DOI":"10.1109\/TIT.1982.1056489","volume":"28","author":"S Lloyd","year":"1982","unstructured":"Lloyd, S.: Least squares quantization in PCM. IEEE Trans. Inf. Theory 28(2), 129\u2013137 (1982)","journal-title":"IEEE Trans. Inf. Theory"},{"key":"9_CR27","doi-asserted-by":"crossref","unstructured":"Machanavajjhala, A., Gehrke, J., Kifer, D., et\u00a0al.: $$l$$-diversity: privacy beyond $$k$$-anonymity. In: Proceedings of the 22nd IEEE International Conference on Data Engineering (ICDE) (2006)","DOI":"10.1109\/ICDE.2006.1"},{"key":"9_CR28","doi-asserted-by":"crossref","unstructured":"Mamede, N., Baptista, J., Dias, F.: Automated anonymization of text documents. In: Proceedings of the IEEE Congress on Evolutionary Computation (CEC), pp. 1287\u20131294 (2016)","DOI":"10.1109\/CEC.2016.7743936"},{"key":"9_CR29","unstructured":"Medlock, B.: An introduction to NLP-based textual anonymisation. In: Proceedings of the 5th International Conference on Language Resources & Evaluation (LREC) (2006)"},{"key":"9_CR30","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s40747-021-00512-9","volume":"7","author":"V Mehta","year":"2021","unstructured":"Mehta, V., Bawa, S., Singh, J.: WEClustering: word embeddings based text clustering technique for large datasets. Complex Intell. Syst. 7, 1\u201314 (2021)","journal-title":"Complex Intell. Syst."},{"key":"9_CR31","unstructured":"Mikolov, T., Chen, K., Corrado, G., Dean, J.: Efficient estimation of word representations in vector space. In: Proceedings of the 1st International Conference on Learning Representations (ICLR) (2013)"},{"key":"9_CR32","doi-asserted-by":"crossref","unstructured":"Mubark, A.A., Elabd, E., Abdulkader, H.: Semantic anonymization in publishing categorical sensitive attributes. In: Proceedings of the 8th International Conference on Knowledge & Smart Technology (KST), pp. 89\u201395 (2016)","DOI":"10.1109\/KST.2016.7440495"},{"key":"9_CR33","doi-asserted-by":"crossref","unstructured":"Muennighoff, N., Tazi, N., Magne, L., Reimers, N.: MTEB: massive text embedding benchmark. In: Proceedings of the 17th Conference of the European Chapter of the Association for Computational Linguistics (EACL), pp. 2014\u20132037 (2023)","DOI":"10.18653\/v1\/2023.eacl-main.148"},{"key":"9_CR34","unstructured":"Pedregosa, F., Vet al.: Scikit-learn: machine learning in Python. J. Mach. Learn. Res. 12, 2825\u20132830 (2011)"},{"key":"9_CR35","doi-asserted-by":"crossref","unstructured":"Pennington, J., Socher, R., Manning, C.: GloVe: global vectors for word representation. In: Proceedings of the Conference on Empirical Methods in Natural Language Processing (EMNLP), pp. 1532\u20131543 (2014)","DOI":"10.3115\/v1\/D14-1162"},{"key":"9_CR36","unstructured":"Prasser, F., Kohlmayer, F., Lautenschlaeger, R., Kuhn, K.: ARX-A comprehensive tool for anonymizing biomedical data. In: Proceedings of the AMIA Annual Symposium, pp. 984\u2013993 (2014)"},{"key":"9_CR37","unstructured":"\u0158eh\u016f\u0159ek, R., Sojka, P.: Software framework for topic modelling with large corpora. In: Proceedings of the LREC Workshop on New Challenges for NLP Frameworks, pp. 45\u201350 (2010)"},{"key":"9_CR38","doi-asserted-by":"crossref","unstructured":"Reimers, N., Gurevych, I.: Sentence-BERT: sentence embeddings using Siamese BERT-networks. In: Proceedings of the Conference on Empirical Methods in Natural Language Processing (EMNLP)and the 9th International Joint Conference on Natural Language Processing (IJCNLP) (2019)","DOI":"10.18653\/v1\/D19-1410"},{"issue":"6","key":"9_CR39","doi-asserted-by":"publisher","first-page":"1010","DOI":"10.1109\/69.971193","volume":"13","author":"P Samarati","year":"2001","unstructured":"Samarati, P.: Protecting respondents identities in microdata release. IEEE Trans. Knowl. Data Eng. 13(6), 1010\u20131027 (2001)","journal-title":"IEEE Trans. Knowl. Data Eng."},{"key":"9_CR40","unstructured":"Samarati, P., Sweeney, L.: Protecting privacy when disclosing information: $$k$$-anonymity and its enforcement through generalization and suppression. Technical report, SRI International (1998)"},{"key":"9_CR41","unstructured":"Song, K., et al.: MPNet: masked and permuted pre-training for language understanding. In: Proceedings of the 34th Annual Conference on Neural Information Processing Systems (NeurIPS), pp. 16857\u201316867 (2020)"},{"key":"9_CR42","unstructured":"Sweeney, L.: Replacing personally-identifying information in medical records, the scrub system. In: Proceedings of the AMIA Annual Symposium, pp. 333\u2013337 (1996)"},{"key":"9_CR43","unstructured":"Turian, J., Ratinov, L., Bengio, Y.: Word representations: a simple and general method for semi-supervised learning. In: Proceedings of the 48th Annual Meeting of the Association for Computational Linguistics (ACL), pp. 384\u2013394 (2010)"},{"issue":"301","key":"9_CR44","doi-asserted-by":"publisher","first-page":"236","DOI":"10.1080\/01621459.1963.10500845","volume":"58","author":"JH Ward Jr","year":"1963","unstructured":"Ward, J.H., Jr.: Hierarchical grouping to optimize an objective function. J. Am. Stat. Assoc. 58(301), 236\u2013244 (1963)","journal-title":"J. Am. Stat. Assoc."},{"issue":"5","key":"9_CR45","doi-asserted-by":"publisher","first-page":"564","DOI":"10.1197\/jamia.M2435","volume":"14","author":"B Wellner","year":"2007","unstructured":"Wellner, B., et al.: Rapidly retargetable approaches to de-identification in medical records. J. Am. Med. Inform. Assoc. 14(5), 564\u2013573 (2007)","journal-title":"J. Am. Med. Inform. Assoc."}],"container-title":["Lecture Notes in Computer Science","Database Engineered Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-83472-1_9","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,6]],"date-time":"2025-09-06T07:43:20Z","timestamp":1757144600000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-83472-1_9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9783031834714","9783031834721"],"references-count":45,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-83472-1_9","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"16 March 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"IDEAS","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Database Engineered Applications Symposium","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Bayonne","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"France","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28 August 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"31 August 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ideas-12024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/conferences.sigappfr.org\/ideas2024\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}