{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,30]],"date-time":"2026-07-30T11:05:14Z","timestamp":1785409514813,"version":"3.56.0"},"reference-count":48,"publisher":"Springer Science and Business Media LLC","issue":"7","license":[{"start":{"date-parts":[[2026,6,24]],"date-time":"2026-06-24T00:00:00Z","timestamp":1782259200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,6,24]],"date-time":"2026-06-24T00:00:00Z","timestamp":1782259200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62476047"],"award-info":[{"award-number":["62476047"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Mach Learn"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1007\/s10994-026-07080-4","type":"journal-article","created":{"date-parts":[[2026,6,24]],"date-time":"2026-06-24T10:08:31Z","timestamp":1782295711000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["TAD-Bench: A Comprehensive Benchmark for Embedding-Based Text Anomaly Detection"],"prefix":"10.1007","volume":"115","author":[{"given":"Yang","family":"Cao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sikun","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chen","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haolong","family":"Xiang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lianyong","family":"Qi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Bo","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rongsheng","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ming","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,6,24]]},"reference":[{"key":"7080_CR1","volume-title":"Outlier analysis","author":"CC Aggarwal","year":"2016","unstructured":"Aggarwal, C. C. (2016). Outlier analysis. Springer."},{"key":"7080_CR2","doi-asserted-by":"crossref","unstructured":"Almeida, T. A., Hidalgo, J. M. G. & Yamakami, A. (2011). Contributions to the study of sms spam filtering: New collection and results. In Proceedings of the 11th acm symposium on document engineering. pp. 259\u2013262.","DOI":"10.1145\/2034691.2034742"},{"issue":"4","key":"7080_CR3","doi-asserted-by":"publisher","first-page":"968","DOI":"10.1111\/coin.12156","volume":"34","author":"TR Bandaragoda","year":"2018","unstructured":"Bandaragoda, T. R., Ting, K. M., Albrecht, D., Liu, F. T., Zhu, Y., & Wells, J. R. (2018). Isolation-based anomaly detection using nearest-neighbor ensembles. Computational Intelligence, 34(4), 968\u2013998.","journal-title":"Computational Intelligence"},{"key":"7080_CR4","doi-asserted-by":"crossref","unstructured":"Bejan, M., Manolache, A. & Popescu, M. (2023). Ad-nlp: A benchmark for anomaly detection in natural language processing. In Proceedings of the 2023 conference on empirical methods in natural language processing. pp. 10766\u201310778.","DOI":"10.18653\/v1\/2023.emnlp-main.664"},{"issue":"3","key":"7080_CR5","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3444690","volume":"54","author":"A Bl\u00e1zquez-Garc\u00eda","year":"2021","unstructured":"Bl\u00e1zquez-Garc\u00eda, A., Conde, A., Mori, U., & Lozano, J. A. (2021). A review on outlier\/anomaly detection in time series data. ACM Computing Surveys (CSUR), 54(3), 1\u201333.","journal-title":"ACM Computing Surveys (CSUR)"},{"key":"7080_CR6","doi-asserted-by":"crossref","unstructured":"Breunig, M. M., Kriegel, H-P., Ng, R. T. & Sander, J. (2000). Lof: identifying density-based local outliers. In Proceedings of the 2000 acm sigmod international conference on management of data. pp. 93\u2013104.","DOI":"10.1145\/342009.335388"},{"key":"7080_CR7","first-page":"1877","volume":"33","author":"T Brown","year":"2020","unstructured":"Brown, T., Mann, B., Ryder, N., Subbiah, M., Kaplan, J. D., Dhariwal, P., et al. (2020). Language models are few-shot learners. Advances in Neural Information Processing Systems, 33, 1877\u20131901.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"7080_CR8","doi-asserted-by":"crossref","unstructured":"Cao, Y., Yu, B., Yang, S., Liu, M. & Yang, Y. (2026). Towards token-level text anomaly detection. In proceedings of the acm web conference 2026. pp. 8733\u20138736.","DOI":"10.1145\/3774904.3792952"},{"issue":"7","key":"7080_CR9","doi-asserted-by":"publisher","first-page":"2761","DOI":"10.1007\/s10115-023-01856-z","volume":"65","author":"LS da Costa","year":"2023","unstructured":"da Costa, L. S., Oliveira, I. L., & Fileto, R. (2023). Text classification using embeddings: A survey. Knowledge and Information Systems, 65(7), 2761\u20132803.","journal-title":"Knowledge and Information Systems"},{"key":"7080_CR10","doi-asserted-by":"crossref","unstructured":"Das, S. D., Basak, A. & Dutta, S. (2021). A heuristic-driven ensemble framework for covid-19 fake news detection. In Combating online hostile posts in regional languages during emergency situation: First international workshop, constraint 2021, collocated with aaai 2021, virtual event, Feb 8, 2021, revised selected papers 1 (pp. 164\u2013176).","DOI":"10.1007\/978-3-030-73696-5_16"},{"key":"7080_CR11","doi-asserted-by":"crossref","unstructured":"Davidson, T., Warmsley, D., Macy, M., & Weber, I. (2017). Automated hate speech detection and the problem of offensive language. In Proceedings of the international aaai conference on web and social media,11, 512\u2013515.","DOI":"10.1609\/icwsm.v11i1.14955"},{"key":"7080_CR12","unstructured":"Devlin, J., Chang, M-W., Lee, K. & Toutanova, K. (2019), 06. BERT: Pre-training of deep bidirectional transformers for language understanding. In Burstein, J., Doran, C. & Solorio, T. (Eds), Proceedings of the 2019 conference of the north American chapter of the association for computational linguistics: Human language technologies, volume 1 (long and short papers). pp. 4171\u20134186. Minneapolis, Minnesota. Association for Computational Linguistics. https:\/\/aclanthology.org\/N19-1423\/."},{"key":"7080_CR13","unstructured":"Durani, W., Leiber, C., Durani, K., Plant, C. & B\u00f6hm, C. (2025). Anomaly detection by an ensemble of random pairs of hyperspheres. In The thirty-ninth annual conference on neural information processing systems."},{"key":"7080_CR14","unstructured":"Goldstein, M. & Dengel, A. (2012). Histogram-based outlier score (hbos): A fast unsupervised anomaly detection algorithm. In KI-2012: poster and demo track. pp. 159\u201363."},{"key":"7080_CR15","doi-asserted-by":"crossref","unstructured":"Goodge, A., Hooi, B., Ng, S.-K., & Ng, W. S. (2022). Lunar: Unifying local outlier detection methods via graph neural networks. In Proceedings of the aaai conference on artificial intelligence,36, 6737\u20136745.","DOI":"10.1609\/aaai.v36i6.20629"},{"key":"7080_CR16","unstructured":"Lan, Z., Chen, M., Goodman, S., Gimpel, K., Sharma, P. & Soricut, R. (2020).  Albert: A lite bert for self-supervised learning of language representations. arxiv: abs\/1909.11942"},{"key":"7080_CR17","doi-asserted-by":"crossref","unstructured":"Li, Y., Li, J., Xiao, Z., Yang, T., Nian, Y., Hu, X. & Zhao, Y. (2025).  NLP-ADBench: NLP anomaly detection benchmark. Findings of the association for computational linguistics: Emnlp 2025, pp. 2464\u20132474.","DOI":"10.18653\/v1\/2025.findings-emnlp.133"},{"key":"7080_CR18","doi-asserted-by":"crossref","unstructured":"Li, Z., Zhao, Y., Botta, N., Ionescu, C. & Hu, X. (2020). Copod: copula-based outlier detection. 2020 ieee international conference on data mining (icdm). pp. 1118\u20131123.","DOI":"10.1109\/ICDM50108.2020.00135"},{"issue":"12","key":"7080_CR21","doi-asserted-by":"publisher","first-page":"12181","DOI":"10.1109\/TKDE.2022.3159580","volume":"35","author":"Z Li","year":"2022","unstructured":"Li, Z., Zhao, Y., Hu, X., Botta, N., Ionescu, C., & Chen, G. H. (2022). Ecod: Unsupervised outlier detection using empirical cumulative distribution functions. IEEE Transactions on Knowledge and Data Engineering, 35(12), 12181\u201312193.","journal-title":"IEEE Transactions on Knowledge and Data Engineering"},{"key":"7080_CR19","doi-asserted-by":"crossref","unstructured":"Liu, F. T., Ting, K. M. & Zhou, Z.-H. (2008). Isolation forest. In 2008 Eighth ieee international conference on data mining. pp. 413\u2013422.","DOI":"10.1109\/ICDM.2008.17"},{"key":"7080_CR20","doi-asserted-by":"crossref","unstructured":"Liu, F. T., Ting, K. M. & Zhou, Z.-H. (2012). Isolation-based anomaly detection. In ACM Transactions on Knowledge Discovery from Data (TKDD), In 6(1), pp. 1\u201339.","DOI":"10.1145\/2133360.2133363"},{"key":"7080_CR49","unstructured":"Li, Z., Huang, Q., Zhu, Y., Yang, L., Mohammadi Amiri, M., van Stein, N., van Leeuwen, M. (2026).Scalable, explainable and provably robust anomaly detection with one-step flow matching. Advances in Neural Information Processing Systems,38, 87834\u201387899."},{"key":"7080_CR22","first-page":"28","volume":"17","author":"V Metsis","year":"2006","unstructured":"Metsis, V., Androutsopoulos, I., & Paliouras, G. (2006). Spam filtering with naive bayes-which naive bayes? CEAS Ceas, 17, 28\u201369.","journal-title":"CEAS,"},{"key":"7080_CR23","unstructured":"Mikolov, T. (2013). Efficient estimation of word representations in vector space. arXiv:1301.3781 arXiv preprint."},{"key":"7080_CR24","unstructured":"OpenAI (2024). New embedding models and API updates.https:\/\/openai.com\/index\/new-embedding-models-and-api-updates\/"},{"issue":"2","key":"7080_CR25","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3439950","volume":"54","author":"G Pang","year":"2021","unstructured":"Pang, G., Shen, C., Cao, L., & Hengel, A. V. D. (2021). Deep learning for anomaly detection: A review. ACM Computing Surveys (CSUR), 54(2), 1\u201338.","journal-title":"ACM Computing Surveys (CSUR)"},{"key":"7080_CR26","doi-asserted-by":"crossref","unstructured":"Pennington, J., Socher, R. & Manning, C. D. (2014). Glove: Global vectors for word representation. In Proceedings of the 2014 conference on empirical methods in natural language processing (emnlp). pp. 1532\u20131543.","DOI":"10.3115\/v1\/D14-1162"},{"key":"7080_CR27","doi-asserted-by":"crossref","unstructured":"Peters, M.E., Neumann, M., Iyyer, M., Gardner, M., Clark, C., Lee, K. & Zettlemoyer, L. (2018) Deep contextualized word representations. In Walker, M., Ji, H. & Stent, A. (Eds.), Proceedings of the 2018 conference of the north American chapter of the association for computational linguistics: Human language technologies, volume 1 (long papers) ( 2227\u20132237). New Orleans, LouisianaAssociation for Computational Linguistics. https:\/\/aclanthology.org\/N18-5000\/","DOI":"10.18653\/v1\/N18-1202"},{"key":"7080_CR28","doi-asserted-by":"publisher","first-page":"215","DOI":"10.1016\/j.sigpro.2013.12.026","volume":"99","author":"MA Pimentel","year":"2014","unstructured":"Pimentel, M. A., Clifton, D. A., Clifton, L., & Tarassenko, L. (2014). A review of novelty detection. Signal Processing, 99, 215\u2013249.","journal-title":"Signal Processing"},{"key":"7080_CR29","doi-asserted-by":"crossref","unstructured":"Qiao, H., Tong, H., An, B., King, I., Aggarwal, C. & Pang, G. (2025). Deep graph anomaly detection: A survey and new perspectives.  IEEE Transactions on Knowledge and Data Engineering. https:\/\/doi.org\/10.1109\/TKDE.2025.3581578","DOI":"10.1109\/TKDE.2025.3581578"},{"key":"7080_CR30","doi-asserted-by":"crossref","unstructured":"Ramaswamy, S., Rastogi, R. & Shim, K. (2000). Efficient algorithms for mining outliers from large data sets. In Proceedings of the 2000 acm sigmod international conference on management of data. pp. 427\u2013438.","DOI":"10.1145\/342009.335437"},{"key":"7080_CR31","unstructured":"Ruff, L., Vandermeulen, R., Goernitz, N., Deecke, L., Siddiqui, S. A., Binder, A., & Kloft, M. (2018). Deep one-class classification. In International conference on machine learning . pp. 4393\u20134402."},{"issue":"5","key":"7080_CR32","doi-asserted-by":"publisher","first-page":"513","DOI":"10.1016\/0306-4573(88)90021-0","volume":"24","author":"G Salton","year":"1988","unstructured":"Salton, G., & Buckley, C. (1988). Term-weighting approaches in automatic text retrieval. Information Processing & Management, 24(5), 513\u2013523.","journal-title":"Information Processing & Management"},{"key":"7080_CR33","unstructured":"Shyu, M-L. , Chen, S-C. , Sarinnapakorn, K. & Chang, L. (2003). A novel anomaly detection scheme based on principal component classifier."},{"issue":"4","key":"7080_CR34","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3441453","volume":"15","author":"G Steinbuss","year":"2021","unstructured":"Steinbuss, G., & B\u00f6hm, K. (2021). Benchmarking unsupervised outlier detection with realistic synthetic data. ACM Transactions on Knowledge Discovery from Data (TKDD), 15(4), 1\u201320.","journal-title":"ACM Transactions on Knowledge Discovery from Data (TKDD)"},{"key":"7080_CR35","unstructured":"Sugiyama, M. & Borgwardt, K. (2013). Rapid distance-based outlier detection via sampling. In Advances in neural information processing systems, 26."},{"key":"7080_CR36","doi-asserted-by":"publisher","unstructured":"Wang, C., Xu, D. & Li, Z. (2024). Log2graphs: An unsupervised framework for log anomaly detection with efficient feature extraction. arXiv preprint. https:\/\/doi.org\/10.48550\/arXiv.2409.11890","DOI":"10.48550\/arXiv.2409.11890"},{"key":"7080_CR37","first-page":"5776","volume":"33","author":"W Wang","year":"2020","unstructured":"Wang, W., Wei, F., Dong, L., Bao, H., Yang, N., & Zhou, M. (2020). Minilm: Deep self-attention distillation for task-agnostic compression of pre-trained transformers. Advances in Neural Information Processing Systems, 33, 5776\u20135788.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"7080_CR38","doi-asserted-by":"publisher","unstructured":"Xiao, F. & Fan, J. (2025). Text-adbench: Text anomaly detection benchmark based on llms embedding. arXiv preprint. https:\/\/doi.org\/10.48550\/arXiv.2507.12295","DOI":"10.48550\/arXiv.2507.12295"},{"key":"7080_CR40","doi-asserted-by":"publisher","first-page":"88006","DOI":"10.1109\/ACCESS.2024.3418340","volume":"12","author":"C Xu","year":"2024","unstructured":"Xu, C., & Kechadi, M.-T. (2024). An enhanced fake news detection system with fuzzy deep learning. IEEE Access, 12, 88006\u201388021. https:\/\/doi.org\/10.1109\/ACCESS.2024.3418340","journal-title":"IEEE Access"},{"issue":"12","key":"7080_CR41","doi-asserted-by":"publisher","first-page":"12591","DOI":"10.1109\/TKDE.2023.3270293","volume":"35","author":"H Xu","year":"2023","unstructured":"Xu, H., Pang, G., Wang, Y., & Wang, Y. (2023b). Deep isolation forest for anomaly detection. IEEE Transactions on Knowledge and Data Engineering, 35(12), 12591\u201312604.","journal-title":"IEEE Transactions on Knowledge and Data Engineering"},{"key":"7080_CR39","doi-asserted-by":"crossref","unstructured":"Xu, Y., Milleret, J. & Segond, F. (2023a). Comparative analysis of anomaly detection algorithms in text data. In Proceedings of the 14th international conference on recent advances in natural language processing. pp. 1234\u20131245.","DOI":"10.26615\/978-954-452-092-2_131"},{"key":"7080_CR42","doi-asserted-by":"crossref","unstructured":"Yang, T., Nian, Y., Li, L., Xu, R., Li, Y., Li, J., . . . others (2025). Ad-llm: Benchmarking large language models for anomaly detection. Findings of the association for computational linguistics: Acl 2025 (pp. 1524\u20131547). https:\/\/doi.org\/10.18653\/v1\/2025.findings-acl.79","DOI":"10.18653\/v1\/2025.findings-acl.79"},{"key":"7080_CR44","doi-asserted-by":"crossref","unstructured":"Zampieri, M., Malmasi, S., Nakov, P., Rosenthal, S., Farra, N. & Kumar, R. (2019) Predicting the type and target of offensive posts in social media. In J.\u00a0Burstein, C.\u00a0Doran& T.\u00a0Solorio (Eds.), Proceedings of the 2019 conference of the north American chapter of the association for computational linguistics: Human language technologies, volume 1 (long and short papers) (pp. 1415\u20131420). Minneapolis, MinnesotaAssociation for Computational Linguistics. https:\/\/doi.org\/aclanthology.org\/N19-1144\/","DOI":"10.18653\/v1\/N19-1144"},{"key":"7080_CR45","doi-asserted-by":"publisher","unstructured":"Zhang, D., Li, J., Zeng, Z. & Wang, F. (2024). Jasper and stella: distillation of sota embedding models. arXiv preprint. https:\/\/doi.org\/10.48550\/arXiv.2412.19048","DOI":"10.48550\/arXiv.2412.19048"},{"key":"7080_CR46","doi-asserted-by":"publisher","unstructured":"Zhao, Y., Nasrullah, Z., & Li, Z. (2019).  Pyod: A python toolbox for scalable outlier detection. Journal of Machine Learning Research,20(96), 1\u20137. https:\/\/doi.org\/10.48550\/arXiv.1901.01588","DOI":"10.48550\/arXiv.1901.01588"},{"key":"7080_CR47","doi-asserted-by":"publisher","unstructured":"Zhu, Y., Yuan, H., Wang, S., Liu, J., Liu, W., Deng, C., . . . Wen, J.-R. (2025). Large language models for information retrieval: A survey. ACM Transactions on Information Systems, 44(1), 1\u201354. https:\/\/doi.org\/10.1145\/3748304","DOI":"10.1145\/3748304"},{"key":"7080_CR48","unstructured":"Zhuang, L., Wayne, L., Ya, S. & Jun, Z. (2021) A robustly optimized BERT pre-training approach with post-training. In S.\u00a0Li et al. (Eds.), Proceedings of the 20th Chinese national conference on computational linguistics 1218\u20131227. Huhhot, China,Chinese Information Processing Society of China. https:\/\/aclanthology.org\/2021.ccl-1.108\/"}],"container-title":["Machine Learning"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10994-026-07080-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10994-026-07080-4","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10994-026-07080-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,30]],"date-time":"2026-07-30T10:25:40Z","timestamp":1785407140000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10994-026-07080-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,24]]},"references-count":48,"journal-issue":{"issue":"7","published-print":{"date-parts":[[2026,7]]}},"alternative-id":["7080"],"URL":"https:\/\/doi.org\/10.1007\/s10994-026-07080-4","relation":{},"ISSN":["0885-6125","1573-0565"],"issn-type":[{"value":"0885-6125","type":"print"},{"value":"1573-0565","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,6,24]]},"assertion":[{"value":"15 April 2026","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 May 2026","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 May 2026","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 June 2026","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors have no conflict of interest to declare that are relevant to the content of this article.","order":1,"name":"Ethics","label":"Conflict of interest","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Not applicable.","order":2,"name":"Ethics","label":"Ethical Approval","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Not applicable.","order":3,"name":"Ethics","label":"Consent for Publication","group":{"name":"EthicsHeading","label":"Declarations"}}],"article-number":"155"}}