{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,6]],"date-time":"2026-02-06T01:30:49Z","timestamp":1770341449930,"version":"3.49.0"},"reference-count":53,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2024,10,25]],"date-time":"2024-10-25T00:00:00Z","timestamp":1729814400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,25]],"date-time":"2024-10-25T00:00:00Z","timestamp":1729814400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"People's Public Security University of China Top-notch Innovative Talent Cultivation Fund Supports Key Projects in Graduate Student Scientific Research and Innovation","award":["2024yjsky008"],"award-info":[{"award-number":["2024yjsky008"]}]},{"name":"People's Public Security University of China Top-notch Innovative Talent Cultivation Fund Supports Key Projects in Graduate Student Scientific Research and Innovation","award":["2024yjsky008"],"award-info":[{"award-number":["2024yjsky008"]}]},{"DOI":"10.13039\/501100012325","name":"National Office for Philosophy and Social Sciences","doi-asserted-by":"publisher","award":["20AZD114"],"award-info":[{"award-number":["20AZD114"]}],"id":[{"id":"10.13039\/501100012325","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Data Sci Anal"],"published-print":{"date-parts":[[2025,10]]},"DOI":"10.1007\/s41060-024-00652-4","type":"journal-article","created":{"date-parts":[[2024,10,25]],"date-time":"2024-10-25T05:52:10Z","timestamp":1729835530000},"page":"3205-3234","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["CPSDbench: a large language model evaluation benchmark and baseline for Chinese public security domain"],"prefix":"10.1007","volume":"20","author":[{"given":"Xin","family":"Tong","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bo","family":"Jin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhi","family":"Lin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Binjun","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qiang","family":"Cheng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ting","family":"Yu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,10,25]]},"reference":[{"key":"652_CR1","unstructured":"Touvron, H., Lavril, T., Izacard, G., Martinet, X., Lachaux, M.-A., Lacroix, T., Rozi\u00e8re, B., Goyal, N., Hambro, E., Azhar, F., et al.: Llama: Open and efficient foundation language models. Preprint at arxiv:2302.13971 (2023)"},{"key":"652_CR2","unstructured":"Touvron, H., Martin, L., Stone, K., Albert, P., Almahairi, A., Babaei, Y., Bashlykov, N., Batra, S., Bhargava, P., Bhosale, S., Bikel, D., Blecher, L., Ferrer, C.C., Chen, M., Cucurull, G., Esiobu, D., Fernandes, J., Fu, J., Fu, W., Fuller, B., Gao, C., Goswami, V., Goyal, N., Hartshorn, A., Hosseini, S., Hou, R., Inan, H., Kardas, M., Kerkez, V., Khabsa, M., Kloumann, I., Korenev, A., Koura, P.S., Lachaux, M.-A., Lavril, T., Lee, J., Liskovich, D., Lu, Y., Mao, Y., Martinet, X., Mihaylov, T., Mishra, P., Molybog, I., Nie, Y., Poulton, A., Reizenstein, J., Rungta, R., Saladi, K., Schelten, A., Silva, R., Smith, E.M., Subramanian, R., Tan, X.E., Tang, B., Taylor, R., Williams, A., Kuan, J.X., Xu, P., Yan, Z., Zarov, I., Zhang, Y., Fan, A., Kambadur, M., Narang, S., Rodriguez, A., Stojnic, R., Edunov, S., Scialom, T.: Llama 2: Open Foundation and Fine-Tuned Chat Models. Preprint at https:\/\/arxiv.org\/abs\/2307.09288 (2023)"},{"key":"652_CR3","first-page":"1950","volume":"35","author":"H Liu","year":"2022","unstructured":"Liu, H., Tam, D., Muqeeth, M., Mohta, J., Huang, T., Bansal, M., Raffel, C.A.: Few-shot parameter-efficient fine-tuning is better and cheaper than in-context learning. Adv. Neural. Inf. Process. Syst. 35, 1950\u20131965 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"issue":"12","key":"652_CR4","doi-asserted-by":"publisher","first-page":"2629","DOI":"10.1007\/s10439-023-03272-4","volume":"51","author":"L Giray","year":"2023","unstructured":"Giray, L.: Prompt engineering with chatgpt: a guide for academic writers. Ann. Biomed. Eng. 51(12), 2629\u20132633 (2023)","journal-title":"Ann. Biomed. Eng."},{"key":"652_CR5","doi-asserted-by":"crossref","unstructured":"Du, Z., Qian, Y., Liu, X., Ding, M., Qiu, J., Yang, Z., Tang, J.: GLM: general language model pretraining with autoregressive blank infilling. In: Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics, vol. 1, Dublin, IRELAND, 22-27 May 2022 (2022)","DOI":"10.18653\/v1\/2022.acl-long.26"},{"key":"652_CR6","unstructured":"Zeng, A., Liu, X., Du, Z., Wang, Z., Lai, H., Ding, M., Yang, Z., Xu, Y., Zheng, W., Xia, X., et al.: GLM-130b: an open bilingual pre-trained model. Preprint at arxiv:2210.02414 (2022)"},{"key":"652_CR7","unstructured":"Wu, S., Irsoy, O., Lu, S., Dabravolski, V., Dredze, M., Gehrmann, S., Kambadur, P., Rosenberg, D., Mann, G.: Bloomberggpt: a large language model for finance. Preprint at arxiv:2303.17564 (2023)"},{"key":"652_CR8","unstructured":"Cui, J., Li, Z., Yan, Y., Chen, B., Yuan, L.: Chatlaw: open-source legal large language model with integrated external knowledge bases. Preprint at arxiv:2306.16092 (2023)"},{"key":"652_CR9","unstructured":"Dan, Y., Lei, Z., Gu, Y., Li, Y., Yin, J., Lin, J., Ye, L., Tie, Z., Zhou, Y., Wang, Y., et al.: Educhat: A large-scale language model-based chatbot system for intelligent education. Preprint at arxiv:2308.02773 (2023)"},{"key":"652_CR10","unstructured":"Zhang, K., Yu, J., Yan, Z., Liu, Y., Adhikarla, E., Fu, S., Chen, X., Chen, C., Zhou, Y., Li, X., et al.: BiomedGPT: a unified and generalist biomedical generative pre-trained transformer for vision, language, and multimodal tasks. Preprint at arxiv:2305.17100 (2023)"},{"key":"652_CR11","unstructured":"Wang, H., Liu, C., Xi, N., Qiang, Z., Zhao, S., Qin, B., Liu, T.: Huatuo: Tuning llama model with chinese medical knowledge. Preprint at arxiv:2304.06975 (2023)"},{"key":"652_CR12","unstructured":"Li, J., Wang, X., Wu, X., Zhang, Z., Xu, X., Fu, J., Tiwari, P., Wan, X., Wang, B.: Huatuo-26M, a large-scale Chinese medical QA dataset. Preprint at arxiv:2305.01526 (2023)"},{"key":"652_CR13","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2024.3404862","author":"P Sarzaeim","year":"2024","unstructured":"Sarzaeim, P., Mahmoud, Q.H., Azim, A.: A framework for LLM-assisted smart policing system. IEEE Access (2024). https:\/\/doi.org\/10.1109\/ACCESS.2024.3404862","journal-title":"IEEE Access"},{"key":"652_CR14","doi-asserted-by":"crossref","unstructured":"Kim, H., Kim, D., Lee, J., Yoon, C., Choi, D., Gim, M., Kang, J.: LAPIS: language model-augmented police investigation system. arXiv preprint arXiv:2407.20248 (2024)","DOI":"10.1145\/3627673.3680044"},{"key":"652_CR15","unstructured":"Wang, A., Pruksachatkun, Y., Nangia, N., Singh, A., Michael, J., Hill, F., Levy, O., Bowman, S.: Superglue: a stickier benchmark for general-purpose language understanding systems. Adv. Neural Inform. Process. Syst. 32 (2019)"},{"key":"652_CR16","unstructured":"Huang, Y., Bai, Y., Zhu, Z., Zhang, J., Zhang, J., Su, T., Liu, J., Lv, C., Zhang, Y., Lei, J., et al.: C-Eval: a multi-level multi-discipline chinese evaluation suite for foundation models. Preprint at arxiv:2305.08322 (2023)"},{"key":"652_CR17","unstructured":"Hendrycks, D., Burns, C., Basart, S., Zou, A., Mazeika, M., Song, D., Steinhardt, J.: Measuring massive multitask language understanding. Preprint at arxiv:2009.03300 (2020)"},{"issue":"7972","key":"652_CR18","doi-asserted-by":"publisher","first-page":"172","DOI":"10.1038\/s41586-023-06291-2","volume":"620","author":"K Singhal","year":"2023","unstructured":"Singhal, K., Azizi, S., Tu, T., Mahdavi, S.S., Wei, J., Chung, H.W., Scales, N., Tanwani, A., Cole-Lewis, H., Pfohl, S.: Large language models encode clinical knowledge. Nature 620(7972), 172\u2013180 (2023)","journal-title":"Nature"},{"key":"652_CR19","unstructured":"Wang, X., Chen, G.H., Song, D., Zhang, Z., Chen, Z., Xiao, Q., Jiang, F., Li, J., Wan, X., Wang, B., et al.: CMB: a comprehensive medical benchmark in chinese. Preprint at arxiv:2308.08833 (2023)"},{"key":"652_CR20","doi-asserted-by":"crossref","unstructured":"Zhu, W., Wang, X., Zheng, H., Chen, M., Tang, B.: PromptCBLUE: a chinese prompt tuning benchmark for the medical domain. Preprint at arxiv:2310.14151 (2023)","DOI":"10.2139\/ssrn.4685921"},{"key":"652_CR21","doi-asserted-by":"crossref","unstructured":"Fei, Z., Shen, X., Zhu, D., Zhou, F., Han, Z., Zhang, S., Chen, K., Shen, Z., Ge, J.: Lawbench: benchmarking legal knowledge of large language models (2023)","DOI":"10.18653\/v1\/2024.emnlp-main.452"},{"key":"652_CR22","unstructured":"Dai, Y., Feng, D., Huang, J., Jia, H., Xie, Q., Zhang, Y., Han, W., Tian, W., Wang, H.: LAiW: a Chinese legal large language models benchmark (a technical report). Preprint at arxiv:2310.05620 (2023)"},{"key":"652_CR23","doi-asserted-by":"crossref","unstructured":"Niklaus, J., Matoshi, V., Rani, P., Galassi, A., St\u00fcrmer, M., Chalkidis, I.: Lextreme: a multi-lingual and multi-task benchmark for the legal domain. Preprint at arxiv:2301.13126 (2023)","DOI":"10.18653\/v1\/2023.findings-emnlp.200"},{"key":"652_CR24","unstructured":"Islam, P., Kannappan, A., Kiela, D., Qian, R., Scherrer, N., Vidgen, B.: FinanceBench: a new benchmark for financial question answering. Preprint at arxiv:2311.11944 (2023)"},{"key":"652_CR25","unstructured":"Zhang, L., Cai, W., Liu, Z., Yang, Z., Dai, W., Liao, Y., Qin, Q., Li, Y., Liu, X., Liu, Z., et al.: Fineval: a Chinese financial domain knowledge evaluation benchmark for large language models. Preprint at arxiv:2308.09975 (2023)"},{"key":"652_CR26","unstructured":"Lei, Y., Li, J., Jiang, M., Hu, J., Cheng, D., Ding, Z., Jiang, C.: CFBenchmark: Chinese financial assistant benchmark for large language model. Preprint at arxiv:2311.05812 (2023)"},{"issue":"12","key":"652_CR27","doi-asserted-by":"publisher","first-page":"6999","DOI":"10.1109\/TNNLS.2021.3084827","volume":"33","author":"Z Li","year":"2021","unstructured":"Li, Z., Liu, F., Yang, W., Peng, S., Zhou, J.: A survey of convolutional neural networks: analysis, applications, and prospects. IEEE Trans. Neural Netw. Learn. Syst. 33(12), 6999\u20137019 (2021)","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"issue":"8","key":"652_CR28","doi-asserted-by":"publisher","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter, S., Schmidhuber, J.: Long short-term memory. Neural Comput. 9(8), 1735\u20131780 (1997)","journal-title":"Neural Comput."},{"key":"652_CR29","unstructured":"Vaswani, A.: Attention is all you need. Adv. Neural Inform. Process. Syst (2017)"},{"key":"652_CR30","unstructured":"Radford, A., Narasimhan, K., Salimans, T., Sutskever, I., et al.: Improving language understanding by generative pre-training. OpenAI https:\/\/cdn.openai.com\/research-covers\/language-unsupervised\/language_understanding_paper.pdf (2018)"},{"key":"652_CR31","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., Radford, A., Klimov, O.: Proximal policy optimization algorithms. Preprint at arxiv:1707.06347 (2017)"},{"key":"652_CR32","first-page":"27730","volume":"35","author":"L Ouyang","year":"2022","unstructured":"Ouyang, L., Wu, J., Jiang, X., Almeida, D., Wainwright, C., Mishkin, P., Zhang, C., Agarwal, S., Slama, K., Ray, A.: Training language models to follow instructions with human feedback. Adv. Neural. Inf. Process. Syst. 35, 27730\u201327744 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"652_CR33","unstructured":"Wei, J., Tay, Y., Bommasani, R., Raffel, C., Zoph, B., Borgeaud, S., Yogatama, D., Bosma, M., Zhou, D., Metzler, D. et al.: Emergent abilities of large language models. Preprint at arxiv:2206.07682 (2022)"},{"key":"652_CR34","first-page":"1877","volume":"33","author":"T Brown","year":"2020","unstructured":"Brown, T., Mann, B., Ryder, N., Subbiah, M., Kaplan, J.D., Dhariwal, P., Neelakantan, A., Shyam, P., Sastry, G., Askell, A.: Language models are few-shot learners. Adv. Neural. Inf. Process. Syst. 33, 1877\u20131901 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"652_CR35","first-page":"24824","volume":"35","author":"J Wei","year":"2022","unstructured":"Wei, J., Wang, X., Schuurmans, D., Bosma, M., Xia, F., Chi, E., Le, Q.V., Zhou, D.: Chain-of-thought prompting elicits reasoning in large language models. Adv. Neural. Inf. Process. Syst. 35, 24824\u201324837 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"issue":"7","key":"652_CR36","first-page":"2578","volume":"31","author":"J Zhang","year":"2019","unstructured":"Zhang, J., Li, C.: Adversarial examples: opportunities and challenges. IEEE Trans. Neural Netw. Learn. Syst. 31(7), 2578\u20132593 (2019)","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"652_CR37","unstructured":"Achiam, J., Adler, S., Agarwal, S., Ahmad, L., Akkaya, I., Aleman, F.L., Almeida, D., Altenschmidt, J., Altman, S., Anadkat, S. et al.: GPT-4 technical report. arXiv preprint arXiv:2303.08774 (2023)"},{"key":"652_CR38","unstructured":"Dubey, A., Jauhri, A., Pandey, A., Kadian, A., Al-Dahle, A., Letman, A., Mathur, A., Schelten, A., Yang, A., Fan, A., et al.: The llama 3 herd of models. arXiv preprint arXiv:2407.21783 (2024)"},{"key":"652_CR39","unstructured":"Liu, A., Feng, B., Wang, B., Wang, B., Liu, B., Zhao, C., Dengr, C., Ruan, C., Dai, D., Guo, D., et al.: Deepseek-v2: a strong, economical, and efficient mixture-of-experts language model. arXiv preprint arXiv:2405.04434 (2024)"},{"key":"652_CR40","unstructured":"Yang, A., Xiao, B., Wang, B., Zhang, B., Bian, C., Yin, C., Lv, C., Pan, D., Wang, D., Yan, D. et al.: Baichuan 2: Open large-scale language models. Preprint at arxiv:2309.10305 (2023)"},{"key":"652_CR41","unstructured":"Bai, J., Bai, S., Chu, Y., Cui, Z., Dang, K., Deng, X., Fan, Y., Ge, W., Han, Y., Huang, F., et al.: Qwen technical report. Preprint at arxiv:2309.16609 (2023)"},{"key":"652_CR42","unstructured":"Sun, Y., Wang, S., Feng, S., Ding, S., Pang, C., Shang, J., Liu, J., Chen, X., Zhao, Y., Lu, Y., et al.: Ernie 3.0: large-scale knowledge enhanced pre-training for language understanding and generation. Preprint at arxiv:2107.02137 (2021)"},{"key":"652_CR43","unstructured":"Zhang, T., Kishore, V., Wu, F., Weinberger, K.Q., Artzi, Y.: BERTScore: evaluating text generation with BERT. Preprint at arxiv:1904.09675 (2019)"},{"key":"652_CR44","unstructured":"Devlin, J.: BERT: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)"},{"key":"652_CR45","doi-asserted-by":"crossref","unstructured":"Reimers, N.: Sentence-BERT: sentence embeddings using siamese BERT-networks. arXiv preprint arXiv:1908.10084 (2019)","DOI":"10.18653\/v1\/D19-1410"},{"key":"652_CR46","doi-asserted-by":"crossref","unstructured":"Xiao, S., Liu, Z., Zhang, P., Muennighoff, N., Lian, D., Nie, J.-Y.: C-pack: packed resources for general chinese embeddings. In: Proceedings of the 47th International ACM SIGIR Conference on Research and Development in Information Retrieval, pp. 641\u2013649 (2024)","DOI":"10.1145\/3626772.3657878"},{"key":"652_CR47","doi-asserted-by":"crossref","unstructured":"Chen, J., Xiao, S., Zhang, P., Luo, K., Lian, D., Liu, Z.: BGE M3-embedding: multi-lingual. Multi-functionality, multi-granularity text embeddings through self-knowledge distillation (2023)","DOI":"10.18653\/v1\/2024.findings-acl.137"},{"key":"652_CR48","unstructured":"Robinson, J., Chuang, C.-Y., Sra, S., Jegelka, S.: Contrastive learning with hard negative samples. In: International Conference on Learning Representations (2021)"},{"key":"652_CR49","doi-asserted-by":"crossref","unstructured":"Gao, T., Yao, X., Chen, D.: SimCSE: simple contrastive learning of sentence embeddings. In: Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing, pp. 6894\u20136910 (2021)","DOI":"10.18653\/v1\/2021.emnlp-main.552"},{"key":"652_CR50","unstructured":"Valmeekam, K., Marquez, M., Sreedharan, S., Kambhampati, S.: On the planning abilities of large language models: a critical investigation. In: Proceedings of the 37th International Conference on Neural Information Processing Systems, pp. 75993\u201376005 (2023)"},{"key":"652_CR51","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2302.01560","author":"Z Wang","year":"2024","unstructured":"Wang, Z., Cai, S., Chen, G., Liu, A., Ma, X.S., Liang, Y.: Describe, explain, plan and select: interactive planning with LLMS enables open-world multi-task agents. Adv. Neural Inform. Process. Syst. (2024). https:\/\/doi.org\/10.48550\/arXiv.2302.01560","journal-title":"Adv. Neural Inform. Process. Syst."},{"key":"652_CR52","doi-asserted-by":"crossref","unstructured":"Lu, J., Holleis, T., Zhang, Y., Aumayer, B., Nan, F., Bai, F., Ma, S., Ma, S., Li, M., Yin, G. et al.: Toolsandbox: a stateful, conversational, interactive evaluation benchmark for LLM tool use capabilities. arXiv preprint arXiv:2408.04682 (2024)","DOI":"10.18653\/v1\/2025.findings-naacl.65"},{"key":"652_CR53","doi-asserted-by":"crossref","unstructured":"Guo, Z., Huang, Y., Xiong, D.: Ctooleval: A chinese benchmark for llm-powered agent evaluation in real-world API interactions. In: Findings of the Association for Computational Linguistics ACL 2024, pp. 15711\u201315724 (2024)","DOI":"10.18653\/v1\/2024.findings-acl.928"}],"container-title":["International Journal of Data Science and Analytics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s41060-024-00652-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s41060-024-00652-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s41060-024-00652-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,25]],"date-time":"2025-09-25T10:52:43Z","timestamp":1758797563000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s41060-024-00652-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,25]]},"references-count":53,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2025,10]]}},"alternative-id":["652"],"URL":"https:\/\/doi.org\/10.1007\/s41060-024-00652-4","relation":{},"ISSN":["2364-415X","2364-4168"],"issn-type":[{"value":"2364-415X","type":"print"},{"value":"2364-4168","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,25]]},"assertion":[{"value":"7 May 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"20 September 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 October 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"This work was supported by the Double First-Class Innovative Research Project on Cyberspace Security Law Enforcement Technology at the People\u2019s Public Security University of China (2023SYL07).","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Funding"}}]}}