{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,23]],"date-time":"2026-07-23T09:00:49Z","timestamp":1784797249732,"version":"3.55.0"},"reference-count":51,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2026,6,17]],"date-time":"2026-06-17T00:00:00Z","timestamp":1781654400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,6,17]],"date-time":"2026-06-17T00:00:00Z","timestamp":1781654400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Mach. Intell. Res."],"published-print":{"date-parts":[[2026,8]]},"DOI":"10.1007\/s11633-025-1602-0","type":"journal-article","created":{"date-parts":[[2026,6,17]],"date-time":"2026-06-17T05:45:54Z","timestamp":1781675154000},"page":"887-902","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Reasoning-targeted Jailbreak Attacks on Large Reasoning Models via Semantic Triggers and Psychological Framing"],"prefix":"10.1007","volume":"23","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-4267-5359","authenticated-orcid":false,"given":"Zehao","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7696-5330","authenticated-orcid":false,"given":"Lanjun","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,6,17]]},"reference":[{"key":"1602_CR1","volume-title":"Deepseek-R1: Incentivizing reasoning capability in LLMs via reinforcement learning","author":"D Guo","year":"2025","unstructured":"D. Guo, D. Yang, H. Zhang, J. Song, R. Zhang, R. Xu, Q. Zhu, S. Ma, P. Wang, X. Bi, X. Zhang, X. Yu, Y. Wu, Z. F. Wu, Z. Gou, Z. Shao, Z. Li, Z. Gao, A. Liu, B. Xue, B. Wang, B. Wu, B. Feng, C. Lu, C. Zhao, C. Deng, C. Zhang, C. Ruan, D. Dai, D. Chen, D. Ji, E. Li, F. Lin, F. Dai, F. Luo, G. Hao, G. Chen, G. Li, H. Zhang, H. Bao, H. Xu, H. Wang, H. Ding, H. Xin, H. Gao, H. Qu, H. Li, J. Guo, J. Li, J. Wang, J. Chen, J. Yuan, J. Qiu, J. Li, J. L. Cai, J. Ni, J. Liang, J. Chen, K. Dong, K. Hu, K. Gao, K. Guan, K. Huang, K. Yu, L. Wang, L. Zhang, L. Zhao, L. Wang, L. Zhang, L. Xu, L. Xia, M. Zhang, M. Zhang, M. Tang, M. Li, M. Wang, M. Li, N. Tian, P. Huang, P. Zhang, Q. Wang, Q. Chen, Q. Du, R. Ge, R. Zhang, R. Pan, R. Wang, R. J. Chen, R. L. Jin, R. Chen, S. Lu, S. Zhou, S. Chen, S. Ye, S. Wang, S. Yu, S. Zhou, S. Pan, S. S. Li, S. Zhou, S. Wu, S. Ye, T. Yun, T. Pei, T. Sun, T. Wang, W. Zeng, W. Zhao, W. Liu, W. Liang, W. Gao, W. Yu, W. Zhang, W. L. Xiao, W. An, X. Liu, X. Wang, X. Chen, X. Nie, X. Cheng, X. Liu, X. Xie, X. Liu, X. Yang, X. Li, X. Su, X. Lin, X. Q. Li, X. Jin, X. Shen, X. Chen, X. Sun, X. Wang, X. Song, X. Zhou, X. Wang, X. Shan, Y. K. Li, Y. Q. Wang, Y. X. Wei, Y. Zhang, Y. Xu, Y. Li, Y. Zhao, Y. Sun, Y. Wang, Y. Yu, Y. Zhang, Y. Shi, Y. Xiong, Y. He, Y. Piao, Y. Wang, Y. Tan, Y. Ma, Y. Liu, Y. Guo, Y. Ou, Y. Wang, Y. Gong, Y. Zou, Y. He, Y. Xiong, Y. Luo, Y. You, Y. Liu, Y. Zhou, Y. X. Zhu, Y. Xu, Y. Huang, Y. Li, Y. Zheng, Y. Zhu, Y. Ma, Y. Tang, Y. Zha, Y. Yan, Z. Z. Ren, Z. Ren, Z. Sha, Z. Fu, Z. Xu, Z. Xie, Z. Zhang, Z. Hao, Z. Ma, Z. Yan, Z. Wu, Z. Gu, Z. Zhu, Z. Liu, Z. Li, Z. Xie, Z. Song, Z. Pan, Z. Huang, Z. Xu, Z. Zhang, Z. Zhang. Deepseek-R1: Incentivizing reasoning capability in LLMs via reinforcement learning, [Online], Available: https:\/\/arxiv.org\/abs\/2501.12948, 2025."},{"issue":"18","key":"1602_CR2","doi-asserted-by":"publisher","first-page":"9516","DOI":"10.1021\/acs.jcim.5c01265","volume":"65","author":"D X Cui","year":"2025","unstructured":"D. X. Cui, S. Y. Long, Y. X. Tang, Y. Zhao, Q. Li. Can reasoning power significantly improve the knowledge of large language models for chemistry?\u2013Based on conversations with LLMs. Journal of Chemical Information and Modeling, vol. 65, no. 18, pp. 9516\u20139527, 2025. DOI: https:\/\/doi.org\/10.1021\/acs.jcim.5c01265.","journal-title":"Journal of Chemical Information and Modeling"},{"key":"1602_CR3","volume-title":"Reasoning with large language models, a survey","author":"A Plaat","year":"2024","unstructured":"A. Plaat, A. Wong, S. Verberne, J. Broekens, N. van Stein, T. Back. Reasoning with large language models, a survey, [Online], Available: https:\/\/arxiv.org\/abs\/2407.11511v1, 2024."},{"key":"1602_CR4","volume-title":"Advancing reasoning in large language models: Promising methods and approaches","author":"A Patil","year":"2025","unstructured":"A. Patil, A. Jadon. Advancing reasoning in large language models: Promising methods and approaches, [Online], Available: https:\/\/arxiv.org\/abs\/2502.03671, 2025."},{"key":"1602_CR5","doi-asserted-by":"publisher","unstructured":"A. Temsah, K. Alhasan, I. Altamimi, A. Jamal, A. Al-Eyadhy, K. H. Malki, M. H. Temsah. Deepseek in healthcare: Revealing opportunities and steering challenges of a new open-source artificial intelligence frontier. Cureus, vol. 17, no. 2, Article number e79221, 2025. DOI: https:\/\/doi.org\/10.7759\/CUREUS.79221.","DOI":"10.7759\/CUREUS.79221"},{"key":"1602_CR6","doi-asserted-by":"publisher","unstructured":"A. Choudhury, Y. Shahsavar, H. Shamszare. User intent to use DeepSeek for health care purposes and their trust in the large language model: Multinational survey study. JMIR Human Factors, vol. 12, Article number 72867, 2025. DOI: https:\/\/doi.org\/10.2196\/72867.","DOI":"10.2196\/72867"},{"issue":"1","key":"1602_CR7","doi-asserted-by":"publisher","first-page":"98","DOI":"10.59652\/jetm.v3i1.439","volume":"3","author":"K T Kotsis","year":"2025","unstructured":"K. T. Kotsis. ChatGPT and DeepSeek evaluate one another for science education. EIKI Journal of Effective Teaching Methods, vol. 3, no. 1, pp. 98\u2013102, 2025. DOI: https:\/\/doi.org\/10.59652\/jetm.v3i1.439.","journal-title":"EIKI Journal of Effective Teaching Methods"},{"issue":"1","key":"1602_CR8","doi-asserted-by":"publisher","first-page":"13","DOI":"10.56540\/jesaf.v4i1.114","volume":"4","author":"A A Q Mohammed","year":"2025","unstructured":"A. A. Q. Mohammed, B. A. Mudhsh, W. R. A. Bin-Hady, A. S. Al-Tamimi. DeepSeek and Grok in the spotlight after ChatGPT in English education: A review study. Journal of English Studies in Arabia Felix, vol. 4, no. 1, pp. 13\u201322, 2025. DOI: https:\/\/doi.org\/10.56540\/jesaf.v4i1.114.","journal-title":"Journal of English Studies in Arabia Felix"},{"key":"1602_CR9","volume-title":"LawPal: A retrieval augmented generation based system for enhanced legal accessibility in India","author":"D Panchal","year":"2025","unstructured":"D. Panchal, A. Gole, V. Narute, R. Joshi. LawPal: A retrieval augmented generation based system for enhanced legal accessibility in India, [Online], Available: https:\/\/arxiv.org\/abs\/2502.16573, 2025."},{"key":"1602_CR10","volume-title":"Evaluating test-time scaling LLMs for legal reasoning: OpenAI o1, DeepSeek-R1, and beyond","author":"Y Hu","year":"2025","unstructured":"Y. Hu, Y. Yu, L. Gan, B. Wei, K. Kuang, F. Wu. Evaluating test-time scaling LLMs for legal reasoning: OpenAI o1, DeepSeek-R1, and beyond, [Online], Available: https:\/\/arxiv.org\/abs\/2503.16040, 2025."},{"key":"1602_CR11","doi-asserted-by":"publisher","unstructured":"A. Kovari. AI for decision support: Balancing accuracy, transparency, and trust across sectors. Information, vol. 15, no. 11, Article number 725, 2024. DOI: https:\/\/doi.org\/10.3390\/info15110725.","DOI":"10.3390\/info15110725"},{"key":"1602_CR12","doi-asserted-by":"publisher","unstructured":"B. Vaassen. AI, opacity, and personal autonomy. Philosophy & Technology, vol. 35, no. 4, Article number 88, 2022. DOI: https:\/\/doi.org\/10.1007\/s13347-022-00577-5.","DOI":"10.1007\/s13347-022-00577-5"},{"key":"1602_CR13","volume-title":"H-CoT: Hijacking the chain-of-thought safety reasoning mechanism to jail-break large reasoning models, including OpenAI o1\/o3, DeepSeek-R1, and Gemini 2.0 flash thinking","author":"M Kuo","year":"2025","unstructured":"M. Kuo, J. Zhang, A. Ding, Q. Wang, L. DiValentin, Y. Bao, W. Wei, H. Li, Y. Chen. H-CoT: Hijacking the chain-of-thought safety reasoning mechanism to jail-break large reasoning models, including OpenAI o1\/o3, DeepSeek-R1, and Gemini 2.0 flash thinking, [Online], Available: https:\/\/arxiv.org\/abs\/2502.12893, 2025."},{"key":"1602_CR14","doi-asserted-by":"publisher","first-page":"3526","DOI":"10.18653\/v1\/2024.findings-naacl.224","volume-title":"Proceedings of Findings of the Association for Computational Linguistics: NAACL","author":"N Xu","year":"2024","unstructured":"N. Xu, F. Wang, B. Zhou, B. Li, C. Xiao, M. Chen. Cognitive overload: Jailbreaking large language models with overloaded logical thinking. In Proceedings of Findings of the Association for Computational Linguistics: NAACL, Mexico City, Mexico, pp. 3526\u20133548, 2024. DOI: https:\/\/doi.org\/10.18653\/v1\/2024.findings-naacl.224."},{"issue":"2","key":"1602_CR15","doi-asserted-by":"publisher","first-page":"180","DOI":"10.1007\/s11633-022-1377-5","volume":"20","author":"Z Zhang","year":"2023","unstructured":"Z. Zhang, G. Xiao, Y. Li, T. Lv, F. Qi, Z. Liu, Y. Wang, X. Jiang, M. Sun. Red alarm for pre-trained models: Universal vulnerability to neuron-level backdoor attacks. Machine Intelligence Research, vol. 20, no. 2, pp. 180\u2013193, 2023. DOI: https:\/\/doi.org\/10.1007\/s11633-022-1377-5.","journal-title":"Machine Intelligence Research"},{"key":"1602_CR16","volume-title":"Obedience to Authority: An Experimental View","author":"S Milgram","year":"1974","unstructured":"S. Milgram. Obedience to Authority: An Experimental View, New York, USA: Harper & Row, 1974."},{"key":"1602_CR17","doi-asserted-by":"publisher","first-page":"199","DOI":"10.1016\/j.copsyc.2015.07.018","volume":"6","author":"C Moore","year":"2015","unstructured":"C. Moore. Moral disengagement. Current Opinion in Psychology, vol. 6, pp. 199\u2013204, 2015. DOI: https:\/\/doi.org\/10.1016\/j.copsyc.2015.07.018.","journal-title":"Current Opinion in Psychology"},{"key":"1602_CR18","volume-title":"Qwen2.5-Vl technical report","author":"S Bai","year":"2025","unstructured":"S. Bai, K. Chen, X. Liu, J. Wang, W. Ge, S. Song, K. Dang, P. Wang, S. Wang, J. Tang, H. Zhong, Y. Zhu, M. Yang, Z. Li, J. Wan, P. Wang, W. Ding, Z. Fu, Y. Xu, J. Ye, X. Zhang, T. Xie, Z. Cheng, H. Zhang, Z. Yang, H. Xu, J. Lin. Qwen2.5-Vl technical report, [Online], Available: https:\/\/arxiv.org\/abs\/2502.13923, 2025."},{"key":"1602_CR19","volume-title":"Large language models for artificial general intelligence (AGI): A survey of foundational principles and approaches","author":"A Mumuni","year":"2025","unstructured":"A. Mumuni, F. Mumuni. Large language models for artificial general intelligence (AGI): A survey of foundational principles and approaches, [Online], Available: https:\/\/arxiv.org\/abs\/2501.03151, 2025."},{"key":"1602_CR20","doi-asserted-by":"publisher","first-page":"27","DOI":"10.1007\/978-981-97-3222-7_2","volume-title":"Artificial General Intelligence (AGI) Security: Smart Applications and Sustainable Technologies","author":"M Fahad","year":"2025","unstructured":"M. Fahad, T. Basri, M. A. Hamza, S. Faisal, A. Akbar, U. Haider, S. El Hajjami. The benefits and risks of artificial general intelligence (AGI). Artificial General Intelligence (AGI) Security: Smart Applications and Sustainable Technologies, S. El Hajjami, K. Kaushik, I. U. Khan, Eds., Singapore: Springer, pp. 27\u201352, 2025. DOI: https:\/\/doi.org\/10.1007\/978-981-97-3222-7_2."},{"key":"1602_CR21","volume-title":"A survey of efficient reasoning for large reasoning models: Language, multimodality, and beyond","author":"X Qu","year":"2025","unstructured":"X. Qu, Y. Li, Z. Su, W. Sun, J. Yan, D. Liu, G. Cui, D. Liu, S. Liang, J. He, P. Li, W. Wei, J. Shao, C. Lu, Y. Zhang, X. S. Hua, B. Zhou, Y. Cheng. A survey of efficient reasoning for large reasoning models: Language, multimodality, and beyond, [Online], Available: https:\/\/arxiv.org\/abs\/2503.21614, 2025."},{"key":"1602_CR22","doi-asserted-by":"publisher","unstructured":"H. Kim, H. Hwang, J. Lee, S. Park, D. Kim, T. Lee, C. Yoon, J. Sohn, J. Park, O. Reykhart, T. Fetherston, D. Choi, S. H. Kwak, Q. Chen, J. Kang. Small language models learn enhanced reasoning skills from medical textbooks. NPJ Digital Medicine, vol. 8, no. 1, Article number 240, 2025. DOI: https:\/\/doi.org\/10.1038\/s41746-025-01653-8.","DOI":"10.1038\/s41746-025-01653-8"},{"issue":"5","key":"1602_CR23","doi-asserted-by":"publisher","first-page":"888","DOI":"10.1007\/s11633-024-1502-8","volume":"21","author":"T Sun","year":"2024","unstructured":"T. Sun, X. Zhang, Z. He, P. Li, Q. Cheng, X. Liu, H. Yan, Y. Shao, Q. Tang, S. Zhang, X. Zhao, K. Chen, Y. Zheng, Z. Zhou, R. Li, J. Zhan, Y. Zhou, L. Li, X. Yang, L. Wu, Z. Yin, X. Huang, Y. G. Jiang, X. Qiu. MOSS: An Open Conversational Large Language Model. Machine Intelligence Research, vol. 21, no. 5, pp. 888\u2013905, 2024. DOI: https:\/\/doi.org\/10.1007\/s11633-024-1502-8.","journal-title":"Machine Intelligence Research"},{"key":"1602_CR24","volume-title":"Proceedings of the 36th International Conference on Neural Information Processing Systems","author":"J Wei","year":"2022","unstructured":"J. Wei, X. Wang, D. Schuurmans, M. Bosma, B. Ichter, F. Xia, E. H. Chi, Q. V. Le, D. Zhou. Chain-of-thought prompting elicits reasoning in large language models. In Proceedings of the 36th International Conference on Neural Information Processing Systems, New Orleans, USA, Article number 1800, 2022."},{"key":"1602_CR25","volume-title":"Proceedings of the 38th International Conference on Neural Information Processing Systems","author":"X Zhang","year":"2024","unstructured":"X. Zhang, C. Dut, T. Pang, Q. Liu, W. Gao, M. Lin. Chain of preference optimization: Improving chain-of-thought reasoning in LLMs. In Proceedings of the 38th International Conference on Neural Information Processing Systems, Vancouver, Canada, Article number 11, 2024."},{"key":"1602_CR26","volume-title":"Proceedings of the 38th International Conference on Neural Information Processing Systems","author":"Z Xiao","year":"2024","unstructured":"Z. Xiao, D. Zhang, X. Han, X. Fu, W. Y. Yu, T. Zhong, S. Wu, Y. Wang, J. Yin, G. Chen. Enhancing LLM reasoning via vision-augmented prompting. In Proceedings of the 38th International Conference on Neural Information Processing Systems, Vancouver, Canada, Article number 905, 2024."},{"key":"1602_CR27","doi-asserted-by":"publisher","first-page":"9989","DOI":"10.18653\/v1\/2025.findings-acl.520","volume-title":"Proceedings of Findings of the Association for Computational Linguistics","author":"H Jin","year":"2025","unstructured":"H. Jin, J. W. Yeom, S. Bae, T. Kim. \u201cWell, keep thinking\u201d: Enhancing LLM reasoning with adaptive injection decoding. In Proceedings of Findings of the Association for Computational Linguistics, Vienna, Austria, pp. 9989\u201310018, 2025. DOI: https:\/\/doi.org\/10.18653\/v1\/2025.findings-acl.520."},{"key":"1602_CR28","doi-asserted-by":"publisher","first-page":"24821","DOI":"10.1609\/aaai.v39i23.34664","volume-title":"Proceedings of the 39th AAAI Conference on Artificial Intelligence","author":"Z Ma","year":"2025","unstructured":"Z. Ma, Z. Huang, J. Liu, M. Wang, H. Zhao, X. Li. Automated creation of reusable and diverse toolsets for enhancing LLM reasoning. In Proceedings of the 39th AAAI Conference on Artificial Intelligence, Philadelphia, USA, pp. 24821\u201324830, 2025. DOI: https:\/\/doi.org\/10.1609\/aaai.v39i23.34664."},{"key":"1602_CR29","volume-title":"Proceedings of the 11th International Conference on Learning Representations","author":"S Yao","year":"2023","unstructured":"S. Yao, J. Zhao, D. Yu, N. Du, I. Shafran, K. R. Narasimhan, Y. Cao. ReAct: Synergizing reasoning and acting in language models. In Proceedings of the 11th International Conference on Learning Representations, Kigali, Rwanda, 2023."},{"key":"1602_CR30","volume-title":"RPRM: Reasoning-driven process reward modeling","author":"S She","year":"2025","unstructured":"S. She, J. Liu, Y. Liu, J. Chen, X. Huang, S. Huang. RPRM: Reasoning-driven process reward modeling, [Online], Available: https:\/\/arxiv.org\/abs\/2503.21295, 2025."},{"key":"1602_CR31","volume-title":"Are smarter LLMs safer? Exploring safety-reasoning trade-offs in prompting and fine-tuning","author":"A Li","year":"2025","unstructured":"A. Li, Y. Mo, M. Li, Y. Wang, Y. Wang. Are smarter LLMs safer? Exploring safety-reasoning trade-offs in prompting and fine-tuning, [Online], Available: https:\/\/arxiv.org\/abs\/2502.09673, 2025."},{"key":"1602_CR32","volume-title":"Safety tax: Safety alignment makes your large reasoning models less reasonable","author":"T Huang","year":"2025","unstructured":"T. Huang, S. Hu, F. Ilhan, S. F. Tekin, Z. Yahn, Y. Xu, L. Liu. Safety tax: Safety alignment makes your large reasoning models less reasonable, [Online], Available: https:\/\/arxiv.org\/abs\/2503.00555, 2025."},{"key":"1602_CR33","volume-title":"Safety in large reasoning models: A survey","author":"C Wang","year":"2025","unstructured":"C. Wang, Y. Liu, B. Bi, D. Zhang, Z. Z. Li, Y. Ma, Y. He, S. Yu, X. Li, J. Fang, J. Zhang, B. Hooi. Safety in large reasoning models: A survey, [Online], Available: https:\/\/arxiv.org\/abs\/2504.17704, 2025."},{"key":"1602_CR34","volume-title":"How should we enhance the safety of large reasoning models: An empirical study","author":"Z Zhang","year":"2025","unstructured":"Z. Zhang, X. Q. Loye, V. S. J. Huang, J. Yang, Q. Zhu, S. Cui, F. Mi, L. Shang, Y. Wang, H. Wang, M. Huang. How should we enhance the safety of large reasoning models: An empirical study, [Online], Available: https:\/\/arxiv.org\/abs\/2505.15404, 2025."},{"key":"1602_CR35","volume-title":"Large language models as software components: A taxonomy for LLM-integrated applications","author":"I Weber","year":"2024","unstructured":"I. Weber. Large language models as software components: A taxonomy for LLM-integrated applications, [Online], Available: https:\/\/arxiv.org\/abs\/2406.10300, 2024."},{"key":"1602_CR36","doi-asserted-by":"publisher","first-page":"416","DOI":"10.1109\/TLT.2025.3561332","volume":"18","author":"X Zhang","year":"2025","unstructured":"X. Zhang, C. Zhang, J. Sun, J. Xiao, Y. Yang, Y. Luo. EduPlanner: LLM-based multiagent systems for customized and intelligent instructional design. IEEE Transactions on Learning Technologies, vol. 18, pp. 416\u2013427, 2025. DOI: https:\/\/doi.org\/10.1109\/TLT.2025.3561332.","journal-title":"IEEE Transactions on Learning Technologies"},{"key":"1602_CR37","volume-title":"From LLMs to MLLMs to agents: A survey of emerging paradigms in jailbreak attacks and defenses within LLM ecosystem","author":"Y Mao","year":"2025","unstructured":"Y. Mao, T. Cui, P. Liu, D. You, H. Zhu. From LLMs to MLLMs to agents: A survey of emerging paradigms in jailbreak attacks and defenses within LLM ecosystem, [Online], Available: https:\/\/arxiv.org\/abs\/2506.15170, 2025."},{"key":"1602_CR38","volume-title":"Attack and defense techniques in large language models: A survey and new perspectives","author":"Z Liao","year":"2025","unstructured":"Z. Liao, K. Chen, Y. Lin, K. Li, Y. Liu, H. Chen, X. Huang, Y. Yu. Attack and defense techniques in large language models: A survey and new perspectives, [Online], Available: https:\/\/arxiv.org\/abs\/2505.00976, 2025."},{"key":"1602_CR39","volume-title":"Breaking down the defenses: A comparative survey of attacks on large language models","author":"A G Chowdhury","year":"2024","unstructured":"A. G. Chowdhury, M. M. Islam, V. Kumar, F. H. Shezan, V. Kumar, V. Jain, A. Chadha. Breaking down the defenses: A comparative survey of attacks on large language models, [Online], Available: https:\/\/arxiv.org\/abs\/2403.04786, 2024."},{"key":"1602_CR40","doi-asserted-by":"publisher","first-page":"20754","DOI":"10.18653\/v1\/2025.findingsacl.1067","volume-title":"Proceedings of Findings of the Association for Computational Linguistics","author":"T Cui","year":"2025","unstructured":"T. Cui, Y. Mao, P. Liu, C. Liu, D. You. Exploring jail-break attacks on LLMs through intent concealment and diversion. In Proceedings of Findings of the Association for Computational Linguistics, Vienna, Austria, pp. 20754\u201320768, 2025. DOI: https:\/\/doi.org\/10.18653\/v1\/2025.findingsacl.1067."},{"key":"1602_CR41","doi-asserted-by":"publisher","DOI":"10.1117\/12.3049302","volume-title":"Proceedings of SPIE 13395, International Conference on Optics, Electronics, and Communication Engineering","author":"D Zhang","year":"2024","unstructured":"D. Zhang, Z. Hu, H. Chen, G. Liu, F. Li, J. Lu. Cognitive pitfalls of LLMs: A system for generating adversarial samples based on cognitive biases. In Proceedings of SPIE 13395, International Conference on Optics, Electronics, and Communication Engineering, Wuhan, China, Article number 133954O, 2024. DOI: https:\/\/doi.org\/10.1117\/12.3049302."},{"key":"1602_CR42","volume-title":"Proceedings of the 38th International Conference on Neural Information Processing Systems","author":"A Mehrotra","year":"2024","unstructured":"A. Mehrotra, M. Zampetakis, P. Kassianik, B. Nelson, H. Anderson, Y. Singer, A. Karbasi. Tree of attacks: Jail-breaking black-box LLMs automatically. In Proceedings of the 38th International Conference on Neural Information Processing Systems, Vancouver, Canada, Article number 1952, 2024."},{"key":"1602_CR43","doi-asserted-by":"publisher","DOI":"10.1002\/9780470672532.wbepp165","volume-title":"The Encyclopedia of Peace Psychology","author":"A Bandura","year":"2011","unstructured":"A. Bandura. Moral disengagement. The Encyclopedia of Peace Psychology, D. J. Christie, Ed., Hoboken, USA: John Wiley & Sons, 2011. DOI: https:\/\/doi.org\/10.1002\/9780470672532.wbepp165."},{"key":"1602_CR44","volume-title":"The instruction hierarchy: Training LLMs to prioritize privileged instructions","author":"E Wallace","year":"2024","unstructured":"E. Wallace, K. Xiao, R. Leike, L. Weng, J. Heidecke, A. Beutel. The instruction hierarchy: Training LLMs to prioritize privileged instructions, [Online], Available: https:\/\/arxiv.org\/abs\/2404.13208, 2024."},{"key":"1602_CR45","doi-asserted-by":"publisher","first-page":"4149","DOI":"10.18653\/v1\/N19-1421","volume-title":"Proceedings of Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies","author":"A Talmor","year":"2019","unstructured":"A. Talmor, J. Herzig, N. Lourie, J. Berant. CommonsenseQA: A question answering challenge targeting commonsense knowledge. In Proceedings of Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Minneapolis, USA, pp. 4149\u20134158, 2019. DOI: https:\/\/doi.org\/10.18653\/v1\/N19-1421."},{"key":"1602_CR46","doi-asserted-by":"publisher","first-page":"346","DOI":"10.1162\/tacl_a_00370","volume-title":"Proceedings of Transactions of the Association for Computational Linguistics","author":"M Geva","year":"2021","unstructured":"M. Geva, D. Khashabi, E. Segal, T. Khot, D. Roth, J. Berant. Did Aristotle use a laptop? A question answering benchmark with implicit reasoning strategies. In Proceedings of Transactions of the Association for Computational Linguistics, Cambridge, USA, pp. 346\u2013361, 2021. DOI: https:\/\/doi.org\/10.1162\/tacl_a_00370."},{"key":"1602_CR47","doi-asserted-by":"publisher","first-page":"13697","DOI":"10.18653\/v1\/2024.findings-acl.813","volume-title":"Proceedings of Findings of the Association for Computational Linguistics","author":"T Vu","year":"2024","unstructured":"T. Vu, M. Iyyer, X. Wang, N. Constant, J. Wei, J. Wei, C. Tar, Y. H. Sung, D. Zhou, Q. Le, T. Luong. FreshLLMs: Refreshing large language models with search engine augmentation. In Proceedings of Findings of the Association for Computational Linguistics, Bangkok, Thailand, pp. 13697\u201313720, 2024. DOI: https:\/\/doi.org\/10.18653\/v1\/2024.findings-acl.813."},{"key":"1602_CR48","doi-asserted-by":"publisher","unstructured":"D. Jin, E. Pan, N. Oufattole, W. H. Weng, H. Fang, P. Szolovits. What disease does this patient have? A large-scale open domain question answering dataset from medical exams. Applied Sciences, vol. 11, no. 14, Article number 6421, 2021. DOI: https:\/\/doi.org\/10.3390\/app11146421.","DOI":"10.3390\/app11146421"},{"key":"1602_CR49","volume-title":"GPT-4o system card","author":"A Hurst","year":"2024","unstructured":"A. Hurst, A. Lerer, A. P. Goucher, A. Perelman, A. Ramesh, A. Clark, A. J. Ostrow, A. Welihinda, A. Hayes, A. Radford, A. M\u0105dry, A. Baker-Whitcomb, A. Beutel, A. Borzunov, A. Carney, A. Chow, A. Kirillov, A. Nichol, A. Paino, A. Renzin, A. T. Passos, A. Kirillov, A. Christakis, A. Conneau, A. Kamali, A. Jabri, A. Moyer, A. Tam, A. Crookes, A. Tootoochian, A. Tootoonchian, A. Kumar, A. Vallone, A. Karpathy, A. Braunstein, A. Cann, A. Codispoti, A. Galu, A. Kondrich, A. Tulloch, A. Mishchenko, A. Baek, A. Jiang, A. Pelisse, A. Woodford, A. Gosalia, A. Dhar, A. Pantuliano, A. Nayak, A. Oliver, B. Zoph, B. Ghorbani, B. Leimberger, B. Rossen, B. Sokolowsky, B. Wang, B. Zweig, B. Hoover, B. Samic, B. McGrew, B. Spero, B. Giertler, B. Cheng, B. Lightcap, B. Walkin, B. Quinn, B. Guarraci, B. Hsu, B. Kellogg, B. Eastman, C. Lugaresi, C. Wainwright, C. Bassin, C. Hudson, C. Chu, C. Nelson, C. Li, C. J. Shern, C. Conger, C. Barette, C. Voss, C. Ding, C. Lu, C. Zhang, C. Beaumont, C. Hallacy, C. Koch, C. Gibson, C. Kim, C. Choi, C. McLeavey, C. Hesse, C. Fischer, C. Winter, C. Czarnecki, C. Jarvis, C. Wei, C. Koumouzelis, D. Sherburn, D. Kappler, D. Levin, D. Levy, D. Carr, D. Farhi, D. Mely, D. Robinson, D. Sasaki, D. Jin, D. Valladares, D. Tsipras, D. Li, D. P. Nguyen, D. Findlay, E. Oiwoh, E. Wong, E. Asdar, E. Proehl, E. Yang, E. Antonow, E. Kramer, E. Peterson, E. Sigler, E. Wallace, E. Brevdo, E. Mays, F. Khorasani, F. P. Such, F. Raso, F. Zhang, F. von Lohmann, F. Sulit, G. Goh, G. Oden, G. Salmon, G. Starace, G. Brockman, H. Salman, H. Bao, H. Hu, H. Wong, H. Wang, H. Schmidt, H. Whitney, H. Jun, H. Kirchner, H. P. de Oliveira Pinto, H. Ren, H. Chang, H. W. Chung, I. Kivlichan, I. O\u2019Connell, I. O\u2019Connell, I. Osband, I. Silber, I. Sohl, I. Okuyucu, I. Lan, I. Kostrikov, I. Sutskever, I. Kanitscheider, I. Gulrajani, J. Coxon, J. Menick, J. Pachocki, J. Aung, J. Betker, J. Crooks, J. Lennon, J. Kiros, J. Leike, J. Park, J. Kwon, J. Phang, J. Teplitz, J. Wei, J. Wolfe, J. Chen, J. Harris, J. Varavva, J. G. Lee, J. Shieh, J. Lin, J. Yu, J. Weng, J. Tang, J. Yu, J. Jang, J. Q. Candela, J. Beutler, J. Landers, J. Parish, J. Heidecke, J. Schulman, J. Lachman, J. McKay, J. Uesato, J. Ward, J. W. Kim, J. Huizinga, J. Sitkin, J. Kraaijeveld, J. Gross, J. Kaplan, J. Snyder, J. Achiam, J. Jiao, J. Lee, J. Zhuang, J. Harriman, K. Fricke, K. Hayashi, K. Singhal, K. Shi, K. Karthik, K. Wood, K. Rimbach, K. Hsu, K. Nguyen, K. Gu-Lemberg, K. Button, K. Liu, K. Howe, K. Muthukumar, K. Luther, L. Ahmad, L. Kai, L. Itow, L. Workman, L. Pathak, L. Chen, L. Jing, L. Guy, L. Fedus, L. Zhou, L. Mamitsuka, L. Weng, L. McCallum, L. Held, L. Ouyang, L. Feuvrier, L. Zhang, L. Kondraciuk, L. Kaiser, L. Hewitt, L. Metz, L. Doshi, M. Aflak, M. Simens, M. Boyd, M. Thompson, M. Dukhan, M. Chen, M. Gray, M. Hudnall, M. Zhang, M. Aljubeh, M. Litwin, M. Zeng, M. Johnson, M. Shetty, M. Gupta, M. Shah, M. Yatbaz, M. J. Yang, M. Zhong, M. Glaese, M. Chen, M. Janner, M. Lampe, M. Petrov, M. Wu, M. Wang, M. Fradin, M. Pokrass, M. Castro, M. O. T. de Castro, M. Pavlov, M. Brundage, M. Wang, M. Khan, M. Murati, M. Bavarian, M. Lin, M. Yesildal, N. Soto, N. Gimelshein, N. Cone, N. Staudacher, N. Summers, N. LaFontaine, N. Chowdhury, N. Ryder, N. Stathas, N. Turley, N. Tezak, N. Felix, N. Kudige, N. Keskar, N. Deutsch, N. Bundick, N. Puckett, O. Nachum, O. Okelola, O. Boiko, O. Murk, O. Jaffe, O. Watkins, O. Godement, O. Campbell-Moore, P. Chao, P. McMillan, P. Belov, P. Su, P. Bak, P. Bakkum, P. Deng, P. Dolan, P. Hoeschele, P. Welinder, P. Tillet, P. Pronin, P. Tillet, P. Dhariwal, Q. Yuan, R. Dias, R. Lim, R. Arora, R. Troll, R. Lin, R. G. Lopes, R. Puri, R. Miyara, R. Leike, R. Gaubert, R. Zamani, R. Wang, R. Donnelly, R. Honsby, R. Smith, R. Sahai, R. Ramchandani, R. Huet, R. Carmichael, R. Zellers, R. Chen, R. Chen, R. Nigmatullin, R. Cheu, S. Jain, S. Altman, S. Schoenholz, S. Toizer, S. Miserendino, S. Agarwal, S. Culver, S. Ethersmith, S. Gray, S. Grove, S. Metzger, S. Hermani, S. Jain, S. Zhao, S. Wu, S. Jomoto, S. Wu, S. Xia, S. Phene, S. Papay, S. Narayanan, S. Coffey, S. Lee, S. Hall, S. Balaji, T. Broda, T. Stramer, T. Xu, T. Gogineni, T. Christianson, T. Sanders, T. Patwardhan, T. Cunninghman, T. Degry, T. Dimson, T. Raoux, T. Shadwell, T. Zheng, T. Underwood, T. Markov, T. Sherbakov, T. Rubin, T. Stasi, T. Kaftan, T. Heywood, T. Peterson, T. Walters, T. Eloundou, V. Qi, V. Moeller, V. Monaco, V. Kuo, V. Fomenko, W. Chang, W. Zheng, W. Zhou, W. Manassra, W. Sheu, W. Zaremba, Y. Patil, Y. Qian, Y. Kim, Y. Cheng, Y. Zhang, Y. He, Y. Zhang, Y. Jin, Y. Dai, Y. Malkov. GPT-4o system card, [Online], Available: https:\/\/arxiv.org\/abs\/2410.21276, 2024."},{"key":"1602_CR50","volume-title":"Towards understanding the safety boundaries of DeepSeek models: Evaluation and findings","author":"Z Ying","year":"2025","unstructured":"Z. Ying, G. Zheng, Y. Huang, D. Zhang, W. Zhang, Q. Zou, A. Liu, X. Liu, D. Tao. Towards understanding the safety boundaries of DeepSeek models: Evaluation and findings, [Online], Available: https:\/\/arxiv.org\/abs\/2503.15092, 2025."},{"key":"1602_CR51","volume-title":"A comprehensive review of Qwen and Deep\u2013Seek LLMs: Architecture, performance and applications","author":"S Joshi","year":"2025","unstructured":"S. Joshi. A comprehensive review of Qwen and Deep\u2013Seek LLMs: Architecture, performance and applications, [Online], Available: https:\/\/sciety.org\/articles\/activity\/10.20944\/preprints202505.2064.v1, 2025."}],"container-title":["Machine Intelligence Research"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11633-025-1602-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11633-025-1602-0","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11633-025-1602-0.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,23]],"date-time":"2026-07-23T08:02:39Z","timestamp":1784793759000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11633-025-1602-0"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,17]]},"references-count":51,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2026,8]]}},"alternative-id":["1602"],"URL":"https:\/\/doi.org\/10.1007\/s11633-025-1602-0","relation":{},"ISSN":["2731-538X","2731-5398"],"issn-type":[{"value":"2731-538X","type":"print"},{"value":"2731-5398","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,6,17]]},"assertion":[{"value":"28 June 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 September 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 June 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors declared that they have no conflicts of interest to this work.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations of conflict of interest"}}]}}