{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,22]],"date-time":"2025-06-22T04:04:15Z","timestamp":1750565055822,"version":"3.41.0"},"publisher-location":"Singapore","reference-count":22,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819681969","type":"print"},{"value":"9789819681976","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-96-8197-6_21","type":"book-chapter","created":{"date-parts":[[2025,6,21]],"date-time":"2025-06-21T17:19:00Z","timestamp":1750526340000},"page":"286-297","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Round Trip Translation Defence Against Large Language Model Jailbreaking Attacks"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-5726-8317","authenticated-orcid":false,"given":"Canaan","family":"Yung","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9418-1487","authenticated-orcid":false,"given":"Hadi Mohaghegh","family":"Dolatabadi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0885-0643","authenticated-orcid":false,"given":"Sarah","family":"Erfani","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4388-0517","authenticated-orcid":false,"given":"Christopher","family":"Leckie","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,6,22]]},"reference":[{"issue":"1","key":"21_CR1","first-page":"1","volume":"14","author":"M Aiken","year":"2010","unstructured":"Aiken, M., Park, M.: The efficacy of round-trip translation for mt evaluation. Translation J. 14(1), 1\u201310 (2010)","journal-title":"Translation J."},{"key":"21_CR2","unstructured":"Barham, S., Feizi, S.: Interpretable adversarial training for text (2019). arXiv preprint arXiv:1905.12864"},{"key":"21_CR3","doi-asserted-by":"crossref","unstructured":"Cao, B., Cao, Y., Lin, L., Chen, J.: Defending against alignment-breaking attacks via robustly aligned llm (2023). arXiv preprint arXiv:2309.14348","DOI":"10.18653\/v1\/2024.acl-long.568"},{"key":"21_CR4","unstructured":"Chao, P., Robey, A., Dobriban, E., Hassani, H., Pappas, G.J., Wong, E.: Jailbreaking black box large language models in twenty queries (2023). arXiv preprint arXiv:2310.08419"},{"key":"21_CR5","unstructured":"Chu, J., Liu, Y., Yang, Z., Shen, X., Backes, M., Zhang, Y.: Comprehensive assessment of jailbreak attacks against llms(2024). arXiv preprint arXiv:2402.05668"},{"key":"21_CR6","unstructured":"Cobbe, K., et al.: Training verifiers to solve math word problems (2021). arXiv preprint arXiv:2110.14168"},{"key":"21_CR7","doi-asserted-by":"crossref","unstructured":"Deng, B., Wang, W., Feng, F., Deng, Y., Wang, Q., He, X.: Attack prompt generation for red teaming and defending large language models (2023). arXiv preprint arXiv:2310.12505","DOI":"10.18653\/v1\/2023.findings-emnlp.143"},{"key":"21_CR8","unstructured":"Edelsbrunner, H., Harer, J.L.: Computational topology: an introduction. American Mathematical Society (2022)"},{"key":"21_CR9","unstructured":"Goldsmith, J.: Wikipedia (Apr 2022). https:\/\/github.com\/goldsmith\/Wikipedia"},{"key":"21_CR10","unstructured":"Google: Google translate api (2023). http:\/\/translate.google.com"},{"key":"21_CR11","doi-asserted-by":"crossref","unstructured":"Ji, J., et al.: Advancing the robustness of large language models through self-denoised smoothing. arXiv preprint arXiv:2404.12274 (2024)","DOI":"10.18653\/v1\/2024.naacl-short.23"},{"key":"21_CR12","unstructured":"Mazeika, M., et\u00a0al.: Harmbench: A standardized evaluation framework for automated red teaming and robust refusal. arXiv preprint arXiv:2402.04249 (2024)"},{"key":"21_CR13","doi-asserted-by":"crossref","unstructured":"McMahon, A., McMahon, R.: Language classification by numbers. Oxford University Press (2005)","DOI":"10.1093\/oso\/9780199279012.001.0001"},{"key":"21_CR14","unstructured":"Mitra, A., Khanpour, H., Rosset, C., Awadallah, A.: Orca-math: unlocking the potential of slms in grade school math (2024)"},{"key":"21_CR15","unstructured":"Oxford: Oxford learner\u2019s dictionary (2023). https:\/\/www.oxfordlearnersdictionaries.com\/"},{"key":"21_CR16","unstructured":"Robey, A., Wong, E., Hassani, H., Pappas, G.J.: Smoothllm: defending large language models against jailbreaking attacks. arXiv preprint arXiv:2310.03684 (2023)"},{"key":"21_CR17","unstructured":"Sheshadri, A., et\u00a0al.: Targeted latent adversarial training improves robustness to persistent harmful behaviors in llms. arXiv preprint arXiv:2407.15549 (2024)"},{"key":"21_CR18","unstructured":"Tulchinskii, E., et al.: Intrinsic dimension estimation for robust detection of ai-generated texts. Adv. Neural Inform. Process. Syst. 36 (2024)"},{"key":"21_CR19","doi-asserted-by":"crossref","unstructured":"Xu, Z., Liu, Y., Deng, G., Li, Y., Picek, S.: A comprehensive study of jailbreak attack versus defense for large language models. In: Proceedings of the Findings of the Association for Computational Linguistics ACL 2024, pp. 7432\u20137449 (2024)","DOI":"10.18653\/v1\/2024.findings-acl.443"},{"key":"21_CR20","unstructured":"Zhang, Y., Ding, L., Zhang, L., Tao, D.: Intention analysis prompting makes large language models a good jailbreak defender. arXiv preprint arXiv:2401.06561 (2024)"},{"key":"21_CR21","unstructured":"Zhou, Z., et al.: Mathattack: Attacking large language models towards math solving ability. arXiv preprint arXiv:2309.01686 (2023)"},{"key":"21_CR22","unstructured":"Zou, A., Wang, Z., Kolter, J.Z., Fredrikson, M.: Universal and transferable adversarial attacks on aligned language models. arXiv preprint arXiv:2307.15043 (2023)"}],"container-title":["Lecture Notes in Computer Science","Trends and Applications in Knowledge Discovery and Data Mining"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-96-8197-6_21","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,21]],"date-time":"2025-06-21T17:19:07Z","timestamp":1750526347000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-96-8197-6_21"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9789819681969","9789819681976"],"references-count":22,"URL":"https:\/\/doi.org\/10.1007\/978-981-96-8197-6_21","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"22 June 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"PAKDD","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Pacific-Asia Conference on Knowledge Discovery and Data Mining","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Sydney, NSW","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Australia","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"10 June 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"13 June 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"pakdd2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/pakdd2025.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}