{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,22]],"date-time":"2026-06-22T23:07:38Z","timestamp":1782169658037,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":24,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,2,22]]},"DOI":"10.1145\/3779211.3795738","type":"proceedings-article","created":{"date-parts":[[2026,5,5]],"date-time":"2026-05-05T11:33:48Z","timestamp":1777980828000},"page":"159-173","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Joint Evaluation : A Human + LLM + Multi-Agents Collaborative Framework for Comprehensive AI Safety (Jo.E)"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9950-4625","authenticated-orcid":false,"given":"Himanshu","family":"Joshi","sequence":"first","affiliation":[{"name":"Science and Technology, Northeastern University, Toronto, Ontario, Canada"},{"name":"CS and AI, Golden Gate University, San Francisco, California, USA"},{"name":"Sloan School of Management, MIT, Cambridge, Massachusetts, USA"},{"name":"AI Research, COHUMAIN Labs, Toronto, Ontario, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,5,5]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"On the opportunities and risks of foundation models. arXiv preprint arXiv:2108.07258","author":"Bommasani Rishi","year":"2021","unstructured":"Rishi Bommasani, Drew A Hudson, Ehsan Adeli, et al. On the opportunities and risks of foundation models. arXiv preprint arXiv:2108.07258, 2021."},{"key":"e_1_3_2_1_2_1","volume-title":"Ethical and social risks of harm from language models. arXiv preprint arXiv:2112.04359","author":"Weidinger Laura","year":"2021","unstructured":"Laura Weidinger, John Mellor, Maribeth Rauh, et al. Ethical and social risks of harm from language models. arXiv preprint arXiv:2112.04359, 2021."},{"key":"e_1_3_2_1_3_1","volume-title":"Red teaming language models to reduce harms: Methods, scaling behaviors, and lessons learned. arXiv preprint arXiv:2209.07858","author":"Ganguli Deep","year":"2022","unstructured":"Deep Ganguli, Liane Lovitt, Jackson Kernion, et al. Red teaming language models to reduce harms: Methods, scaling behaviors, and lessons learned. arXiv preprint arXiv:2209.07858, 2022."},{"key":"e_1_3_2_1_4_1","volume-title":"Proceedings of NeurIPS","author":"Zheng Lianmin","year":"2023","unstructured":"Lianmin Zheng, Wei-Lin Chiang, Ying Sheng, et al. Judging LLM-as-a-judge with MT-Bench and Chatbot Arena. In Proceedings of NeurIPS, 2023."},{"key":"e_1_3_2_1_5_1","first-page":"3448","volume-title":"Proceedings of EMNLP","author":"Perez Ethan","year":"2022","unstructured":"Ethan Perez, Sam Huang, Francis Song, et al. Red teaming language models with language models. In Proceedings of EMNLP, pages 3419\u20133448, 2022."},{"key":"e_1_3_2_1_6_1","volume-title":"Proceedings of ICML","author":"Mazeika Mantas","year":"2024","unstructured":"Mantas Mazeika, Long Phan, Xuwang Yin, et al. HarmBench: A standardized evaluation framework for automated red teaming and robust refusal. In Proceedings of ICML, 2024."},{"key":"e_1_3_2_1_7_1","volume-title":"Measuring progress on scalable oversight for large language models. arXiv preprint arXiv:2211.03540","author":"Bowman Samuel R","year":"2022","unstructured":"Samuel R Bowman, Jeeyoon Hyun, Ethan Perez, et al. Measuring progress on scalable oversight for large language models. arXiv preprint arXiv:2211.03540, 2022."},{"key":"e_1_3_2_1_8_1","volume-title":"Proceedings of COLM","author":"Dubois Yann","year":"2024","unstructured":"Yann Dubois, Balazs Galambosi, Percy Liang, and Tatsunori B Hashimoto. Length-controlled AlpacaEval: A simple way to debias automatic evaluators. In Proceedings of COLM, 2024."},{"key":"e_1_3_2_1_9_1","first-page":"2522","volume-title":"Proceedings of EMNLP","author":"Liu Yang","year":"2023","unstructured":"Yang Liu, Dan Iter, Yichong Xu, Shuohang Wang, Ruochen Xu, and Chenguang Zhu. G-Eval: NLG evaluation using GPT-4 with better human alignment. In Proceedings of EMNLP, pages 2511\u20132522, 2023."},{"key":"e_1_3_2_1_10_1","volume-title":"Proceedings of ICLR","author":"Zhu Lianghui","year":"2024","unstructured":"Lianghui Zhu, Xinggang Wang, and Xinlong Wang. JudgeLM: Fine-tuned large language models are scalable judges. In Proceedings of ICLR, 2024."},{"key":"e_1_3_2_1_11_1","volume-title":"Proceedings of ICLR","author":"Chao Patrick","year":"2024","unstructured":"Patrick Chao, Alexander Robey, Edgar Dobriban, Hamed Hassani, George J Pappas, and Eric Wong. Jailbreaking black box large language models in twenty queries. In Proceedings of ICLR, 2024."},{"key":"e_1_3_2_1_12_1","volume-title":"Tree of attacks: Jailbreaking black-box LLMs automatically. arXiv preprint arXiv:2312.02119","author":"Mehrotra Anay","year":"2023","unstructured":"Anay Mehrotra, Manolis Zampetakis, Paul Kassianik, et al. Tree of attacks: Jailbreaking black-box LLMs automatically. arXiv preprint arXiv:2312.02119, 2023."},{"key":"e_1_3_2_1_13_1","volume-title":"Proceedings of ICLR","author":"Liu Xiaogeng","year":"2024","unstructured":"Xiaogeng Liu, Nan Xu, Muhao Chen, and Chaowei Xiao. AutoDAN: Generating stealthy jailbreak prompts on aligned large language models. In Proceedings of ICLR, 2024."},{"key":"e_1_3_2_1_14_1","volume-title":"Universal and transferable adversarial attacks on aligned language models. arXiv preprint arXiv:2307.15043","author":"Zou Andy","year":"2023","unstructured":"Andy Zou, Zifan Wang, Nicholas Carlini, Milad Nasr, J Zico Kolter, and Matt Fredrikson. Universal and transferable adversarial attacks on aligned language models. arXiv preprint arXiv:2307.15043, 2023."},{"key":"e_1_3_2_1_15_1","volume-title":"NeurIPS Datasets and Benchmarks","author":"Chao Patrick","year":"2024","unstructured":"Patrick Chao, Edoardo Debenedetti, Alexander Robey, et al. JailbreakBench: An open robustness benchmark for jailbreaking large language models. In NeurIPS Datasets and Benchmarks, 2024."},{"key":"e_1_3_2_1_16_1","first-page":"3252","volume-title":"Proceedings of ACL","author":"Lin Stephanie","year":"2022","unstructured":"Stephanie Lin, Jacob Hilton, and Owain Evans. TruthfulQA: Measuring how models mimic human falsehoods. In Proceedings of ACL, pages 3214\u20133252, 2022."},{"key":"e_1_3_2_1_17_1","first-page":"2105","volume-title":"Findings of ACL","author":"Parrish Alicia","year":"2022","unstructured":"Alicia Parrish, Angelica Chen, Nikita Nangia, et al. BBQ: A hand-built bias benchmark for question answering. In Findings of ACL, pages 2086\u20132105, 2022."},{"key":"e_1_3_2_1_18_1","volume-title":"Proceedings of ACL","author":"Zhang Zhexin","year":"2024","unstructured":"Zhexin Zhang, Leqi Lei, Lindong Wu, et al. SafetyBench: Evaluating the safety of large language models with multiple choice questions. In Proceedings of ACL, 2024."},{"key":"e_1_3_2_1_19_1","volume-title":"et al. Constitutional AI: Harmlessness from AI feedback. arXiv preprint arXiv:2212.08073","author":"Bai Yuntao","year":"2022","unstructured":"Yuntao Bai, Saurav Kadavath, Sandipan Kundu, et al. Constitutional AI: Harmlessness from AI feedback. arXiv preprint arXiv:2212.08073, 2022."},{"key":"e_1_3_2_1_20_1","volume-title":"Proceedings of ICML","author":"Lee Harrison","year":"2024","unstructured":"Harrison Lee, Samrat Phatale, Hassan Mansoor, et al. RLAIF vs. RLHF: Scaling reinforcement learning from human feedback with AI feedback. In Proceedings of ICML, 2024."},{"key":"e_1_3_2_1_21_1","volume-title":"AI safety via debate. arXiv preprint arXiv:1805.00899","author":"Irving Geoffrey","year":"2018","unstructured":"Geoffrey Irving, Paul Christiano, and Dario Amodei. AI safety via debate. arXiv preprint arXiv:1805.00899, 2018."},{"key":"e_1_3_2_1_22_1","volume-title":"Transactions on Machine Learning Research","author":"Wei Jason","year":"2022","unstructured":"Jason Wei, Yi Tay, Rishi Bommasani, et al. Emergent abilities of large language models. Transactions on Machine Learning Research, 2022."},{"key":"e_1_3_2_1_23_1","first-page":"3992","volume-title":"Proceedings of EMNLP-IJCNLP","author":"Reimers Nils","year":"2019","unstructured":"Nils Reimers and Iryna Gurevych. Sentence-BERT: Sentence embeddings using Siamese BERT-networks. In Proceedings of EMNLP-IJCNLP, pages 3982\u20133992, 2019."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/TBDATA.2019.2921572"}],"event":{"name":"WSDM Companion '26: Nineteenth ACM International Conference on Web Search and Data Mining","location":"Boise Centre Boise ID USA","acronym":"WSDM Companion '26","sponsor":["SIGKDD ACM Special Interest Group on Knowledge Discovery in Data","SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web","SIGIR ACM Special Interest Group on Information Retrieval","SIGMOD ACM Special Interest Group on Management of Data"]},"container-title":["Proceedings of the Nineteenth ACM International Conference on Web Search and Data Mining"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3779211.3795738","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,5]],"date-time":"2026-05-05T11:36:03Z","timestamp":1777980963000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3779211.3795738"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,2,22]]},"references-count":24,"alternative-id":["10.1145\/3779211.3795738","10.1145\/3779211"],"URL":"https:\/\/doi.org\/10.1145\/3779211.3795738","relation":{},"subject":[],"published":{"date-parts":[[2026,2,22]]},"assertion":[{"value":"2026-05-05","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}