{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T08:19:24Z","timestamp":1783153164334,"version":"3.54.6"},"publisher-location":"New York, NY, USA","reference-count":60,"publisher":"ACM","funder":[{"name":"Australian Research Council Discovery Project ARC DP","award":["DP230101753"],"award-info":[{"award-number":["DP230101753"]}]},{"name":"School of Electrical Engineering and Computer Science at the University of Queensland","award":["grant NS-2401"],"award-info":[{"award-number":["grant NS-2401"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,4,13]]},"DOI":"10.1145\/3774904.3792273","type":"proceedings-article","created":{"date-parts":[[2026,4,9]],"date-time":"2026-04-09T21:54:34Z","timestamp":1775771674000},"page":"1574-1585","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["PIXEL: Adaptive Steering Via Position-wise Injection with eXact Estimated Levels under a Subspace Calibration"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-9144-7856","authenticated-orcid":false,"given":"Manjiang","family":"Yu","sequence":"first","affiliation":[{"name":"The University of Queensland, Brisbane, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-7656-4333","authenticated-orcid":false,"given":"Hongji","family":"Li","sequence":"additional","affiliation":[{"name":"MBZUAI, Abu Dhabi, United Arab Emirates"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5002-8800","authenticated-orcid":false,"given":"Priyanka","family":"Singh","sequence":"additional","affiliation":[{"name":"The University of Queensland, Brisbane, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4515-6792","authenticated-orcid":false,"given":"Xue","family":"Li","sequence":"additional","affiliation":[{"name":"The University of Queensland, Brisbane, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4908-0243","authenticated-orcid":false,"given":"Di","family":"Wang","sequence":"additional","affiliation":[{"name":"KAUST, Thuwal, Saudi Arabia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1251-8504","authenticated-orcid":false,"given":"Lijie","family":"Hu","sequence":"additional","affiliation":[{"name":"MBZUAI, Abu Dhabi, United Arab Emirates"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,4,12]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Sheer El-Showk Gomez, et al","author":"Bai Yuntao","year":"2022","unstructured":"Yuntao Bai, Saurav Kadavath, Sandipan Kundu, Amanda Askell, Jackson Kernion, Sheer El-Showk Gomez, et al., 2022. Constitutional AI: Harmlessness from AI Feedback. arXiv preprint arXiv:2212.08073 (2022)."},{"key":"e_1_3_2_1_2_1","volume-title":"Steering large language model activations in sparse spaces. arXiv preprint arXiv:2503.00177","author":"Bayat Reza","year":"2025","unstructured":"Reza Bayat, Ali Rahimi-Kalahroudi, Mohammad Pezeshki, Sarath Chandar, and Pascal Vincent. 2025. Steering large language model activations in sparse spaces. arXiv preprint arXiv:2503.00177 (2025)."},{"key":"e_1_3_2_1_3_1","first-page":"1877","article-title":"Language models are few-shot learners","volume":"33","author":"Brown Tom B","year":"2020","unstructured":"Tom B Brown, Benjamin Mann, Nick Ryder, Melanie Subbiah, Jared Kaplan, Prafulla Dhariwal, et al., 2020. Language models are few-shot learners. Advances in Neural Information Processing Systems, Vol. 33 (2020), 1877-1901.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.52202\/079017-1567"},{"key":"e_1_3_2_1_5_1","volume-title":"Lijie Hu, Di Wang, et al.","author":"Cheng Keyuan","year":"2024","unstructured":"Keyuan Cheng, Gang Lin, Haoyang Fei, Lu Yu, Muhammad Asif Ali, Lijie Hu, Di Wang, et al., 2024. Multi-hop question answering under temporal knowledge editing. arXiv preprint arXiv:2404.00492 (2024)."},{"key":"e_1_3_2_1_6_1","volume-title":"Advances in Neural Information Processing Systems","volume":"30","author":"Christiano Paul","year":"2017","unstructured":"Paul Christiano, Jan Leike, Tom Brown, Miljan Martic, Shane Legg, and Dario Amodei. 2017. Deep reinforcement learning from human preferences. Advances in Neural Information Processing Systems, Vol. 30 (2017)."},{"key":"e_1_3_2_1_7_1","volume-title":"Understanding and Mitigating Cross-lingual Privacy Leakage via Language-specific and Universal Privacy Neurons. arXiv preprint arXiv:2506.00759","author":"Dong Wenshuo","year":"2025","unstructured":"Wenshuo Dong, Qingsong Yang, Shu Yang, Lijie Hu, Meng Ding, Wanyu Lin, Tianhang Zheng, and Di Wang. 2025. Understanding and Mitigating Cross-lingual Privacy Leakage via Language-specific and Universal Privacy Neurons. arXiv preprint arXiv:2506.00759 (2025)."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.751"},{"key":"e_1_3_2_1_9_1","unstructured":"Aaron Grattafiori Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Alex Vaughan et al. 2024. The llama 3 herd of models. arXiv e-prints (2024) arXiv-2407."},{"key":"e_1_3_2_1_10_1","volume-title":"Patterns and Mechanisms of Contrastive Activation Engineering in Large Language Models. arXiv preprint arXiv:2505.03189","author":"Hao Yiming","year":"2025","unstructured":"Yiming Hao, Zihan Liu, Mor Geva, Yoav Goldberg, Fei Sha, and Xiang Lisa Li Zhang. 2025. Patterns and Mechanisms of Contrastive Activation Engineering in Large Language Models. arXiv preprint arXiv:2505.03189 (2025)."},{"key":"e_1_3_2_1_11_1","volume-title":"Measuring massive multitask language understanding. arXiv preprint arXiv:2009.03300","author":"Hendrycks Dan","year":"2020","unstructured":"Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, and Jacob Steinhardt. 2020. Measuring massive multitask language understanding. arXiv preprint arXiv:2009.03300 (2020)."},{"key":"e_1_3_2_1_12_1","first-page":"18305","article-title":". Linearity of Large Language Models Explains Their Universal and Efficient Text Editing","volume":"36","author":"Hernandez Evan","year":"2023","unstructured":"Evan Hernandez, Noam Nisan, Sebastian Bordt, et al., 2023. Linearity of Large Language Models Explains Their Universal and Efficient Text Editing. In Advances in Neural Information Processing Systems, Vol. 36. 18305-18337.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-eacl.33"},{"key":"e_1_3_2_1_14_1","volume-title":"Muhammad Asif Ali, and Di Wang","author":"Hu Lijie","year":"2024","unstructured":"Lijie Hu, Liang Liu, Shu Yang, Xin Chen, Hongru Xiao, Mengdi Li, Pan Zhou, Muhammad Asif Ali, and Di Wang. 2024b. A hopfieldian view-based interpretation for chain-of-thought reasoning. arXiv preprint arXiv:2406.12255 (2024)."},{"key":"e_1_3_2_1_15_1","volume-title":"A Unified Understanding and Evaluation of Steering Methods. arXiv preprint arXiv:2502.02716","author":"Im Shawn","year":"2025","unstructured":"Shawn Im and Yixuan Li. 2025. A Unified Understanding and Evaluation of Steering Methods. arXiv preprint arXiv:2502.02716 (2025)."},{"key":"e_1_3_2_1_16_1","unstructured":"Albert Q. Jiang Alexandre Sablayrolles Arthur Mensch Chris Bamford Devendra Singh Chaplot Diego de las Casas Florian Bressand Gianna Lengyel Guillaume Lample Lucile Saulnier L\u00e9lio Renard Lavaud Marie-Anne Lachaux Pierre Stock Teven Le Scao Thibaut Lavril Thomas Wang Timoth\u00e9e Lacroix and William El Sayed. 2023. Mistral 7B. arXiv:2310.06825 [cs.CL] https:\/\/arxiv.org\/abs\/2310.06825"},{"key":"e_1_3_2_1_17_1","volume-title":"MSRS: Adaptive Multi-Subspace Representation Steering for Attribute Alignment in Large Language Models. arXiv preprint arXiv:2508.10599","author":"Jiang Xinyan","year":"2025","unstructured":"Xinyan Jiang, Lin Zhang, Jiayi Zhang, Qingsong Yang, Guimin Hu, Di Wang, and Lijie Hu. 2025. MSRS: Adaptive Multi-Subspace Representation Steering for Attribute Alignment in Large Language Models. arXiv preprint arXiv:2508.10599 (2025)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D17-1082"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3701551.3703507"},{"key":"e_1_3_2_1_20_1","volume-title":"Manjiang Yu, Priyanka Singh, Xue Li, Di Wang, and Lijie Hu.","author":"Li Hongji","year":"2025","unstructured":"Hongji Li, Junchi yao, Manjiang Yu, Priyanka Singh, Xue Li, Di Wang, and Lijie Hu. 2025b. Towards Reasoning-Preserving Unlearning in Multimodal Large Language Models. arXiv:2512.17911 [cs.CL] https:\/\/arxiv.org\/abs\/2512.17911"},{"key":"e_1_3_2_1_21_1","first-page":"41451","volume-title":"Levine (Eds.)","volume":"36","author":"Li Kenneth","year":"2023","unstructured":"Kenneth Li, Oam Patel, Fernanda Vi\u00e9gas, Hanspeter Pfister, and Martin Wattenberg. 2023. Inference-Time Intervention: Eliciting Truthful Answers from a Language Model. In Advances in Neural Information Processing Systems, A. Oh, T. Naumann, A. Globerson, K. Saenko, M. Hardt, and S. Levine (Eds.), Vol. 36. Curran Associates, Inc., 41451-41530. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2023\/file\/81b8390039b7302c909cb769f8b6cd93-Paper-Conference.pdf"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.229"},{"key":"e_1_3_2_1_23_1","volume-title":"Token-wise Adaptive Modulation for Controlled Text Generation. arXiv preprint arXiv:2405.06543","author":"Liu Yu","year":"2024","unstructured":"Yu Liu, Jingyi Wang, Yilun Xu, and Rui Zhang. 2024b. Token-wise Adaptive Modulation for Controlled Text Generation. arXiv preprint arXiv:2405.06543 (2024)."},{"key":"e_1_3_2_1_24_1","volume-title":"Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing.","author":"Liu Zihan","year":"2024","unstructured":"Zihan Liu, Yanda Chen, Tianyu Zhang, et al., 2024a. Steerability of Large Language Models: A Causal Perspective. In Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.52202\/068431-1262"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1260"},{"key":"e_1_3_2_1_27_1","volume-title":"FairSteer: Inference-Time Debiasing of Large Language Models with Adaptive Steering. In Findings of the Association for Computational Linguistics: ACL","author":"Nguyen An","year":"2025","unstructured":"An Nguyen, David Lee, Rohan Kumar, et al., 2025a. FairSteer: Inference-Time Debiasing of Large Language Models with Adaptive Steering. In Findings of the Association for Computational Linguistics: ACL 2025."},{"key":"e_1_3_2_1_28_1","volume-title":"Multi-attribute steering of language models via targeted intervention. arXiv preprint arXiv:2502.12446","author":"Nguyen Duy","year":"2025","unstructured":"Duy Nguyen, Archiki Prasad, Elias Stengel-Eskin, and Mohit Bansal. 2025b. Multi-attribute steering of language models via targeted intervention. arXiv preprint arXiv:2502.12446 (2025)."},{"key":"e_1_3_2_1_29_1","volume-title":"Representation Engineering: Editing Model Behavior via Activation Space Geometry. arXiv preprint arXiv:2402.01862","author":"Olson Kyle","year":"2024","unstructured":"Kyle Olson, Sheng Shen, Catherine Olsson, and Deep Ganguli. 2024. Representation Engineering: Editing Model Behavior via Activation Space Geometry. arXiv preprint arXiv:2402.01862 (2024)."},{"key":"e_1_3_2_1_30_1","volume-title":"Beyond Linear Steering: Unified Multi-Attribute Control for Language Models. arXiv preprint arXiv:2505.24535","author":"Oozeer Narmeen","year":"2025","unstructured":"Narmeen Oozeer, Luke Marks, Fazl Barez, and Amirali Abdullah. 2025. Beyond Linear Steering: Unified Multi-Attribute Control for Language Models. arXiv preprint arXiv:2505.24535 (2025)."},{"key":"e_1_3_2_1_31_1","unstructured":"OpenAI. 2023. GPT-4 Technical Report. https:\/\/openai.com\/research\/gpt-4. Accessed: 2025-09-19."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.52202\/068431-2011"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.findings-acl.165"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.828"},{"key":"e_1_3_2_1_35_1","volume-title":"Proceedings of the 2022 International Conference on Learning Representations.","author":"Ross Alexis","year":"2022","unstructured":"Alexis Ross, Adam Santoro, Wojciech M. Czarnecki, et al., 2022. Attention is Not All You Need: Pure Attention Loses Rank Doubly Exponentially with Depth. In Proceedings of the 2022 International Conference on Learning Representations."},{"key":"e_1_3_2_1_36_1","volume-title":"Smith","author":"Sheng Emily","year":"2025","unstructured":"Emily Sheng, Yizhong Xu, Rowan Zellers, Yejin Choi, and Noah A. Smith. 2025. AlphaSteer: Learning Refusal Steering with Principled Null-space Constraints. arXiv preprint arXiv:2506.07022 (2025)."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-emnlp.827"},{"key":"e_1_3_2_1_38_1","volume-title":"Jonibek Mansurov, Di Wang, and Preslav Nakov.","author":"Su Jinyan","year":"2023","unstructured":"Jinyan Su, Terry Yue Zhuo, Jonibek Mansurov, Di Wang, and Preslav Nakov. 2023a. Fake news detectors are biased against texts generated by large language models. arXiv preprint arXiv:2309.08674 (2023)."},{"key":"e_1_3_2_1_39_1","volume-title":"Qwen2 technical report. arXiv preprint arXiv:2412.15115","author":"Team Qwen","year":"2024","unstructured":"Qwen Team. 2024. Qwen2 technical report. arXiv preprint arXiv:2412.15115 (2024)."},{"key":"e_1_3_2_1_40_1","unstructured":"Hugo Touvron Thibaut Lavril Gautier Izacard Xavier Martinet Marie-Anne Lachaux Timothee Lacroix et al. 2023. LLaMA: Open and Efficient Foundation Language Models. arXiv preprint arXiv:2302.13971 (2023)."},{"key":"e_1_3_2_1_41_1","volume-title":"GLUE: A multi-task benchmark and analysis platform for natural language understanding. arXiv preprint arXiv:1804.07461","author":"Wang Alex","year":"2018","unstructured":"Alex Wang, Amanpreet Singh, Julian Michael, Felix Hill, Omer Levy, and Samuel R Bowman. 2018. GLUE: A multi-task benchmark and analysis platform for natural language understanding. arXiv preprint arXiv:1804.07461 (2018)."},{"key":"e_1_3_2_1_42_1","volume-title":"When truth is overridden: Uncovering the internal origins of sycophancy in large language models. arXiv preprint arXiv:2508.02087","author":"Wang Keyu","year":"2025","unstructured":"Keyu Wang, Jin Li, Shu Yang, Zhuoran Zhang, and Di Wang. 2025c. When truth is overridden: Uncovering the internal origins of sycophancy in large language models. arXiv preprint arXiv:2508.02087 (2025)."},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/3696410.3714640"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/3696410.3714640"},{"key":"e_1_3_2_1_45_1","volume-title":"Daniel Egert, Olivier Delalleau, Jane Polak Scowcroft, Neel Kant, Aidan Swope, et al.","author":"Wang Zhilin","year":"2023","unstructured":"Zhilin Wang, Yi Dong, Jiaqi Zeng, Virginia Adams, Makesh Narsimhan Sreedhar, Daniel Egert, Olivier Delalleau, Jane Polak Scowcroft, Neel Kant, Aidan Swope, et al., 2023. Helpsteer: Multi-attribute helpfulness dataset for steerlm. arXiv preprint arXiv:2311.09528 (2023)."},{"key":"e_1_3_2_1_46_1","unstructured":"Jason Wei Maarten Bosma Vincent Zhao Kelvin Guu Adams Yu Brian Lester et al. 2022. Finetuned Language Models Are Zero-Shot Learners. arXiv preprint arXiv:2201.08239 (2022)."},{"key":"e_1_3_2_1_47_1","first-page":"63908","article-title":"Reft: Representation finetuning for language models","volume":"37","author":"Wu Zhengxuan","year":"2024","unstructured":"Zhengxuan Wu, Aryaman Arora, Zheng Wang, Atticus Geiger, Dan Jurafsky, Christopher D Manning, and Christopher Potts. 2024. Reft: Representation finetuning for language models. Advances in Neural Information Processing Systems, Vol. 37 (2024), 63908-63962.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_48_1","volume-title":"SORRY-Bench: Systematically Evaluating Large Language Model Safety Refusal. In The Thirteenth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=YfKNaRktan","author":"Xie Tinghao","year":"2025","unstructured":"Tinghao Xie, Xiangyu Qi, Yi Zeng, Yangsibo Huang, Udari Madhushani Sehwag, Kaixuan Huang, Luxi He, Boyi Wei, Dacheng Li, Ying Sheng, Ruoxi Jia, Bo Li, Kai Li, Danqi Chen, Peter Henderson, and Prateek Mittal. 2025. SORRY-Bench: Systematically Evaluating Large Language Model Safety Refusal. In The Thirteenth International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=YfKNaRktan"},{"key":"e_1_3_2_1_49_1","volume-title":"AutoSteer: Scalable Discovery of Steering Vectors. arXiv preprint arXiv:2404.00667","author":"Yang Lingfan","year":"2024","unstructured":"Lingfan Yang, Yilun Zheng, Yizhong Wu, Jingbo Zhu, and Hannaneh Hajishirzi. 2024c. AutoSteer: Scalable Discovery of Steering Vectors. arXiv preprint arXiv:2404.00667 (2024)."},{"key":"e_1_3_2_1_50_1","volume-title":"First Conference on Language Modeling.","author":"Yang Shu","year":"2024","unstructured":"Shu Yang, Muhammad Asif Ali, Lu Yu, Lijie Hu, and Di Wang. 2024a. Model autophagy analysis to explicate self-consumption within human-ai interactions. In First Conference on Language Modeling."},{"key":"e_1_3_2_1_51_1","volume-title":"Lijie Hu, and Di Wang.","author":"Yang Shu","year":"2024","unstructured":"Shu Yang, Jiayuan Su, Han Jiang, Mengdi Li, Keyuan Cheng, Muhammad Asif Ali, Lijie Hu, and Di Wang. 2024b. Dialectical alignment: Resolving the tension of 3h and security threats of llms. arXiv preprint arXiv:2404.00486 (2024)."},{"key":"e_1_3_2_1_52_1","volume-title":"Understanding aha moments: from external observations to internal mechanisms. arXiv preprint arXiv:2504.02956","author":"Yang Shu","year":"2025","unstructured":"Shu Yang, Junchao Wu, Xin Chen, Yunze Xiao, Xinyi Yang, Derek F Wong, and Di Wang. 2025a. Understanding aha moments: from external observations to internal mechanisms. arXiv preprint arXiv:2504.02956 (2025)."},{"key":"e_1_3_2_1_53_1","volume-title":"Exploring the Personality Traits of LLMs through Latent Features Steering. arXiv preprint arXiv:2410.10863","author":"Yang Shu","year":"2024","unstructured":"Shu Yang, Shenzhe Zhu, Liang Liu, Lijie Hu, Mengdi Li, and Di Wang. 2024d. Exploring the Personality Traits of LLMs through Latent Features Steering. arXiv preprint arXiv:2410.10863 (2024)."},{"key":"e_1_3_2_1_54_1","volume-title":"Fraud-r1: A multi-round benchmark for assessing the robustness of llm against augmented fraud and phishing inducements. arXiv preprint arXiv:2502.12904","author":"Yang Shu","year":"2025","unstructured":"Shu Yang, Shenzhe Zhu, Zeyu Wu, Keyu Wang, Junchi Yao, Junchao Wu, Lijie Hu, Mengdi Li, Derek F Wong, and Di Wang. 2025c. Fraud-r1: A multi-round benchmark for assessing the robustness of llm against augmented fraud and phishing inducements. arXiv preprint arXiv:2502.12904 (2025)."},{"key":"e_1_3_2_1_55_1","volume-title":"D-LEAF: Localizing and Correcting Hallucinations in Multimodal LLMs via Layer-to-head Attention Diagnostics. arXiv preprint arXiv:2509.07864","author":"Yang Tiancheng","year":"2025","unstructured":"Tiancheng Yang, Lin Zhang, Jiaye Lin, Guimin Hu, Di Wang, and Lijie Hu. 2025b. D-LEAF: Localizing and Correcting Hallucinations in Multimodal LLMs via Layer-to-head Attention Diagnostics. arXiv preprint arXiv:2509.07864 (2025)."},{"key":"e_1_3_2_1_56_1","volume-title":"Understanding the repeat curse in large language models from a feature perspective. arXiv preprint arXiv:2504.14218","author":"Yao Junchi","year":"2025","unstructured":"Junchi Yao, Shu Yang, Jianhua Xu, Lijie Hu, Mengdi Li, and Di Wang. 2025. Understanding the repeat curse in large language models from a feature perspective. arXiv preprint arXiv:2504.14218 (2025)."},{"key":"e_1_3_2_1_57_1","volume-title":"Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing.","author":"Zhang Chen","year":"2025","unstructured":"Chen Zhang, Meng Li, Rui Zhao, et al., 2025a. SADI: Semantics-Adaptive Dynamic Interventions for Safer Language Models. In Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing."},{"key":"e_1_3_2_1_58_1","volume-title":"Understanding and Mitigating Political Stance Cross-topic Generalization in Large Language Models. arXiv preprint arXiv:2508.02360","author":"Zhang Jiayi","year":"2025","unstructured":"Jiayi Zhang, Shu Yang, Junchao Wu, Derek F Wong, and Di Wang. 2025b. Understanding and Mitigating Political Stance Cross-topic Generalization in Large Language Models. arXiv preprint arXiv:2508.02360 (2025)."},{"key":"e_1_3_2_1_59_1","volume-title":"Locate-then-edit for multi-hop factual recall under knowledge editing. arXiv preprint arXiv:2410.06331","author":"Zhang Zhuoran","year":"2024","unstructured":"Zhuoran Zhang, Yongxiang Li, Zijian Kan, Keyuan Cheng, Lijie Hu, and Di Wang. 2024. Locate-then-edit for multi-hop factual recall under knowledge editing. arXiv preprint arXiv:2410.06331 (2024)."},{"key":"e_1_3_2_1_60_1","volume-title":"Flattery in motion: Benchmarking and analyzing sycophancy in video-llms. arXiv preprint arXiv:2506.07180","author":"Zhou Wenrui","year":"2025","unstructured":"Wenrui Zhou, Mohamed Hendy, Shu Yang, Qingsong Yang, Zikun Guo, Yuyu Luo, Lijie Hu, and Di Wang. 2025. Flattery in motion: Benchmarking and analyzing sycophancy in video-llms. arXiv preprint arXiv:2506.07180 (2025)."}],"event":{"name":"WWW '26: The ACM Web Conference 2026","location":"Dubai United Arab Emirates","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web"]},"container-title":["Proceedings of the ACM Web Conference 2026"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3774904.3792273","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T07:58:08Z","timestamp":1783151888000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3774904.3792273"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4,12]]},"references-count":60,"alternative-id":["10.1145\/3774904.3792273","10.1145\/3774904"],"URL":"https:\/\/doi.org\/10.1145\/3774904.3792273","relation":{},"subject":[],"published":{"date-parts":[[2026,4,12]]},"assertion":[{"value":"2026-04-12","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}