{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T03:08:43Z","timestamp":1784603323299,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":75,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,8,3]]},"DOI":"10.1145\/3711896.3737120","type":"proceedings-article","created":{"date-parts":[[2025,8,3]],"date-time":"2025-08-03T21:07:39Z","timestamp":1754255259000},"page":"3250-3260","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Self-Regularization with Sparse Autoencoders for Controllable LLM-based Classification"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7816-7658","authenticated-orcid":false,"given":"Xuansheng","family":"Wu","sequence":"first","affiliation":[{"name":"University of Georgia, Athens, Georgia, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4075-5980","authenticated-orcid":false,"given":"Wenhao","family":"Yu","sequence":"additional","affiliation":[{"name":"Tencent AI Lab, Seattle, WA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4519-1931","authenticated-orcid":false,"given":"Xiaoming","family":"Zhai","sequence":"additional","affiliation":[{"name":"University of Georgia, Athens, GA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9170-2424","authenticated-orcid":false,"given":"Ninghao","family":"Liu","sequence":"additional","affiliation":[{"name":"University of Georgia, Athens, GA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,8,3]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al.","author":"Achiam Josh","year":"2023","unstructured":"Josh Achiam, Steven Adler, Sandhini Agarwal, Lama Ahmad, Ilge Akkaya, Florencia Leoni Aleman, Diogo Almeida, Janko Altenschmidt, Sam Altman, Shyamal Anadkat, et al., 2023. Gpt-4 technical report. arXiv(2023)."},{"key":"e_1_3_2_2_2_1","unstructured":"Mistral AI. 2024. Mistral AI API (0.0.2). https:\/\/docs.mistral.ai\/api\/"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00034"},{"key":"e_1_3_2_2_4_1","unstructured":"Sina Baharlouei Maher Nouiehed Ahmad Beirami and Meisam Razaviyayn. 2019. Renyi fair inference. arXiv preprint arXiv:1906.12005(2019)."},{"key":"e_1_3_2_2_5_1","unstructured":"Yuntao Bai Andy Jones Kamal Ndousse Amanda Askell Anna Chen Nova DasSarma Dawn Drain Stanislav Fort Deep Ganguli Tom Henighan et al. 2022. Training a helpful and harmless assistant with reinforcement learning from human feedback. arXiv preprint arXiv:2204.05862(2022)."},{"key":"e_1_3_2_2_6_1","unstructured":"Yonatan Belinkov Llu\u00eds M\u00e0rquez Hassan Sajjad Nadir Durrani Fahim Dalvi and James Glass. 2018. Evaluating layers of representation in neural machine translation on part-of-speech and semantic tagging tasks. arXiv(2018)."},{"key":"e_1_3_2_2_7_1","unstructured":"Steven Bills Nick Cammarata Dan Mossing Henk Tillman Leo Gao Gabriel Goh Ilya Sutskever Jan Leike Jeff Wu and William Saunders. 2023. Language models can explain neurons in language models. URL https:\/\/openaipublic. blob. core. windows. net\/neuron-explainer\/paper\/index. html.(2023)."},{"key":"e_1_3_2_2_8_1","unstructured":"Trenton Bricken Jonathan Marcus Siddharth Mishra-Sharma Meg Tong Ethan Perez Mrinank Sharma Kelley Rivoire Thomas Henighan and Adam Jermyn. 2024. Using Dictionary Learning Features as Classifiers. https:\/\/transformer-circuits.pub\/2024\/features-as-classifiers\/index.html."},{"key":"e_1_3_2_2_9_1","volume-title":"Towards Monosemanticity: Decomposing Language Models With Dictionary Learning. Transformer Circuits Thread(2023). https:\/\/transformer-circuits.pub\/2023\/monosemantic-features\/index.html.","author":"Bricken Trenton","year":"2023","unstructured":"Trenton Bricken, Adly Templeton, Joshua Batson, Brian Chen, Adam Jermyn, Tom Conerly, Nick Turner, Cem Anil, Carson Denison, Amanda Askell, Robert Lasenby, Yifan Wu, Shauna Kravec, Nicholas Schiefer, Tim Maxwell, Nicholas Joseph, Zac Hatfield-Dodds, Alex Tamkin, Karina Nguyen, Brayden McLean, Josiah E Burke, Tristan Hume, Shan Carter, Tom Henighan, and Christopher Olah. 2023. Towards Monosemanticity: Decomposing Language Models With Dictionary Learning. Transformer Circuits Thread(2023). https:\/\/transformer-circuits.pub\/2023\/monosemantic-features\/index.html."},{"key":"e_1_3_2_2_10_1","unstructured":"Maheep Chaudhary and Atticus Geiger. 2024. Evaluating Open-Source Sparse Autoencoders on Disentangling Factual Knowledge in GPT-2 Small. arXiv(2024)."},{"key":"e_1_3_2_2_11_1","unstructured":"Boli Chen Yao Fu Guangwei Xu Pengjun Xie Chuanqi Tan Mosha Chen and Liping Jing. [n.d.]. Probing BERT in Hyperbolic Spaces. In ICLR."},{"key":"e_1_3_2_2_12_1","unstructured":"Junying Chen Chi Gui Anningzhe Gao Ke Ji Xidong Wang Xiang Wan and Benyou Wang. 2024. CoD Towards an Interpretable Medical Agent using Chain of Diagnosis. arxiv:2407.13301 [cs.CL] https:\/\/arxiv.org\/abs\/2407.13301"},{"key":"e_1_3_2_2_13_1","unstructured":"Hoagy Cunningham Aidan Ewart Logan Riggs Robert Huben and Lee Sharkey. 2023. Sparse autoencoders find highly interpretable features in language models. arXiv preprint arXiv:2309.08600(2023)."},{"key":"e_1_3_2_2_14_1","first-page":"16124","article-title":"Analyzing Transformers in Embedding Space","author":"Dar Guy","year":"2023","unstructured":"Guy Dar, Mor Geva, Ankit Gupta, and Jonathan Berant. 2023. Analyzing Transformers in Embedding Space. In ACL. 16124-16170.","journal-title":"ACL."},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"crossref","unstructured":"Ning Ding Yulin Chen Bokai Xu Yujia Qin Zhi Zheng Shengding Hu Zhiyuan Liu Maosong Sun and Bowen Zhou. 2023. Enhancing Chat Language Models by Scaling High-quality Instructional Conversations. arXiv preprint arXiv:2305.14233(2023).","DOI":"10.18653\/v1\/2023.emnlp-main.183"},{"key":"e_1_3_2_2_16_1","unstructured":"Abhimanyu Dubey Abhinav Jauhri Abhinav Pandey Abhishek Kadian Ahmad Al-Dahle Aiesha Letman Akhil Mathur Alan Schelten Amy Yang Angela Fan et al. 2024. The llama 3 herd of models. arXiv preprint arXiv:2407.21783(2024)."},{"key":"e_1_3_2_2_17_1","volume-title":"ICML 2024 Workshop on Mechanistic Interpretability.","author":"Dumas Cl\u00e9ment","unstructured":"Cl\u00e9ment Dumas, Veniamin Veselovsky, Giovanni Monea, Robert West, and Chris Wendler. [n.d.]. How do Llamas process multilingual text? A latent exploration through activation patching. In ICML 2024 Workshop on Mechanistic Interpretability."},{"key":"e_1_3_2_2_18_1","unstructured":"Nelson Elhage Tristan Hume Catherine Olsson Nicholas Schiefer Tom Henighan Shauna Kravec Zac Hatfield-Dodds Robert Lasenby Dawn Drain Carol Chen et al. 2022. Toy models of superposition. arXiv(2022)."},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-0-387-87857-7"},{"key":"e_1_3_2_2_20_1","unstructured":"Deep Ganguli Liane Lovitt Jackson Kernion Amanda Askell Yuntao Bai Saurav Kadavath Ben Mann Ethan Perez Nicholas Schiefer Kamal Ndousse et al. 2022. Red teaming language models to reduce harms: Methods scaling behaviors and lessons learned. arXiv preprint arXiv:2209.07858(2022)."},{"key":"e_1_3_2_2_21_1","volume-title":"Henk Tillman, Gabriel Goh, Rajan Troll, Alec Radford, Ilya Sutskever, Jan Leike, and Jeffrey Wu.","author":"Gao Leo","year":"2024","unstructured":"Leo Gao, Tom Dupr\u00e9 la Tour, Henk Tillman, Gabriel Goh, Rajan Troll, Alec Radford, Ilya Sutskever, Jan Leike, and Jeffrey Wu. 2024. Scaling and evaluating sparse autoencoders. arXiv preprint arXiv:2406.04093(2024)."},{"key":"e_1_3_2_2_22_1","first-page":"1026","article-title":"Delving deep into rectifiers: Surpassing human-level performance on imagenet classification","author":"He Kaiming","year":"2015","unstructured":"Kaiming He, Xiangyu Zhang, Shaoqing Ren, and Jian Sun. 2015. Delving deep into rectifiers: Surpassing human-level performance on imagenet classification. In Proceedings of the IEEE ICCV. 1026-1034.","journal-title":"Proceedings of the IEEE ICCV."},{"key":"e_1_3_2_2_23_1","unstructured":"John Hewitt and Christopher D Manning. 2019. A structural probe for finding syntax in word representations. In NAACL."},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.1080\/00401706.1970.10488634"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/3631326"},{"key":"e_1_3_2_2_26_1","first-page":"2879","article-title":"Stable and fair classification","author":"Huang Lingxiao","year":"2019","unstructured":"Lingxiao Huang and Nisheeth Vishnoi. 2019. Stable and fair classification. In ICML. PMLR, 2879-2890.","journal-title":"ICML. PMLR"},{"key":"e_1_3_2_2_27_1","volume-title":"Aidan Ewart, and Lee Sharkey.","author":"Huben Robert","year":"2023","unstructured":"Robert Huben, Hoagy Cunningham, Logan Riggs Smith, Aidan Ewart, and Lee Sharkey. 2023. Sparse Autoencoders Find Highly Interpretable Features in Language Models. In The Twelfth ICLR."},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"crossref","unstructured":"Ganesh Jawahar Beno^it Sagot and Djam\u00e9 Seddah. 2019. What does BERT learn about the structure of language?. In ACL.","DOI":"10.18653\/v1\/P19-1356"},{"key":"e_1_3_2_2_29_1","unstructured":"Albert Q. Jiang Alexandre Sablayrolles Arthur Mensch Chris Bamford Devendra Singh Chaplot Diego de las Casas Florian Bressand Gianna Lengyel Guillaume Lample Lucile Saulnier L\u00e9lio Renard Lavaud Marie-Anne Lachaux Pierre Stock Teven Le Scao Thibaut Lavril Thomas Wang Timoth\u00e9e Lacroix and William El Sayed. 2023. Mistral 7B. arxiv:2310.06825 [cs.CL] https:\/\/arxiv.org\/abs\/2310.06825"},{"key":"e_1_3_2_2_30_1","unstructured":"Mingyu Jin Kai Mei Wujiang Xu Mingjie Sun Ruixiang Tang Mengnan Du Zirui Liu and Yongfeng Zhang. 2025. Massive Values in Self-Attention Modules are the Key to Contextual Knowledge Understanding. arXiv(2025)."},{"key":"e_1_3_2_2_31_1","unstructured":"Mingyu Jin Qinkai Yu Jingyuan Huang Qingcheng Zeng Zhenting Wang Wenyue Hua Haiyan Zhao Kai Mei Yanda Meng Kaize Ding et al. 2024. Exploring Concept Depth: How Large Language Models Acquire Knowledge and Concept at Different Layers? arXiv preprint arXiv:2404.07066(2024)."},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDMW.2011.83"},{"key":"e_1_3_2_2_33_1","first-page":"5830","article-title":"Measuring Fairness of Text Classifiers via Prediction Sensitivity","author":"Krishna Satyapriya","year":"2022","unstructured":"Satyapriya Krishna, Rahul Gupta, Apurv Verma, Jwala Dhamala, Yada Pruksachatkun, and Kai-Wei Chang. 2022. Measuring Fairness of Text Classifiers via Prediction Sensitivity. In ACL. 5830-5842.","journal-title":"ACL."},{"key":"e_1_3_2_2_34_1","volume-title":"Khyathi Chandu, Nouha Dziri, Sachin Kumar, Tom Zick, Yejin Choi, et al.","author":"Lambert Nathan","year":"2024","unstructured":"Nathan Lambert, Valentina Pyatkin, Jacob Morrison, LJ Miranda, Bill Yuchen Lin, Khyathi Chandu, Nouha Dziri, Sachin Kumar, Tom Zick, Yejin Choi, et al., 2024. Rewardbench: Evaluating reward models for language modeling. arXiv(2024)."},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"crossref","unstructured":"Tom Lieberum Senthooran Rajamanoharan Arthur Conmy Lewis Smith Nicolas Sonnerat Vikrant Varma J\u00e1nos Kram\u00e1r Anca Dragan Rohin Shah and Neel Nanda. 2024. Gemma scope: Open sparse autoencoders everywhere all at once on gemma 2. arXiv preprint arXiv:2408.05147(2024).","DOI":"10.18653\/v1\/2024.blackboxnlp-1.19"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"crossref","unstructured":"Zi Lin Zihan Wang Yongqi Tong Yangkun Wang Yuxin Guo Yujia Wang and Jingbo Shang. 2023. ToxicChat: Unveiling Hidden Challenges of Toxicity Detection in Real-World User-AI Conversation. arxiv:2310.17389 [cs.CL]","DOI":"10.18653\/v1\/2023.findings-emnlp.311"},{"key":"e_1_3_2_2_37_1","unstructured":"Aixin Liu Bei Feng Bing Xue Bingxuan Wang Bochao Wu Chengda Lu Chenggang Zhao Chengqi Deng Chenyu Zhang Chong Ruan et al. 2024. Deepseek-v3 technical report. arXiv preprint arXiv:2412.19437(2024)."},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"crossref","unstructured":"Xiao Liu Hanyu Lai Hao Yu Yifan Xu Aohan Zeng Zhengxiao Du Peng Zhang Yuxiao Dong and Jie Tang. 2023. WebGLM: Towards an efficient web-enhanced question answering system with human preferences. In KDD.","DOI":"10.1145\/3580305.3599931"},{"key":"e_1_3_2_2_39_1","unstructured":"AI @ Meta Llama Team. 2024. The Llama 3 Herd of Models. arxiv:2407.21783 [cs.AI] https:\/\/arxiv.org\/abs\/2407.21783"},{"key":"e_1_3_2_2_40_1","unstructured":"Ilya Loshchilov Frank Hutter et al. 2017. Fixing weight decay regularization in adam. arXiv preprint arXiv:1711.05101 Vol. 5 (2017)."},{"key":"e_1_3_2_2_41_1","unstructured":"Alireza Makhzani and Brendan Frey. 2013. K-sparse autoencoders. arXiv(2013)."},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i12.26752"},{"key":"e_1_3_2_2_43_1","unstructured":"Samuel Marks Can Rager Eric J Michaud Yonatan Belinkov David Bau and Aaron Mueller. 2024. Sparse feature circuits: Discovering and editing interpretable causal graphs in language models. arXiv preprint arXiv:2403.19647(2024)."},{"key":"e_1_3_2_2_44_1","unstructured":"Beren Millidge and Sid Black. 2022. The singular value decompositions of transformer weight matrices are highly interpretable. https:\/\/www.alignmentforum.org\/posts\/mkbGjzxD8d8XqKHzA\/the-singular-value-decompositions-of-transformer-weight"},{"key":"e_1_3_2_2_45_1","volume-title":"NIPS","volume":"32","author":"M\u00fcller Rafael","year":"2019","unstructured":"Rafael M\u00fcller, Simon Kornblith, and Geoffrey E Hinton. 2019. When does label smoothing help? NIPS, Vol. 32 (2019)."},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.23915\/distill.00024.001"},{"key":"e_1_3_2_2_47_1","volume-title":"Sparse coding with an overcomplete basis set: A strategy employed by V1? Vision research","author":"Olshausen Bruno A","year":"1997","unstructured":"Bruno A Olshausen and David J Field. 1997. Sparse coding with an overcomplete basis set: A strategy employed by V1? Vision research, Vol. 37, 23 (1997), 3311-3325."},{"key":"e_1_3_2_2_48_1","first-page":"27730","volume-title":"NIPS","volume":"35","author":"Ouyang Long","year":"2022","unstructured":"Long Ouyang, Jeffrey Wu, Xu Jiang, Diogo Almeida, Carroll Wainwright, Pamela Mishkin, Chong Zhang, Sandhini Agarwal, Katarina Slama, Alex Ray, et al., 2022. Training language models to follow instructions with human feedback. NIPS, Vol. 35 (2022), 27730-27744."},{"key":"e_1_3_2_2_49_1","volume-title":"NIPS","volume":"30","author":"Quadrianto Novi","year":"2017","unstructured":"Novi Quadrianto and Viktoriia Sharmanska. 2017. Recycling privileged learning and distribution matching for fairness. NIPS, Vol. 30 (2017)."},{"key":"e_1_3_2_2_50_1","unstructured":"Senthooran Rajamanoharan Arthur Conmy Lewis Smith Tom Lieberum Vikrant Varma J\u00e1nos Kram\u00e1r Rohin Shah and Neel Nanda. 2024. Improving dictionary learning with gated sparse autoencoders. arXiv preprint arXiv:2404.16014(2024)."},{"key":"e_1_3_2_2_51_1","doi-asserted-by":"crossref","unstructured":"N Reimers. 2019. Sentence-BERT: Sentence Embeddings using Siamese BERT-Networks. arXiv preprint arXiv:1908.10084(2019).","DOI":"10.18653\/v1\/D19-1410"},{"key":"e_1_3_2_2_52_1","first-page":"2980","article-title":"Focal loss for dense object detection","author":"Ross T-YLPG","year":"2017","unstructured":"T-YLPG Ross and GKHP Doll\u00e1r. 2017. Focal loss for dense object detection. In CVPR. 2980-2988.","journal-title":"CVPR."},{"key":"e_1_3_2_2_53_1","unstructured":"RyokoAI. 2023. ShareGPT Dataset. https:\/\/huggingface.co\/datasets\/RyokoAI\/ShareGPT52K. (2023)."},{"key":"e_1_3_2_2_54_1","unstructured":"Adam Scherlis Kshitij Sachan Adam S Jermyn Joe Benton and Buck Shlegeris. 2022. Polysemanticity and capacity in neural networks. arXiv(2022)."},{"key":"e_1_3_2_2_55_1","doi-asserted-by":"crossref","unstructured":"Dong Shu Xuansheng Wu Haiyan Zhao Mengnan Du and Ninghao Liu. 2025 a. Beyond Input Activations: Identifying Influential Latents by Gradient Sparse Autoencoders. arXiv preprint arXiv:2505.08080(2025).","DOI":"10.18653\/v1\/2025.emnlp-main.87"},{"key":"e_1_3_2_2_56_1","doi-asserted-by":"crossref","unstructured":"Dong Shu Xuansheng Wu Haiyan Zhao Daking Rai Ziyu Yao Ninghao Liu and Mengnan Du. 2025 b. A survey on sparse autoencoders: Interpreting the internal mechanisms of large language models. arXiv preprint arXiv:2503.05613(2025).","DOI":"10.18653\/v1\/2025.findings-emnlp.89"},{"key":"e_1_3_2_2_57_1","volume-title":"Dropout: a simple way to prevent neural networks from overfitting. The journal of machine learning research","author":"Srivastava Nitish","year":"2014","unstructured":"Nitish Srivastava, Geoffrey Hinton, Alex Krizhevsky, Ilya Sutskever, and Ruslan Salakhutdinov. 2014. Dropout: a simple way to prevent neural networks from overfitting. The journal of machine learning research, Vol. 15, 1 (2014), 1929-1958."},{"key":"e_1_3_2_2_58_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1159"},{"key":"e_1_3_2_2_59_1","volume-title":"Scaling Monosemanticity: Extracting Interpretable Features from Claude 3 Sonnet. Transformer Circuits Thread(2024). https:\/\/transformer-circuits.pub\/2024\/scaling-monosemanticity\/index.html","author":"Templeton Adly","year":"2024","unstructured":"Adly Templeton, Tom Conerly, Jonathan Marcus, Jack Lindsey, Trenton Bricken, Brian Chen, Adam Pearce, Craig Citro, Emmanuel Ameisen, Andy Jones, Hoagy Cunningham, Nicholas L Turner, Callum McDougall, Monte MacDiarmid, C. Daniel Freeman, Theodore R. Sumers, Edward Rees, Joshua Batson, Adam Jermyn, Shan Carter, Chris Olah, and Tom Henighan. 2024. Scaling Monosemanticity: Extracting Interpretable Features from Claude 3 Sonnet. Transformer Circuits Thread(2024). https:\/\/transformer-circuits.pub\/2024\/scaling-monosemanticity\/index.html"},{"key":"e_1_3_2_2_60_1","volume-title":"Md Arafatur Rahman, Ali H Alenezi, and Mueen Uddin.","author":"Tusher Ekramul Haque","year":"2024","unstructured":"Ekramul Haque Tusher, Mohd Arfian Ismail, Md Arafatur Rahman, Ali H Alenezi, and Mueen Uddin. 2024. Email spam: A comprehensive review of optimize detection methods, challenges, and open research problems. IEEE Access(2024)."},{"key":"e_1_3_2_2_61_1","volume-title":"Glue: A multi-task benchmark and analysis platform for natural language understanding. arXiv preprint arXiv:1804.07461(2018).","author":"Wang Alex","year":"2018","unstructured":"Alex Wang. 2018. Glue: A multi-task benchmark and analysis platform for natural language understanding. arXiv preprint arXiv:1804.07461(2018)."},{"key":"e_1_3_2_2_62_1","doi-asserted-by":"crossref","unstructured":"Lean Wang Lei Li Damai Dai Deli Chen Hao Zhou Fandong Meng Jie Zhou and Xu Sun. 2023a. Label Words are Anchors: An Information Flow Perspective for Understanding In-Context Learning. arXiv arXiv:2305.14160(2023).","DOI":"10.18653\/v1\/2023.emnlp-main.609"},{"key":"e_1_3_2_2_63_1","unstructured":"Liang Wang Nan Yang Xiaolong Huang Linjun Yang Rangan Majumder and Furu Wei. 2023b. Improving text embeddings with large language models. arXiv preprint arXiv:2401.00368(2023)."},{"key":"e_1_3_2_2_64_1","volume-title":"Makesh Narsimhan Sreedhar, and Oleksii Kuchaiev","author":"Wang Zhilin","year":"2024","unstructured":"Zhilin Wang, Yi Dong, Olivier Delalleau, Jiaqi Zeng, Gerald Shen, Daniel Egert, Jimmy J. Zhang, Makesh Narsimhan Sreedhar, and Oleksii Kuchaiev. 2024. HelpSteer2: Open-source dataset for training top-performing reward models. arxiv:2406.08673"},{"key":"e_1_3_2_2_65_1","volume-title":"Denny Zhou, et al.","author":"Wei Jason","year":"2022","unstructured":"Jason Wei, Xuezhi Wang, Dale Schuurmans, Maarten Bosma, Fei Xia, Ed Chi, Quoc V Le, Denny Zhou, et al., 2022. Chain-of-thought prompting elicits reasoning in large language models. NIPS(2022)."},{"key":"e_1_3_2_2_66_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.naacl-long.130"},{"key":"e_1_3_2_2_67_1","unstructured":"Xuansheng Wu Jiayi Yuan Wenlin Yao Xiaoming Zhai and Ninghao Liu. 2025. Interpreting and steering llms with mutual information-based explanations on sparse autoencoders. arXiv preprint arXiv:2502.15576(2025)."},{"key":"e_1_3_2_2_68_1","volume-title":"Wizardlm: Empowering large language models to follow complex instructions. arXiv preprint arXiv:2304.12244(2023).","author":"Xu Can","year":"2023","unstructured":"Can Xu, Qingfeng Sun, Kai Zheng, Xiubo Geng, Pu Zhao, Jiazhan Feng, Chongyang Tao, and Daxin Jiang. 2023. Wizardlm: Empowering large language models to follow complex instructions. arXiv preprint arXiv:2304.12244(2023)."},{"key":"e_1_3_2_2_69_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33017346"},{"key":"e_1_3_2_2_70_1","unstructured":"Yuichi Yoshida and Takeru Miyato. 2017. Spectral norm regularization for improving the generalizability of deep learning. arXiv(2017)."},{"key":"e_1_3_2_2_71_1","first-page":"962","article-title":"Fairness constraints: Mechanisms for fair classification. In Artificial intelligence and statistics","author":"Zafar Muhammad Bilal","year":"2017","unstructured":"Muhammad Bilal Zafar, Isabel Valera, Manuel Gomez Rogriguez, and Krishna P Gummadi. 2017. Fairness constraints: Mechanisms for fair classification. In Artificial intelligence and statistics. PMLR, 962-970.","journal-title":"PMLR"},{"key":"e_1_3_2_2_72_1","doi-asserted-by":"crossref","unstructured":"Haiyan Zhao Xuansheng Wu Fan Yang Bo Shen Ninghao Liu and Mengnan Du. 2025. Denoising Concept Vectors with Sparse Autoencoders for Improved Language Model Steering. arXiv preprint arXiv:2505.15038(2025).","DOI":"10.18653\/v1\/2026.findings-eacl.40"},{"key":"e_1_3_2_2_73_1","unstructured":"Lianmin Zheng Wei-Lin Chiang Ying Sheng Tianle Li Siyuan Zhuang Zhanghao Wu Yonghao Zhuang Zhuohan Li Zi Lin Eric P Xing et al. 2023. Lmsys-chat-1m: A large-scale real-world llm conversation dataset. arXiv(2023)."},{"key":"e_1_3_2_2_74_1","doi-asserted-by":"crossref","unstructured":"Yuqing Zhou Ruixiang Tang Ziyu Yao and Ziwei Zhu. 2024. Navigating the shortcut maze: A comprehensive analysis of shortcut learning in text classification by language models. arXiv preprint arXiv:2409.17455(2024).","DOI":"10.18653\/v1\/2024.findings-emnlp.146"},{"key":"e_1_3_2_2_75_1","unstructured":"Yutao Zhu Huaying Yuan Shuting Wang Jiongnan Liu Wenhan Liu Chenlong Deng Zhicheng Dou and Ji-Rong Wen. 2023. Large language models for information retrieval: A survey. arXiv preprint arXiv:2308.07107(2023)."}],"event":{"name":"KDD '25: The 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining","location":"Toronto ON Canada","acronym":"KDD '25","sponsor":["SIGKDD ACM Special Interest Group on Knowledge Discovery in Data","SIGMOD ACM Special Interest Group on Management of Data"]},"container-title":["Proceedings of the 31st ACM SIGKDD Conference on Knowledge Discovery and Data Mining V.2"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3711896.3737120","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T18:11:37Z","timestamp":1777572697000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3711896.3737120"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,8,3]]},"references-count":75,"alternative-id":["10.1145\/3711896.3737120","10.1145\/3711896"],"URL":"https:\/\/doi.org\/10.1145\/3711896.3737120","relation":{},"subject":[],"published":{"date-parts":[[2025,8,3]]},"assertion":[{"value":"2025-08-03","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}