{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,24]],"date-time":"2026-03-24T00:08:59Z","timestamp":1774310939400,"version":"3.50.1"},"publisher-location":"Cham","reference-count":40,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032213204","type":"print"},{"value":"9783032213211","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-21321-1_6","type":"book-chapter","created":{"date-parts":[[2026,3,23]],"date-time":"2026-03-23T11:10:58Z","timestamp":1774264258000},"page":"44-52","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Tutorial on\u00a0Mechanistic Interpretability"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-8734-436X","authenticated-orcid":false,"given":"Catherine","family":"Chen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5360-9627","authenticated-orcid":false,"given":"Maria","family":"Heuss","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9895-4061","authenticated-orcid":false,"given":"Carsten","family":"Eickhoff","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,3,24]]},"reference":[{"key":"6_CR1","unstructured":"Anand, A., Lyu, L., Idahl, M., Wang, Y., Wallat, J., Zhang, Z.: Explainable information retrieval: a survey. arxiv preprint arXiv:2211.02405 (2022)"},{"key":"6_CR2","doi-asserted-by":"crossref","unstructured":"Anand, A., Saha, S., Venktesh, V.: Explainable information retrieval. In: European Conference on Information Retrieval. Springer (2025)","DOI":"10.1007\/978-3-031-88720-8_40"},{"key":"6_CR3","doi-asserted-by":"crossref","unstructured":"Anand, A., Sen, P., Saha, S., Verma, M., Mitra, M.: Explainable information retrieval. In: Proceedings of the 46th International ACM SIGIR Conference on Research and Development in Information Retrieval (2023)","DOI":"10.1145\/3539618.3594249"},{"issue":"1","key":"6_CR4","doi-asserted-by":"publisher","first-page":"207","DOI":"10.1162\/coli_a_00422","volume":"48","author":"Y Belinkov","year":"2022","unstructured":"Belinkov, Y.: Probing classifiers: promises, shortcomings, and advances. Comput. Linguist. 48(1), 207\u2013219 (2022)","journal-title":"Comput. Linguist."},{"key":"6_CR5","unstructured":"Bereska, L., Gavves, S.: Mechanistic interpretability for ai safety-a review. Trans. Mach. Learn. Res. (2024)"},{"key":"6_CR6","unstructured":"Bricken, T., et al.: Towards monosemanticity: decomposing language models with dictionary learning. Transformer Circuits Thread (2023)"},{"key":"6_CR7","doi-asserted-by":"crossref","unstructured":"Chen, C., Merullo, J., Eickhoff, C.: Axiomatic causal interventions for reverse engineering relevance computation in neural retrieval models. In: Proceedings of the 47th International ACM SIGIR Conference on Research and Development in Information Retrieval (2024)","DOI":"10.1145\/3626772.3657841"},{"key":"6_CR8","doi-asserted-by":"crossref","unstructured":"Chowdhury, T., Nijasure, A., Allan, J.: Probing ranking LLMs: a mechanistic analysis for information retrieval. In: Proceedings of the 2025 International ACM SIGIR Conference on Innovative Concepts and Theories in Information Retrieval (2025)","DOI":"10.1145\/3731120.3744603"},{"key":"6_CR9","doi-asserted-by":"crossref","unstructured":"van Dort, I., Maria, H.: How do LLMs cite? A mechanistic interpretation of attribution in rag. In: 48th European Conference on Information Retrieval (2026)","DOI":"10.1007\/978-3-032-21324-2_35"},{"issue":"1","key":"6_CR10","first-page":"12","volume":"1","author":"N Elhage","year":"2021","unstructured":"Elhage, N., et al.: A mathematical framework for transformer circuits. Transformer Circuits Thread 1(1), 12 (2021)","journal-title":"Transformer Circuits Thread"},{"key":"6_CR11","unstructured":"Geiger, A., Lu, H., Icard, T., Potts, C.: Causal abstractions of neural networks. In: Advances in Neural Information Processing Systems (2021)"},{"key":"6_CR12","doi-asserted-by":"crossref","unstructured":"Geva, M., Bastings, J., Filippova, K., Globerson, A.: Dissecting recall of factual associations in auto-regressive language models. In: Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.751"},{"key":"6_CR13","doi-asserted-by":"crossref","unstructured":"Geva, M., Caciularu, A., Wang, K., Goldberg, Y.: Transformer feed-forward layers build predictions by promoting concepts in the vocabulary space. In: Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing (2022)","DOI":"10.18653\/v1\/2022.emnlp-main.3"},{"key":"6_CR14","unstructured":"Goldowsky-Dill, N., MacLeod, C., Sato, L., Arora, A.: Localizing model behavior with path patching. arXiv preprint arXiv:2304.05969 (2023)"},{"key":"6_CR15","unstructured":"Gurnee, W., et al.: Universal neurons in gpt2 language models. arXiv preprint arXiv:2401.12181 (2024)"},{"key":"6_CR16","doi-asserted-by":"crossref","unstructured":"Hanna, M., Liu, O., Variengien, A.: How does GPT-2 compute greater-than?: interpreting mathematical abilities in a pre-trained language model. In: Advances in Neural Information Processing Systems (2023)","DOI":"10.52202\/075280-3322"},{"key":"6_CR17","doi-asserted-by":"crossref","unstructured":"Hendel, R., Geva, M., Globerson, A.: In-context learning creates task vectors. In: Findings of the Association for Computational Linguistics: EMNLP 2023 (2023)","DOI":"10.18653\/v1\/2023.findings-emnlp.624"},{"key":"6_CR18","doi-asserted-by":"crossref","unstructured":"Heuss, M., Chen, C., Anand, A., Eickhoff, C., Verberne, S.: Workshop on explainability in information retrieval. In: Proceedings of the 48th International ACM SIGIR Conference on Research and Development in Information Retrieval (2025)","DOI":"10.1145\/3726302.3730361"},{"key":"6_CR19","doi-asserted-by":"crossref","unstructured":"Huang, J., Geiger, A., D\u2019Oosterlinck, K., Wu, Z., Potts, C.: Rigorously assessing natural language explanations of neurons. In: Proceedings of the 6th BlackboxNLP Workshop: Analyzing and Interpreting Neural Networks for NLP (2023)","DOI":"10.18653\/v1\/2023.blackboxnlp-1.24"},{"key":"6_CR20","unstructured":"Huben, R., Cunningham, H., Smith, L.R., Ewart, A., Sharkey, L.: Sparse autoencoders find highly interpretable features in language models. In: The Twelfth International Conference on Learning Representations (2023)"},{"key":"6_CR21","doi-asserted-by":"crossref","unstructured":"Konen, K., et al.: Style vectors for steering generative large language model. arXiv preprint arXiv:2402.01618 (2024)","DOI":"10.18653\/v1\/2024.findings-eacl.52"},{"key":"6_CR22","doi-asserted-by":"crossref","unstructured":"Li, K., Patel, O., Vi\u00e9gas, F., Pfister, H., Wattenberg, M.: Inference-time intervention: eliciting truthful answers from a language model. In: Advances in Neural Information Processing Systems (2023)","DOI":"10.52202\/075280-1797"},{"key":"6_CR23","doi-asserted-by":"crossref","unstructured":"Liu, Q., Duan, H., Mao, J., Wen, J.R.: How do large language models understand relevance? a mechanistic interpretability perspective. ACM Trans. Inf. Syst. (2025)","DOI":"10.1145\/3774942"},{"key":"6_CR24","doi-asserted-by":"crossref","unstructured":"Lu, M., Chen, C., Eickhoff, C.: Pathway to relevance: how cross-encoders implement a semantic variant of BM25. In: Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing (2025)","DOI":"10.18653\/v1\/2025.emnlp-main.1297"},{"key":"6_CR25","unstructured":"Marks, S., Rager, C., Michaud, E.J., Belinkov, Y., Bau, D., Mueller, A.: Sparse feature circuits: Discovering and editing interpretable causal graphs in language models. arXiv preprint arXiv:2403.19647 (2024)"},{"key":"6_CR26","unstructured":"Meng, K., Bau, D., Andonian, A., Belinkov, Y.: Locating and editing factual associations in GPT. In: Advances in Neural Information Processing Systems (2022)"},{"key":"6_CR27","doi-asserted-by":"crossref","unstructured":"Merullo, J., Eickhoff, C., Pavlick, E.: Language models implement simple word2vec-style vector arithmetic. In: Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers) (2024)","DOI":"10.18653\/v1\/2024.naacl-long.281"},{"key":"6_CR28","unstructured":"nostalgebraist: interpreting GPT: the logit lens. AI alignment forum. AI Alignment Forum (2020)"},{"key":"6_CR29","doi-asserted-by":"crossref","unstructured":"Parry, A., Chen, C., Eickhoff, C., MacAvaney, S.: Mechir: a mechanistic interpretability framework for information retrieval. In: European Conference on Information Retrieval. Springer (2025)","DOI":"10.1007\/978-3-031-88720-8_16"},{"key":"6_CR30","unstructured":"Rai, D., Zhou, Y., Feng, S., Saparov, A., Yao, Z.: A practical review of mechanistic interpretability for transformer-based language models. arXiv preprint arXiv:2407.02646 (2024)"},{"key":"6_CR31","doi-asserted-by":"crossref","unstructured":"Reusch, A., Belinkov, Y.: Reverse-engineering the retrieval process in genir models. In: Proceedings of the 48th International ACM SIGIR Conference on Research and Development in Information Retrieval (2025)","DOI":"10.1145\/3726302.3730076"},{"key":"6_CR32","doi-asserted-by":"crossref","unstructured":"Rimsky, N., Gabrieli, N., Schulz, J., Tong, M., Hubinger, E., Turner, A.: Steering llama 2 via contrastive activation addition. In: Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers) (2024)","DOI":"10.18653\/v1\/2024.acl-long.828"},{"key":"6_CR33","doi-asserted-by":"crossref","unstructured":"Saphra, N., Wiegreffe, S.: Mechanistic? arXiv preprint arXiv:2410.09087 (2024)","DOI":"10.18653\/v1\/2024.blackboxnlp-1.30"},{"key":"6_CR34","doi-asserted-by":"crossref","unstructured":"Subramani, N., Suresh, N., Peters, M.E.: Extracting latent steering vectors from pretrained language models. In: Findings of the Association for Computational Linguistics: ACL 2022 (2022)","DOI":"10.18653\/v1\/2022.findings-acl.48"},{"key":"6_CR35","unstructured":"Templeton, A., et al.: Scaling monosemanticity: extracting interpretable features from claude 3 sonnet. Transformer Circuits Thread (2024). https:\/\/transformer-circuits.pub\/2024\/scaling-monosemanticity\/index.html"},{"key":"6_CR36","unstructured":"Tenney, I., et al.: What do you learn from context? Probing for sentence structure in contextualized word representations. In: International Conference on Learning Representations (2019)"},{"key":"6_CR37","unstructured":"Todd, E., Li, M., Sharma, A.S., Mueller, A., Wallace, B.C., Bau, D.: Function vectors in large language models. In: The Twelfth International Conference on Learning Representations (2024)"},{"key":"6_CR38","unstructured":"Turner, A.M., et al.: Steering language models with activation engineering. arXiv preprint arXiv:2308.10248 (2023)"},{"key":"6_CR39","unstructured":"Vig, J., et al.: Investigating gender bias in language models using causal mediation analysis. In: Advances in Neural Information Processing Systems (2020)"},{"key":"6_CR40","unstructured":"Wang, K.R., Variengien, A., Conmy, A., Shlegeris, B., Steinhardt, J.: Interpretability in the wild: a circuit for indirect object identification in GPT-2 small. In: The Eleventh International Conference on Learning Representations (2023)"}],"container-title":["Lecture Notes in Computer Science","Advances in Information Retrieval"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-21321-1_6","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,23]],"date-time":"2026-03-23T23:13:57Z","timestamp":1774307637000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-21321-1_6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9783032213204","9783032213211"],"references-count":40,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-21321-1_6","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"24 March 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"ECIR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Information Retrieval","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Delft","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"The Netherlands","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 March 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2 April 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"48","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ecir2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ecir2026.eu\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}