{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T16:54:11Z","timestamp":1777654451895,"version":"3.51.4"},"publisher-location":"Cham","reference-count":51,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031727740","type":"print"},{"value":"9783031727757","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,9,30]],"date-time":"2024-09-30T00:00:00Z","timestamp":1727654400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,9,30]],"date-time":"2024-09-30T00:00:00Z","timestamp":1727654400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72775-7_15","type":"book-chapter","created":{"date-parts":[[2024,9,29]],"date-time":"2024-09-29T07:01:50Z","timestamp":1727593310000},"page":"257-273","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Finding Visual Task Vectors"],"prefix":"10.1007","author":[{"given":"Alberto","family":"Hojel","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yutong","family":"Bai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Trevor","family":"Darrell","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Amir","family":"Globerson","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Amir","family":"Bar","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,9,30]]},"reference":[{"key":"15_CR1","unstructured":"Aky\u00fcrek, E., Schuurmans, D., Andreas, J., Ma, T., Zhou, D.: What learning algorithm is in-context learning? Investigations with linear models. arXiv preprint arXiv:2211.15661 (2022)"},{"key":"15_CR2","unstructured":"Bahng, H., Jahanian, A., Sankaranarayanan, S., Isola, P.: Exploring visual prompts for adapting large-scale models. arXiv preprint arXiv:2203.17274 (2022)"},{"key":"15_CR3","doi-asserted-by":"crossref","unstructured":"Bai, Y., et al.: Sequential modeling enables scalable learning for large vision models. arXiv preprint arXiv:2312.00785 (2023)","DOI":"10.1109\/CVPR52733.2024.02157"},{"key":"15_CR4","unstructured":"Bar, A., Gandelsman, Y., Darrell, T., Globerson, A., Efros, A.: Visual prompting via image inpainting. In: Advances in Neural Information Processing Systems, vol. 35, pp. 25005\u201325017 (2022)"},{"key":"15_CR5","unstructured":"Bau, D., et al.: Gan dissection: visualizing and understanding generative adversarial networks. arXiv preprint arXiv:1811.10597 (2018)"},{"key":"15_CR6","unstructured":"Brown, T., et al.: Language models are few-shot learners. In: Advances in Neural Information Processing Systems, vol. 33, pp. 1877\u20131901 (2020)"},{"key":"15_CR7","doi-asserted-by":"crossref","unstructured":"Dai, D., Sun, Y., Dong, L., Hao, Y., Sui, Z., Wei, F.: Why can GPT learn in-context? Language models secretly perform gradient descent as meta optimizers. arXiv preprint arXiv:2212.10559 (2022)","DOI":"10.18653\/v1\/2023.findings-acl.247"},{"key":"15_CR8","doi-asserted-by":"publisher","unstructured":"Davies, D., Bouldin, D.: A cluster separation measure. IEEE Trans. Pattern Anal. Mach. Intell. PAMI-1, 224\u2013227 (1979). https:\/\/doi.org\/10.1109\/TPAMI.1979.4766909","DOI":"10.1109\/TPAMI.1979.4766909"},{"key":"15_CR9","unstructured":"Dosovitskiy, A., et al.: An image is worth 16x16 words: transformers for image recognition at scale (2021)"},{"key":"15_CR10","doi-asserted-by":"crossref","unstructured":"Esser, P., Rombach, R., Ommer, B.: Taming transformers for high-resolution image synthesis. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 12873\u201312883, June 2021","DOI":"10.1109\/CVPR46437.2021.01268"},{"key":"15_CR11","unstructured":"Ferry, Q.R., Ching, J., Kawai, T.: Emergence and function of abstract representations in self-supervised transformers. arXiv preprint arXiv:2312.05361 (2023)"},{"key":"15_CR12","unstructured":"Gandelsman, Y., Efros, A.A., Steinhardt, J.: Interpreting CLIP\u2019s image representation via text-based decomposition. arXiv preprint arXiv:2310.05916 (2023)"},{"key":"15_CR13","unstructured":"Garg, S., Tsipras, D., Liang, P.S., Valiant, G.: What can transformers learn in-context? A case study of simple function classes. In: Advances in Neural Information Processing Systems, vol. 35, pp. 30583\u201330598 (2022)"},{"key":"15_CR14","unstructured":"Hahn, M., Goyal, N.: A theory of emergent in-context learning as implicit structure induction. arXiv preprint arXiv:2303.07971 (2023)"},{"key":"15_CR15","unstructured":"Han, C., Wang, Z., Zhao, H., Ji, H.: In-context learning of large language models explained as kernel regression. arXiv preprint arXiv:2305.12766 (2023)"},{"key":"15_CR16","unstructured":"He, K., Chen, X., Xie, S., Li, Y., Doll\u00e1r, P., Girshick, R.B.: Masked autoencoders are scalable vision learners. CoRR abs\/2111.06377 (2021). https:\/\/arxiv.org\/abs\/2111.06377"},{"key":"15_CR17","doi-asserted-by":"crossref","unstructured":"Hendel, R., Geva, M., Globerson, A.: In-context learning creates task vectors. arXiv preprint arXiv:2310.15916 (2023)","DOI":"10.18653\/v1\/2023.findings-emnlp.624"},{"key":"15_CR18","doi-asserted-by":"publisher","unstructured":"Jia, M., et al.: Visual prompt tuning. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13693, pp. 709\u2013727. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19827-4_41","DOI":"10.1007\/978-3-031-19827-4_41"},{"key":"15_CR19","doi-asserted-by":"crossref","unstructured":"Jin, Z., et al.: Cutting off the head ends the conflict: a mechanism for interpreting and mitigating knowledge conflicts in language models. arXiv preprint arXiv:2402.18154 (2024)","DOI":"10.18653\/v1\/2024.findings-acl.70"},{"key":"15_CR20","unstructured":"Kingma, D.P., Ba, J.: Adam: a method for stochastic optimization (2017)"},{"key":"15_CR21","doi-asserted-by":"crossref","unstructured":"Li, X.L., Liang, P.: Prefix-tuning: optimizing continuous prompts for generation. arXiv preprint arXiv:2101.00190 (2021)","DOI":"10.18653\/v1\/2021.acl-long.353"},{"key":"15_CR22","doi-asserted-by":"crossref","unstructured":"Liu, J., Shen, D., Zhang, Y., Dolan, B., Carin, L., Chen, W.: What makes good in-context examples for GPT-$$3 $$? arXiv preprint arXiv:2101.06804 (2021)","DOI":"10.18653\/v1\/2022.deelio-1.10"},{"key":"15_CR23","unstructured":"Liu, S., Xing, L., Zou, J.: In-context vectors: making in context learning more effective and controllable through latent space steering. arXiv preprint arXiv:2311.06668 (2023)"},{"key":"15_CR24","unstructured":"Lu, S., Schuff, H., Gurevych, I.: How are prompts different in terms of sensitivity? arXiv preprint arXiv:2311.07230 (2023)"},{"key":"15_CR25","doi-asserted-by":"crossref","unstructured":"Lu, Y., Bartolo, M., Moore, A., Riedel, S., Stenetorp, P.: Fantastically ordered prompts and where to find them: overcoming few-shot prompt order sensitivity. arXiv preprint arXiv:2104.08786 (2021)","DOI":"10.18653\/v1\/2022.acl-long.556"},{"key":"15_CR26","unstructured":"Luo, H., Specia, L.: From understanding to utilization: a survey on explainability for large language models. arXiv preprint arXiv:2401.12874 (2024)"},{"key":"15_CR27","unstructured":"van\u00a0der Maaten, L., Hinton, G.: Visualizing data using t-SNE. J. Mach. Learn. Res. 9(86), 2579\u20132605 (2008). http:\/\/jmlr.org\/papers\/v9\/vandermaaten08a.html"},{"key":"15_CR28","unstructured":"Meng, K., Bau, D., Andonian, A., Belinkov, Y.: Locating and editing factual associations in GPT. In: Advances in Neural Information Processing Systems, vol. 35, pp. 17359\u201317372 (2022)"},{"issue":"1","key":"15_CR29","doi-asserted-by":"publisher","first-page":"18","DOI":"10.1145\/3400051.3400058","volume":"22","author":"R Moraffah","year":"2020","unstructured":"Moraffah, R., Karami, M., Guo, R., Raglin, A., Liu, H.: Causal interpretability for machine learning-problems, methods and evaluation. ACM SIGKDD Explor. Newsl. 22(1), 18\u201333 (2020)","journal-title":"ACM SIGKDD Explor. Newsl."},{"key":"15_CR30","doi-asserted-by":"crossref","unstructured":"Palit, V., Pandey, R., Arora, A., Liang, P.P.: Towards vision-language mechanistic interpretability: a causal tracing tool for blip. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2856\u20132861 (2023)","DOI":"10.1109\/ICCVW60793.2023.00307"},{"key":"15_CR31","unstructured":"Park, K., Choe, Y.J., Veitch, V.: The linear representation hypothesis and the geometry of large language models. arXiv preprint arXiv:2311.03658 (2023)"},{"key":"15_CR32","doi-asserted-by":"crossref","unstructured":"Pearl, J.: Direct and indirect effects. In: Probabilistic and Causal Inference: The Works of Judea Pearl, pp. 373\u2013392 (2022)","DOI":"10.1145\/3501714.3501736"},{"issue":"8","key":"15_CR33","first-page":"9","volume":"1","author":"A Radford","year":"2019","unstructured":"Radford, A., Wu, J., Child, R., Luan, D., Amodei, D., Sutskever, I., et al.: Language models are unsupervised multitask learners. OpenAI Blog 1(8), 9 (2019)","journal-title":"OpenAI Blog"},{"key":"15_CR34","doi-asserted-by":"publisher","unstructured":"Rousseeuw, P.J.: Silhouettes: a graphical aid to the interpretation and validation of cluster analysis. J. Comput. Appl. Math. 20, 53\u201365 (1987). https:\/\/doi.org\/10.1016\/0377-0427(87)90125-7. https:\/\/www.sciencedirect.com\/science\/article\/pii\/0377042787901257","DOI":"10.1016\/0377-0427(87)90125-7"},{"key":"15_CR35","doi-asserted-by":"crossref","unstructured":"Russakovsky, O., et al.: ImageNet large scale visual recognition challenge (2015)","DOI":"10.1007\/s11263-015-0816-y"},{"key":"15_CR36","doi-asserted-by":"crossref","unstructured":"Shaban, A., Bansal, S., Liu, Z., Essa, I., Boots, B.: One-shot learning for semantic segmentation. arXiv preprint arXiv:1709.03410 (2017)","DOI":"10.5244\/C.31.167"},{"key":"15_CR37","unstructured":"Singh, C., Inala, J.P., Galley, M., Caruana, R., Gao, J.: Rethinking interpretability in the era of large language models. arXiv preprint arXiv:2402.01761 (2024)"},{"key":"15_CR38","unstructured":"Todd, E., Li, M.L., Sharma, A.S., Mueller, A., Wallace, B.C., Bau, D.: Function vectors in large language models. arXiv preprint arXiv:2310.15213 (2023)"},{"key":"15_CR39","unstructured":"Touvron, H., et\u00a0al.: LLaMA: open and efficient foundation language models. arXiv preprint arXiv:2302.13971 (2023)"},{"key":"15_CR40","unstructured":"Vaswani, A., et al.: Attention is all you need. CoRR abs\/1706.03762 (2017). http:\/\/arxiv.org\/abs\/1706.03762"},{"key":"15_CR41","unstructured":"Wang, B., Komatsuzaki, A.: GPT-J-6B: a 6 billion parameter autoregressive language model (2021)"},{"key":"15_CR42","unstructured":"Wei, J., et al.: Chain-of-thought prompting elicits reasoning in large language models. In: Advances in Neural Information Processing Systems, vol. 35, pp. 24824\u201324837 (2022)"},{"key":"15_CR43","doi-asserted-by":"publisher","first-page":"229","DOI":"10.1007\/BF00992696","volume":"8","author":"RJ Williams","year":"1992","unstructured":"Williams, R.J.: Simple statistical gradient-following algorithms for connectionist reinforcement learning. Mach. Learn. 8, 229\u2013256 (1992)","journal-title":"Mach. Learn."},{"key":"15_CR44","unstructured":"Wu, X., Varshney, L.R.: Transformer-based causal language models perform clustering. arXiv preprint arXiv:2402.12151 (2024)"},{"key":"15_CR45","unstructured":"Xie, S.M., Raghunathan, A., Liang, P., Ma, T.: An explanation of in-context learning as implicit Bayesian inference. arXiv preprint arXiv:2111.02080 (2021)"},{"key":"15_CR46","unstructured":"Xu, J., et al.: IMProv: inpainting-based multimodal prompting for computer vision tasks. arXiv preprint arXiv:2312.01771 (2023)"},{"key":"15_CR47","doi-asserted-by":"crossref","unstructured":"Xu, S., Dong, W., Guo, Z., Wu, X., Xiong, D.: Exploring multilingual human value concepts in large language models: is value alignment consistent, transferable and controllable across languages? arXiv preprint arXiv:2402.18120 (2024)","DOI":"10.18653\/v1\/2024.findings-emnlp.96"},{"key":"15_CR48","unstructured":"Zhang, F., Nanda, N.: Towards best practices of activation patching in language models: metrics and methods. arXiv preprint arXiv:2309.16042 (2023)"},{"key":"15_CR49","doi-asserted-by":"crossref","unstructured":"Zhang, K., Lv, A., Chen, Y., Ha, H., Xu, T., Yan, R.: Batch-ICL: effective, efficient, and order-agnostic in-context learning. arXiv preprint arXiv:2401.06469 (2024)","DOI":"10.18653\/v1\/2024.findings-acl.638"},{"issue":"5","key":"15_CR50","doi-asserted-by":"publisher","first-page":"726","DOI":"10.1109\/TETCI.2021.3100641","volume":"5","author":"Y Zhang","year":"2021","unstructured":"Zhang, Y., Ti\u0148o, P., Leonardis, A., Tang, K.: A survey on neural network interpretability. IEEE Trans. Emerging Top. Comput. Intell. 5(5), 726\u2013742 (2021)","journal-title":"IEEE Trans. Emerging Top. Comput. Intell."},{"key":"15_CR51","unstructured":"Zhang, Y., Zhou, K., Liu, Z.: What makes good examples for visual in-context learning? In: Advances in Neural Information Processing Systems, vol. 36 (2024)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72775-7_15","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,28]],"date-time":"2024-11-28T21:21:57Z","timestamp":1732828917000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72775-7_15"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,9,30]]},"ISBN":["9783031727740","9783031727757"],"references-count":51,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72775-7_15","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,9,30]]},"assertion":[{"value":"30 September 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}