{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T14:13:17Z","timestamp":1784556797353,"version":"3.55.0"},"publisher-location":"Cham","reference-count":30,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032051844","type":"print"},{"value":"9783032051851","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,9,20]],"date-time":"2025-09-20T00:00:00Z","timestamp":1758326400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,9,20]],"date-time":"2025-09-20T00:00:00Z","timestamp":1758326400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-05185-1_29","type":"book-chapter","created":{"date-parts":[[2025,9,19]],"date-time":"2025-09-19T23:47:31Z","timestamp":1758325651000},"page":"295-304","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Interpretable fMRI Captioning via\u00a0Contrastive Learning"],"prefix":"10.1007","author":[{"given":"Vyacheslav","family":"Shen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kassymzhomart","family":"Kunanbayev","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Donggon","family":"Jang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Daeshik","family":"Kim","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,9,20]]},"reference":[{"key":"29_CR1","doi-asserted-by":"crossref","unstructured":"Allen, E.J., et al.: A massive 7t fMRI dataset to bridge cognitive neuroscience and artificial intelligence. Nat. Neurosci. 25(1), 116\u2013126 (2022)","DOI":"10.1038\/s41593-021-00962-x"},{"key":"29_CR2","doi-asserted-by":"crossref","unstructured":"Epstein, R., Kanwisher, N.: A Cortical representation of the local visual environment. Nature 392(6676), 598\u2013601 (1998)","DOI":"10.1038\/33402"},{"key":"29_CR3","unstructured":"Falcon, W.A.: Pytorch lightning. GitHub 3 (2019)"},{"key":"29_CR4","unstructured":"Ferrante, M., Ozcelik, F., Boccato, T., VanRullen, R., Toschi, N.: Brain captioning: decoding human brain activity into images and text. arXiv preprint arXiv:2305.11560 (2023)"},{"key":"29_CR5","doi-asserted-by":"crossref","unstructured":"Haxby, J.V., Gobbini, M.I., Furey, M.L., Ishai, A., Schouten, J.L., Pietrini, P.: Distributed and overlapping representations of faces and objects in ventral temporal cortex. Science 293(5539), 2425\u20132430 (2001)","DOI":"10.1126\/science.1063736"},{"key":"29_CR6","doi-asserted-by":"crossref","unstructured":"Hubel, D.H., Wiesel, T.N.: Receptive fields, binocular interaction and functional architecture in the cat\u2019s visual cortex. J. Physiol. 160(1), 106 (1962)","DOI":"10.1113\/jphysiol.1962.sp006837"},{"key":"29_CR7","doi-asserted-by":"crossref","unstructured":"Huth, A.G., De Heer, W.A., Griffiths, T.L., Theunissen, F.E., Gallant, J.L.: Natural speech reveals the semantic maps that tile human cerebral cortex. Nature 532(7600), 453\u2013458 (2016)","DOI":"10.1038\/nature17637"},{"key":"29_CR8","doi-asserted-by":"crossref","unstructured":"LeCun, Y., Boser, B., Denker, J.S., Henderson, D., Howard, R.E., Hubbard, W., Jackel, L.D.: Backpropagation applied to handwritten zip code recognition. Neural Comput. 1(4), 541\u2013551 (1989)","DOI":"10.1162\/neco.1989.1.4.541"},{"key":"29_CR9","unstructured":"Li, J., Li, D., Savarese, S., Hoi, S.: Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models. In: International Conference on Machine Learning, pp. 19730\u201319742. PMLR (2023)"},{"key":"29_CR10","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"740","DOI":"10.1007\/978-3-319-10602-1_48","volume-title":"Computer Vision \u2013 ECCV 2014","author":"T-Y Lin","year":"2014","unstructured":"Lin, T.-Y., Maire, M., Belongie, S., Hays, J., Perona, P., Ramanan, D., Doll\u00e1r, P., Zitnick, C.L.: Microsoft COCO: Common Objects in Context. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8693, pp. 740\u2013755. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10602-1_48"},{"key":"29_CR11","doi-asserted-by":"crossref","unstructured":"Liu, H., Li, C., Wu, Q., Lee, Y.J.: Visual instruction tuning. In: Advances in Neural Information Processing Systems, vol. 36, pp. 34892\u201334916 (2023)","DOI":"10.52202\/075280-1516"},{"key":"29_CR12","unstructured":"Mai, W., Zhang, Z.: Unibrain: Unify image reconstruction and captioning all in one diffusion model from human brain activity. arXiv preprint arXiv:2308.07428 (2023)"},{"key":"29_CR13","doi-asserted-by":"crossref","unstructured":"Ozcelik, F., VanRullen, R.: Natural scene reconstruction from fMRI signals using generative latent diffusion. Sci. Rep. 13(1), 15666 (2023)","DOI":"10.1038\/s41598-023-42891-8"},{"key":"29_CR14","unstructured":"Radford, A., et\u00a0al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PMLR (2021)"},{"key":"29_CR15","unstructured":"Radford, A., Wu, J., Child, R., Luan, D., Amodei, D., Sutskever, I., et al.: Language models are unsupervised multitask learners. OpenAI blog 1(8), 9 (2019)"},{"key":"29_CR16","doi-asserted-by":"crossref","unstructured":"Reimers, N., Gurevych, I.: Making monolingual sentence embeddings multilingual using knowledge distillation. arXiv preprint arXiv:2004.09813 (2020)","DOI":"10.18653\/v1\/2020.emnlp-main.365"},{"key":"29_CR17","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., Ommer, B.: High-resolution image synthesis with latent diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10684\u201310695 (2022)","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"29_CR18","doi-asserted-by":"crossref","unstructured":"Scotti, P., et\u00a0al.: Reconstructing the mind\u2019s eye: fMRI-to-image with contrastive learning and diffusion priors. In: Advances in Neural Information Processing Systems, vol. 36 (2024)","DOI":"10.52202\/075280-1073"},{"key":"29_CR19","unstructured":"Scotti, P.S., et\u00a0al.: Mindeye2: shared-subject models enable fMRI-to-image with 1 hour of data. arXiv preprint arXiv:2403.11207 (2024)"},{"key":"29_CR20","doi-asserted-by":"crossref","unstructured":"Smith, L.N.: Cyclical learning rates for training neural networks. In: 2017 IEEE Winter Conference on Applications of Computer Vision (WACV), pp. 464\u2013472. IEEE (2017)","DOI":"10.1109\/WACV.2017.58"},{"key":"29_CR21","doi-asserted-by":"crossref","unstructured":"Spiridon, M., Kanwisher, N.: How distributed is visual category information in human occipito-temporal cortex? an fMRI study. Neuron 35(6), 1157\u20131165 (2002)","DOI":"10.1016\/S0896-6273(02)00877-2"},{"key":"29_CR22","doi-asserted-by":"crossref","unstructured":"Takagi, Y., Nishimoto, S.: High-resolution image reconstruction with latent diffusion models from human brain activity. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14453\u201314463 (2023)","DOI":"10.1109\/CVPR52729.2023.01389"},{"key":"29_CR23","doi-asserted-by":"crossref","unstructured":"Tang, J., LeBel, A., Jain, S., Huth, A.G.: Semantic reconstruction of continuous language from non-invasive brain recordings. Nat. Neurosci. 26(5), 858\u2013866 (2023)","DOI":"10.1038\/s41593-023-01304-9"},{"key":"29_CR24","unstructured":"Vaswani, A.: Attention is all you need. In: Advances in Neural Information Processing Systems (2017)"},{"key":"29_CR25","unstructured":"Wang, J., et al.: Git: a generative image-to-text transformer for vision and language. arXiv preprint arXiv:2205.14100 (2022)"},{"key":"29_CR26","doi-asserted-by":"crossref","unstructured":"Wen, H., Shi, J., Zhang, Y., Lu, K.H., Cao, J., Liu, Z.: Neural encoding and decoding with deep learning for dynamic natural vision. Cereb. Cortex 28(12), 4136\u20134160 (2018)","DOI":"10.1093\/cercor\/bhx268"},{"key":"29_CR27","doi-asserted-by":"crossref","unstructured":"Xu, X., Wang, Z., Zhang, G., Wang, K., Shi, H.: Versatile diffusion: text, images and variations all in one diffusion model. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7754\u20137765 (2023)","DOI":"10.1109\/ICCV51070.2023.00713"},{"key":"29_CR28","doi-asserted-by":"crossref","unstructured":"Yamins, D.L., Hong, H., Cadieu, C.F., Solomon, E.A., Seibert, D., DiCarlo, J.J.: Performance-optimized hierarchical models predict neural responses in higher visual cortex. Proc. Natl. Acad. Sci. 111(23), 8619\u20138624 (2014)","DOI":"10.1073\/pnas.1403112111"},{"key":"29_CR29","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"818","DOI":"10.1007\/978-3-319-10590-1_53","volume-title":"Computer Vision \u2013 ECCV 2014","author":"MD Zeiler","year":"2014","unstructured":"Zeiler, M.D., Fergus, R.: Visualizing and Understanding Convolutional Networks. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8689, pp. 818\u2013833. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10590-1_53"},{"key":"29_CR30","unstructured":"Zhang, S., et\u00a0al.: OPT: Open pre-trained transformer language models. arXiv preprint arXiv:2205.01068 (2022)"}],"container-title":["Lecture Notes in Computer Science","Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2025"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-05185-1_29","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T13:13:47Z","timestamp":1784553227000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-05185-1_29"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,9,20]]},"ISBN":["9783032051844","9783032051851"],"references-count":30,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-05185-1_29","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,9,20]]},"assertion":[{"value":"20 September 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests.","order":1,"name":"Ethics","label":"Disclosure of Interests","group":{"name":"EthicsHeading","label":"Ethics"}},{"value":"MICCAI","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Medical Image Computing and Computer-Assisted Intervention","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Daejeon","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Korea (Republic of)","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 September 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 September 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"miccai2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/conferences.miccai.org\/2025\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}