{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,25]],"date-time":"2025-09-25T13:38:46Z","timestamp":1758807526358,"version":"3.40.3"},"publisher-location":"Cham","reference-count":35,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783031064326"},{"type":"electronic","value":"9783031064333"}],"license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022]]},"DOI":"10.1007\/978-3-031-06433-3_5","type":"book-chapter","created":{"date-parts":[[2022,5,14]],"date-time":"2022-05-14T18:03:24Z","timestamp":1652551404000},"page":"50-61","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["Recurrent Vision Transformer for\u00a0Solving Visual Reasoning Problems"],"prefix":"10.1007","author":[{"given":"Nicola","family":"Messina","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Giuseppe","family":"Amato","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fabio","family":"Carrara","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Claudio","family":"Gennaro","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fabrizio","family":"Falchi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,5,15]]},"reference":[{"unstructured":"Banino, A., Balaguer, J., Blundell, C.: PonderNet: learning to ponder. arXiv preprint arXiv:2107.05407 (2021)","key":"5_CR1"},{"unstructured":"Bertasius, G., Wang, H., Torresani, L.: Is space-time attention all you need for video understanding? arXiv preprint arXiv:2102.05095 (2021)","key":"5_CR2"},{"doi-asserted-by":"crossref","unstructured":"Borowski, J., Funke, C.M., Stosio, K., Brendel, W., Wallis, T., Bethge, M.: The notorious difficulty of comparing human and machine perception. In: 2019 Conference on Cognitive Computational Neuroscience, pp. 2019\u20131295 (2019)","key":"5_CR3","DOI":"10.32470\/CCN.2019.1295-0"},{"key":"5_CR4","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"213","DOI":"10.1007\/978-3-030-58452-8_13","volume-title":"Computer Vision \u2013 ECCV 2020","author":"N Carion","year":"2020","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., Zagoruyko, S.: End-to-end object detection with transformers. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12346, pp. 213\u2013229. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58452-8_13"},{"doi-asserted-by":"crossref","unstructured":"Cho, K., et al.: Learning phrase representations using RNN encoder-decoder for statistical machine translation. arXiv preprint arXiv:1406.1078 (2014)","key":"5_CR5","DOI":"10.3115\/v1\/D14-1179"},{"issue":"18","key":"5_CR6","doi-asserted-by":"publisher","first-page":"5250","DOI":"10.3390\/s20185250","volume":"20","author":"L Ciampi","year":"2020","unstructured":"Ciampi, L., Messina, N., Falchi, F., Gennaro, C., Amato, G.: Virtual to real adaptation of pedestrian detectors. Sensors 20(18), 5250 (2020)","journal-title":"Sensors"},{"doi-asserted-by":"crossref","unstructured":"Coccomini, D., Messina, N., Gennaro, C., Falchi, F.: Combining efficientnet and vision transformers for video deepfake detection. arXiv preprint arXiv:2107.02612 (2021)","key":"5_CR7","DOI":"10.1007\/978-3-031-06433-3_19"},{"unstructured":"Dehghani, M., Gouws, S., Vinyals, O., Uszkoreit, J., Kaiser, \u0141.: Universal transformers. arXiv preprint arXiv:1807.03819 (2018)","key":"5_CR8"},{"unstructured":"Devlin, J., Chang, M., Lee, K., Toutanova, K.: BERT: pre-training of deep bidirectional transformers for language understanding. In: NAACL-HLT 2019, pp. 4171\u20134186. Association for Computational Linguistics (2019)","key":"5_CR9"},{"unstructured":"Doersch, C., Gupta, A., Zisserman, A.: Crosstransformers: spatially-aware few-shot transfer. arXiv preprint arXiv:2007.11498 (2020)","key":"5_CR10"},{"unstructured":"Dosovitskiy, A., et al.: An image is worth 16x16 words: transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)","key":"5_CR11"},{"issue":"43","key":"5_CR12","doi-asserted-by":"publisher","first-page":"17621","DOI":"10.1073\/pnas.1109168108","volume":"108","author":"F Fleuret","year":"2011","unstructured":"Fleuret, F., Li, T., Dubout, C., Wampler, E.K., Yantis, S., Geman, D.: Comparing machines and humans on a visual categorization test. Proc. Natl. Acad. Sci. 108(43), 17621\u201317625 (2011)","journal-title":"Proc. Natl. Acad. Sci."},{"unstructured":"Foret, P., Kleiner, A., Mobahi, H., Neyshabur, B.: Sharpness-aware minimization for efficiently improving generalization. arXiv preprint arXiv:2010.01412 (2020)","key":"5_CR13"},{"issue":"3","key":"5_CR14","doi-asserted-by":"publisher","first-page":"16","DOI":"10.1167\/jov.21.3.16","volume":"21","author":"CM Funke","year":"2021","unstructured":"Funke, C.M., Borowski, J., Stosio, K., Brendel, W., Wallis, T.S., Bethge, M.: Five points to check when comparing visual perception in humans and machines. J. Vis. 21(3), 16\u201316 (2021)","journal-title":"J. Vis."},{"unstructured":"Graves, A., Wayne, G., Danihelka, I.: Neural turing machines. arXiv preprint arXiv:1410.5401 (2014)","key":"5_CR15"},{"issue":"8","key":"5_CR16","doi-asserted-by":"publisher","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter, S., Schmidhuber, J.: Long short-term memory. Neural Comput. 9(8), 1735\u20131780 (1997)","journal-title":"Neural Comput."},{"doi-asserted-by":"crossref","unstructured":"Johnson, J., Hariharan, B., van der Maaten, L., Fei-Fei, L., Lawrence Zitnick, C., Girshick, R.: CLEVR: a diagnostic dataset for compositional language and elementary visual reasoning. In: Proceedings of IEEE CVPR, pp. 2901\u20132910 (2017)","key":"5_CR17","DOI":"10.1109\/CVPR.2017.215"},{"issue":"6","key":"5_CR18","doi-asserted-by":"publisher","first-page":"974","DOI":"10.1038\/s41593-019-0392-5","volume":"22","author":"K Kar","year":"2019","unstructured":"Kar, K., Kubilius, J., Schmidt, K., Issa, E.B., DiCarlo, J.J.: Evidence that recurrent circuits are critical to the ventral stream\u2019s execution of core object recognition behavior. Nat. Neurosci. 22(6), 974\u2013983 (2019)","journal-title":"Nat. Neurosci."},{"unstructured":"Kendall, A., Gal, Y., Cipolla, R.: Multi-task learning using uncertainty to weigh losses for scene geometry and semantics. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7482\u20137491 (2018)","key":"5_CR19"},{"issue":"4","key":"5_CR20","doi-asserted-by":"publisher","first-page":"20180011","DOI":"10.1098\/rsfs.2018.0011","volume":"8","author":"J Kim","year":"2018","unstructured":"Kim, J., Ricci, M., Serre, T.: Not-so-CLEVR: learning same-different relations strains feedforward neural networks. Interface Focus 8(4), 20180011 (2018)","journal-title":"Interface Focus"},{"doi-asserted-by":"crossref","unstructured":"Messina, N., Amato, G., Carrara, F., Falchi, F., Gennaro, C.: Testing deep neural networks on the same-different task. In: 2019 International Conference on Content-Based Multimedia Indexing (CBMI), pp. 1\u20136. IEEE (2019)","key":"5_CR21","DOI":"10.1109\/CBMI.2019.8877412"},{"key":"5_CR22","doi-asserted-by":"publisher","first-page":"75","DOI":"10.1016\/j.patrec.2020.12.019","volume":"143","author":"N Messina","year":"2021","unstructured":"Messina, N., Amato, G., Carrara, F., Gennaro, C., Falchi, F.: Solving the same-different task with convolutional neural networks. Pattern Recogn. Lett. 143, 75\u201380 (2021)","journal-title":"Pattern Recogn. Lett."},{"doi-asserted-by":"crossref","unstructured":"Messina, N., Amato, G., Esuli, A., Falchi, F., Gennaro, C., Marchand-Maillet, S.: Fine-grained visual textual alignment for cross-modal retrieval using transformer encoders. arXiv preprint arXiv:2008.05231 (2020)","key":"5_CR23","DOI":"10.1145\/3451390"},{"doi-asserted-by":"crossref","unstructured":"Messina, N., Falchi, F., Esuli, A., Amato, G.: Transformer reasoning network for image-text matching and retrieval. In: 2020 25th International Conference on Pattern Recognition (ICPR), pp. 5222\u20135229. IEEE (2021)","key":"5_CR24","DOI":"10.1109\/ICPR48806.2021.9413172"},{"doi-asserted-by":"crossref","unstructured":"Puebla, G., Bowers, J.S.: Can deep convolutional neural networks learn same-different relations? bioRxiv (2021)","key":"5_CR25","DOI":"10.1101\/2021.04.06.438551"},{"unstructured":"Redmon, J., Farhadi, A.: YOLOv3: an incremental improvement. arXiv preprint arXiv:1804.02767 (2018)","key":"5_CR26"},{"key":"5_CR27","first-page":"91","volume":"28","author":"S Ren","year":"2015","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster R-CNN: towards real-time object detection with region proposal networks. Adv. Neural. Inf. Process. Syst. 28, 91\u201399 (2015)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"unstructured":"Santoro, A., Hill, F., Barrett, D., Morcos, A., Lillicrap, T.: Measuring abstract reasoning in neural networks. In: International Conference on Machine Learning, pp. 4477\u20134486 (2018)","key":"5_CR28"},{"unstructured":"Santoro, A., et al.: A simple neural network module for relational reasoning. In: Advances in Neural Information Processing Systems, pp. 4967\u20134976 (2017)","key":"5_CR29"},{"key":"5_CR30","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"380","DOI":"10.1007\/978-3-319-44781-0_45","volume-title":"Artificial Neural Networks and Machine Learning \u2013 ICANN 2016","author":"S Stabinger","year":"2016","unstructured":"Stabinger, S., Rodr\u00edguez-S\u00e1nchez, A., Piater, J.: 25 years of CNNs: can we compare to human abstraction capabilities? In: Villa, A.E.P., Masulli, P., Pons Rivero, A.J. (eds.) ICANN 2016. LNCS, vol. 9887, pp. 380\u2013387. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-44781-0_45"},{"doi-asserted-by":"crossref","unstructured":"Szegedy, C., et al.: Going deeper with convolutions. In: Proceedings of IEEE CVPR, pp. 1\u20139 (2015)","key":"5_CR31","DOI":"10.1109\/CVPR.2015.7298594"},{"doi-asserted-by":"crossref","unstructured":"Vaishnav, M., Cadene, R., Alamia, A., Linsley, D., Vanrullen, R., Serre, T.: Understanding the computational demands underlying visual reasoning. arXiv preprint arXiv:2108.03603 (2021)","key":"5_CR32","DOI":"10.1162\/neco_a_01485"},{"unstructured":"Vaswani, A., et al.: Attention is all you need. In: Advances in Neural Information Processing Systems, pp. 5998\u20136008 (2017)","key":"5_CR33"},{"unstructured":"Weiler, M., Cesa, G.: General E(2)-equivariant steerable CNNs. In: Conference on Neural Information Processing Systems (NeurIPS) (2019)","key":"5_CR34"},{"doi-asserted-by":"crossref","unstructured":"Xie, S., Girshick, R., Doll\u00e1r, P., Tu, Z., He, K.: Aggregated residual transformations for deep neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1492\u20131500 (2017)","key":"5_CR35","DOI":"10.1109\/CVPR.2017.634"}],"container-title":["Lecture Notes in Computer Science","Image Analysis and Processing \u2013 ICIAP 2022"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-06433-3_5","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,7]],"date-time":"2024-03-07T12:08:15Z","timestamp":1709813295000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-06433-3_5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"ISBN":["9783031064326","9783031064333"],"references-count":35,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-06433-3_5","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2022]]},"assertion":[{"value":"15 May 2022","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICIAP","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Image Analysis and Processing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Lecce","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2022","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 May 2022","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 May 2022","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"21","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"iciap2022","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.iciap2021.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Microsoft","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"307","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"168","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"55% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"4","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}