{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T12:08:33Z","timestamp":1781611713320,"version":"3.54.5"},"publisher-location":"Cham","reference-count":79,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031200588","type":"print"},{"value":"9783031200595","type":"electronic"}],"license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022]]},"DOI":"10.1007\/978-3-031-20059-5_32","type":"book-chapter","created":{"date-parts":[[2022,10,28]],"date-time":"2022-10-28T16:02:50Z","timestamp":1666972970000},"page":"558-575","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":22,"title":["The Abduction of\u00a0Sherlock Holmes: A Dataset for\u00a0Visual Abductive Reasoning"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4012-8979","authenticated-orcid":false,"given":"Jack","family":"Hessel","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3801-294X","authenticated-orcid":false,"given":"Jena D.","family":"Hwang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3678-6925","authenticated-orcid":false,"given":"Jae Sung","family":"Park","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1361-9868","authenticated-orcid":false,"given":"Rowan","family":"Zellers","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6264-0378","authenticated-orcid":false,"given":"Chandra","family":"Bhagavatula","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1161-6006","authenticated-orcid":false,"given":"Anna","family":"Rohrbach","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5704-7614","authenticated-orcid":false,"given":"Kate","family":"Saenko","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3032-5378","authenticated-orcid":false,"given":"Yejin","family":"Choi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2022,10,29]]},"reference":[{"key":"32_CR1","series-title":"Springer Handbooks","doi-asserted-by":"publisher","first-page":"219","DOI":"10.1007\/978-3-319-30526-4_10","volume-title":"Springer Handbook of Model-Based Science","author":"A Aliseda","year":"2017","unstructured":"Aliseda, A.: The logic of abduction: an introduction. In: Magnani, L., Bertolotti, T. (eds.) Springer Handbook of Model-Based Science. SH, pp. 219\u2013230. Springer, Cham (2017). https:\/\/doi.org\/10.1007\/978-3-319-30526-4_10"},{"key":"32_CR2","doi-asserted-by":"crossref","unstructured":"Anderson, P., et al.: Bottom-up and top-down attention for image captioning and visual question answering. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00636"},{"key":"32_CR3","doi-asserted-by":"crossref","unstructured":"Antol, S., et al.: VQA: visual question answering. In: ICCV (2015)","DOI":"10.1109\/ICCV.2015.279"},{"key":"32_CR4","doi-asserted-by":"publisher","first-page":"587","DOI":"10.1162\/tacl_a_00041","volume":"6","author":"EM Bender","year":"2018","unstructured":"Bender, E.M., Friedman, B.: Data statements for natural language processing: toward mitigating system bias and enabling better science. TACL 6, 587\u2013604 (2018)","journal-title":"TACL"},{"key":"32_CR5","doi-asserted-by":"crossref","unstructured":"Berg, A.C., et al.: Understanding and predicting importance in images. In: CVPR (2012)","DOI":"10.1109\/CVPR.2012.6248100"},{"key":"32_CR6","unstructured":"Bhagavatula, C., et al.: Abductive commonsense reasoning. In: ICLR (2020)"},{"key":"32_CR7","first-page":"993","volume":"3","author":"DM Blei","year":"2003","unstructured":"Blei, D.M., Ng, A.Y., Jordan, M.I.: Latent dirichlet allocation. JMLR 3, 993\u20131022 (2003)","journal-title":"JMLR"},{"issue":"2","key":"32_CR8","doi-asserted-by":"publisher","first-page":"193","DOI":"10.1350\/ijps.2009.11.2.123","volume":"11","author":"D Carson","year":"2009","unstructured":"Carson, D.: The abduction of sherlock holmes. Int. J. Police Sci. Manage. 11(2), 193\u2013202 (2009)","journal-title":"Int. J. Police Sci. Manage."},{"key":"32_CR9","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"104","DOI":"10.1007\/978-3-030-58577-8_7","volume-title":"Computer Vision \u2013 ECCV 2020","author":"Y-C Chen","year":"2020","unstructured":"Chen, Y.-C., Li, L., Yu, L., El Kholy, A., Ahmed, F., Gan, Z., Cheng, Yu., Liu, J.: UNITER: UNiversal image-TExt representation learning. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12375, pp. 104\u2013120. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58577-8_7"},{"key":"32_CR10","unstructured":"Dosovitskiy, A., et al.: An image is worth 16 $$\\times $$ 16 words: transformers for image recognition at scale. In: ICLR (2021)"},{"key":"32_CR11","doi-asserted-by":"crossref","unstructured":"Du, L., Ding, X., Liu, T., Qin, B.: Learning event graph knowledge for abductive reasoning. In: ACL (2021)","DOI":"10.18653\/v1\/2021.acl-long.403"},{"key":"32_CR12","doi-asserted-by":"crossref","unstructured":"Fang, Z., Gokhale, T., Banerjee, P., Baral, C., Yang, Y.: Video2Commonsense: generating commonsense descriptions to enrich video captioning. In: EMNLP (2020)","DOI":"10.18653\/v1\/2020.emnlp-main.61"},{"key":"32_CR13","doi-asserted-by":"crossref","unstructured":"Garcia, N., Otani, M., Chu, C., Nakashima, Y.: KnowIT vqa: answering knowledge-based questions about videos. In: AAAI (2020)","DOI":"10.1609\/aaai.v34i07.6713"},{"issue":"12","key":"32_CR14","doi-asserted-by":"publisher","first-page":"86","DOI":"10.1145\/3458723","volume":"64","author":"T Gebru","year":"2021","unstructured":"Gebru, T., et al.: Datasheets for datasets. Commun. ACM 64(12), 86\u201392 (2021)","journal-title":"Commun. ACM"},{"key":"32_CR15","doi-asserted-by":"crossref","unstructured":"Grice, H.P.: Logic and conversation. In: Speech Acts, pp. 41\u201358. Brill (1975)","DOI":"10.1163\/9789004368811_003"},{"key":"32_CR16","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.90"},{"issue":"1\u20132","key":"32_CR17","doi-asserted-by":"publisher","first-page":"69","DOI":"10.1016\/0004-3702(93)90015-4","volume":"63","author":"JR Hobbs","year":"1993","unstructured":"Hobbs, J.R., Stickel, M.E., Appelt, D.E., Martin, P.: Interpretation as abduction. Artif. Intell. 63(1\u20132), 69\u2013142 (1993)","journal-title":"Artif. Intell."},{"key":"32_CR18","unstructured":"Hosseini, H., Kannan, S., Zhang, B., Poovendran, R.: Deceiving google\u2019s perspective api built for detecting toxic comments. arXiv preprint arXiv:1702.08138 (2017)"},{"key":"32_CR19","doi-asserted-by":"crossref","unstructured":"Ignat, O., Castro, S., Miao, H., Li, W., Mihalcea, R.: WhyAct: identifying action reasons in lifestyle vlogs. In: EMNLP (2021)","DOI":"10.18653\/v1\/2021.emnlp-main.392"},{"key":"32_CR20","doi-asserted-by":"crossref","unstructured":"Jang, Y., Song, Y., Yu, Y., Kim, Y., Kim, G.: Tgif-QA: toward spatio-temporal reasoning in visual question answering. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.149"},{"key":"32_CR21","doi-asserted-by":"crossref","unstructured":"Johnson, J., Hariharan, B., Van Der Maaten, L., Fei-Fei, L., Lawrence Zitnick, C., Girshick, R.: Clevr: a diagnostic dataset for compositional language and elementary visual reasoning. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.215"},{"key":"32_CR22","doi-asserted-by":"crossref","unstructured":"Johnson, J., Karpathy, A., Fei-Fei, L.: Densecap: fully convolutional localization networks for dense captioning. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.494"},{"key":"32_CR23","doi-asserted-by":"crossref","unstructured":"Johnson, J., et al.: Image retrieval using scene graphs. In: CVPR (2015)","DOI":"10.1109\/CVPR.2015.7298990"},{"issue":"4","key":"32_CR24","doi-asserted-by":"publisher","first-page":"325","DOI":"10.1007\/BF02278710","volume":"38","author":"R Jonker","year":"1987","unstructured":"Jonker, R., Volgenant, A.: A shortest augmenting path algorithm for dense and sparse linear assignment problems. Computing 38(4), 325\u2013340 (1987). https:\/\/doi.org\/10.1007\/BF02278710","journal-title":"Computing"},{"key":"32_CR25","doi-asserted-by":"crossref","unstructured":"Kazemzadeh, S., Ordonez, V., Matten, M., Berg, T.: ReferItGame: referring to objects in photographs of natural scenes. In: EMNLP (2014)","DOI":"10.3115\/v1\/D14-1086"},{"key":"32_CR26","doi-asserted-by":"crossref","unstructured":"Kim, H., Zala, A., Bansal, M.: CoSIm: commonsense reasoning for counterfactual scene imagination. In: NAACL (2022)","DOI":"10.18653\/v1\/2022.naacl-main.66"},{"key":"32_CR27","unstructured":"Kingma, D.P., Ba, J.: Adam: a method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)"},{"issue":"1","key":"32_CR28","doi-asserted-by":"publisher","first-page":"173","DOI":"10.1162\/COLI_a_00088","volume":"38","author":"E Krahmer","year":"2012","unstructured":"Krahmer, E., Van Deemter, K.: Computational generation of referring expressions: a survey. Comput. Linguist. 38(1), 173\u2013218 (2012)","journal-title":"Comput. Linguist."},{"key":"32_CR29","doi-asserted-by":"publisher","unstructured":"Krishna, R., et al.: Visual genome: connecting language and vision using crowdsourced dense image annotations. In: IJCV (2016). https:\/\/doi.org\/10.1007\/S11263-016-0981-7","DOI":"10.1007\/S11263-016-0981-7"},{"issue":"1\u20132","key":"32_CR30","doi-asserted-by":"publisher","first-page":"83","DOI":"10.1002\/nav.3800020109","volume":"2","author":"HW Kuhn","year":"1955","unstructured":"Kuhn, H.W.: The hungarian method for the assignment problem. Naval Res. Logistics Q. 2(1\u20132), 83\u201397 (1955)","journal-title":"Naval Res. Logistics Q."},{"key":"32_CR31","doi-asserted-by":"crossref","unstructured":"Lei, J., Yu, L., Berg, T.L., Bansal, M.: TVQA+: spatio-temporal grounding for video question answering. In: ACL (2020)","DOI":"10.18653\/v1\/2020.acl-main.730"},{"key":"32_CR32","doi-asserted-by":"crossref","unstructured":"Lei, J., Yu, L., Berg, T.L., Bansal, M.: What is more likely to happen next? video-and-language future event prediction. In: EMNLP (2020)","DOI":"10.18653\/v1\/2020.emnlp-main.706"},{"key":"32_CR33","doi-asserted-by":"crossref","unstructured":"Lin, T.Y., et al.: Microsoft COCO: common objects in context. In: ECCV (2014)","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"32_CR34","doi-asserted-by":"crossref","unstructured":"Liu, J., et al.: Violin: a large-scale dataset for video-and-language inference. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.01091"},{"key":"32_CR35","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. In: ICLR (2019)"},{"key":"32_CR36","doi-asserted-by":"crossref","unstructured":"Marino, K., Rastegari, M., Farhadi, A., Mottaghi, R.: OK-VQA: a visual question answering benchmark requiring external knowledge. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00331"},{"key":"32_CR37","doi-asserted-by":"crossref","unstructured":"Mishra, A., Shekhar, S., Singh, A.K., Chakraborty, A.: OCR-VQA: visual question answering by reading text in images. In: ICDAR (2019)","DOI":"10.1109\/ICDAR.2019.00156"},{"key":"32_CR38","doi-asserted-by":"crossref","unstructured":"Mitchell, M., et al.: Model cards for model reporting. In: FAccT (2019)","DOI":"10.1145\/3287560.3287596"},{"key":"32_CR39","doi-asserted-by":"publisher","first-page":"S436","DOI":"10.1086\/392744","volume":"66","author":"I Niiniluoto","year":"1999","unstructured":"Niiniluoto, I.: Defending abduction. Philos. Sci. 66, S436\u2013S451 (1999)","journal-title":"Philos. Sci."},{"key":"32_CR40","unstructured":"Oord, A.V.D., Li, Y., Vinyals, O.: Representation learning with contrastive predictive coding. arXiv preprint arXiv:1807.03748 (2018)"},{"key":"32_CR41","unstructured":"Ordonez, V., Kulkarni, G., Berg, T.L.: Im2text: describing images using 1 million captioned photographs. In: NeurIPS (2011)"},{"key":"32_CR42","unstructured":"Ovchinnikova, E., Montazeri, N., Alexandrov, T., Hobbs, J.R., McCord, M.C., Mulkar-Mehta, R.: Abductive reasoning with a large knowledge base for discourse processing. In: IWCS (2011)"},{"key":"32_CR43","doi-asserted-by":"crossref","unstructured":"Park, D.H., Darrell, T., Rohrbach, A.: Robust change captioning. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00472"},{"key":"32_CR44","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"508","DOI":"10.1007\/978-3-030-58558-7_30","volume-title":"Computer Vision \u2013 ECCV 2020","author":"JS Park","year":"2020","unstructured":"Park, J.S., Bhagavatula, C., Mottaghi, R., Farhadi, A., Choi, Y.: VisualCOMET: reasoning about the dynamic context of a still image. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12350, pp. 508\u2013524. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58558-7_30"},{"key":"32_CR45","doi-asserted-by":"crossref","unstructured":"Paul, D., Frank, A.: Generating hypothetical events for abductive inference. In: *SEM (2021)","DOI":"10.18653\/v1\/2021.starsem-1.6"},{"key":"32_CR46","unstructured":"Peirce, C.S.: Philosophical Writings of Peirce, vol. 217. Courier Corporation (1955)"},{"key":"32_CR47","unstructured":"Peirce, C.S.: Pragmatism and Pragmaticism, vol. 5. Belknap Press of Harvard University Press (1965)"},{"key":"32_CR48","doi-asserted-by":"crossref","unstructured":"Pezzelle, S., Greco, C., Gandolfi, G., Gualdoni, E., Bernardi, R.: Be different to be better! a benchmark to leverage the complementarity of language and vision. In: Findings of EMNLP (2020)","DOI":"10.18653\/v1\/2020.findings-emnlp.248"},{"key":"32_CR49","doi-asserted-by":"crossref","unstructured":"Pirsiavash, H., Vondrick, C., Torralba, A.: Inferring the why in images. Technical report (2014)","DOI":"10.21236\/ADA612444"},{"key":"32_CR50","doi-asserted-by":"crossref","unstructured":"Qin, L., et al.: Back to the future: unsupervised backprop-based decoding for counterfactual and abductive commonsense reasoning. In: EMNLP (2020)","DOI":"10.18653\/v1\/2020.emnlp-main.58"},{"key":"32_CR51","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. arXiv preprint arXiv:2103.00020 (2021)"},{"key":"32_CR52","unstructured":"Raffel, C., et al.: Exploring the limits of transfer learning with a unified text-to-text transformer. In: JMLR (2020)"},{"key":"32_CR53","doi-asserted-by":"crossref","unstructured":"Reimers, N., Gurevych, I.: Sentence-bert: sentence embeddings using siamese bert-networks. In: EMNLP (2019)","DOI":"10.18653\/v1\/D19-1410"},{"key":"32_CR54","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster R-CNN: towards real-time object detection with region proposal networks. In: NeurIPS (2015)"},{"key":"32_CR55","doi-asserted-by":"crossref","unstructured":"Sap, M., Card, D., Gabriel, S., Choi, Y., Smith, N.A.: The risk of racial bias in hate speech detection. In: ACL (2019)","DOI":"10.18653\/v1\/P19-1163"},{"issue":"6","key":"32_CR56","doi-asserted-by":"publisher","first-page":"841","DOI":"10.1177\/0959354398086007","volume":"8","author":"G Shank","year":"1998","unstructured":"Shank, G.: The extraordinary ordinary powers of abductive reasoning. Theor. Psychol. 8(6), 841\u2013860 (1998)","journal-title":"Theor. Psychol."},{"key":"32_CR57","doi-asserted-by":"crossref","unstructured":"Sharma, P., Ding, N., Goodman, S., Soricut, R.: Conceptual captions: a cleaned, hypernymed, image alt-text dataset for automatic image captioning. In: ACL (2018)","DOI":"10.18653\/v1\/P18-1238"},{"key":"32_CR58","unstructured":"Shazeer, N., Stern, M.: Adafactor: adaptive learning rates with sublinear memory cost. In: ICML (2018)"},{"key":"32_CR59","unstructured":"Sohn, K.: Improved deep metric learning with multi-class n-pair loss objective. In: NeurIPS (2016)"},{"key":"32_CR60","doi-asserted-by":"crossref","unstructured":"Tafjord, O., Mishra, B.D., Clark, P.: ProofWriter: generating implications, proofs, and abductive statements over natural language. In: Findings of ACL (2021)","DOI":"10.18653\/v1\/2021.findings-acl.317"},{"key":"32_CR61","doi-asserted-by":"crossref","unstructured":"Tan, H., Bansal, M.: LXMERT: learning cross-modality encoder representations from transformers. In: EMNLP (2019)","DOI":"10.18653\/v1\/D19-1514"},{"key":"32_CR62","unstructured":"Tan, M., Le, Q.: Efficientnet: rethinking model scaling for convolutional neural networks. In: ICML (2019)"},{"key":"32_CR63","doi-asserted-by":"crossref","unstructured":"Tapaswi, M., Zhu, Y., Stiefelhagen, R., Torralba, A., Urtasun, R., Fidler, S.: MovieQA: understanding stories in movies through question-answering. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.501"},{"key":"32_CR64","unstructured":"Vaswani, A., et al.: Attention is all you need. In: NeurIPS (2017)"},{"key":"32_CR65","doi-asserted-by":"crossref","unstructured":"Vedantam, R., Lin, X., Batra, T., Zitnick, C.L., Parikh, D.: Learning common sense through visual abstraction. In: ICCV (2015)","DOI":"10.1109\/ICCV.2015.292"},{"issue":"10","key":"32_CR66","doi-asserted-by":"publisher","first-page":"2413","DOI":"10.1109\/TPAMI.2017.2754246","volume":"40","author":"P Wang","year":"2017","unstructured":"Wang, P., Wu, Q., Shen, C., Dick, A., Van Den Hengel, A.: FVQA: fact-based visual question answering. TPAMI 40(10), 2413\u20132427 (2017)","journal-title":"TPAMI"},{"key":"32_CR67","doi-asserted-by":"crossref","unstructured":"Wang, P., Wu, Q., Shen, C., Hengel, A.V.D., Dick, A.: Explicit knowledge-based reasoning for visual question answering. In: IJCAI (2017)","DOI":"10.24963\/ijcai.2017\/179"},{"key":"32_CR68","doi-asserted-by":"crossref","unstructured":"Wolf, T., et al.: Transformers: state-of-the-art natural language processing. In: EMNLP: System Demonstrations (2020)","DOI":"10.18653\/v1\/2020.emnlp-demos.6"},{"key":"32_CR69","doi-asserted-by":"crossref","unstructured":"Xie, S., Girshick, R., Doll\u00e1r, P., Tu, Z., He, K.: Aggregated residual transformations for deep neural networks. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.634"},{"key":"32_CR70","doi-asserted-by":"crossref","unstructured":"Yao, Y., Zhang, A., Zhang, Z., Liu, Z., Chua, T.S., Sun, M.: CPT: colorful prompt tuning for pre-trained vision-language models. arXiv preprint arXiv:2109.11797 (2021)","DOI":"10.18653\/v1\/2022.findings-acl.273"},{"key":"32_CR71","unstructured":"Yi, K., et al.: CLEVRER: collision events for video representation and reasoning. In: ICLR (2020)"},{"key":"32_CR72","doi-asserted-by":"crossref","unstructured":"Yu, L., Park, E., Berg, A.C., Berg, T.L.: Visual Madlibs: fill in the blank image generation and question answering. In: ICCV (2015)","DOI":"10.1109\/ICCV.2015.283"},{"key":"32_CR73","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"69","DOI":"10.1007\/978-3-319-46475-6_5","volume-title":"Computer Vision \u2013 ECCV 2016","author":"L Yu","year":"2016","unstructured":"Yu, L., Poirson, P., Yang, S., Berg, A.C., Berg, T.L.: Modeling context in referring expressions. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9906, pp. 69\u201385. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46475-6_5"},{"key":"32_CR74","doi-asserted-by":"crossref","unstructured":"Zadeh, A., Chan, M., Liang, P.P., Tong, E., Morency, L.P.: Social-iq: a question answering benchmark for artificial social intelligence. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00901"},{"key":"32_CR75","doi-asserted-by":"crossref","unstructured":"Zellers, R., Bisk, Y., Farhadi, A., Choi, Y.: From recognition to cognition: Visual commonsense reasoning. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00688"},{"key":"32_CR76","unstructured":"Zellers, R., et al.: MERLOT: multimodal neural script knowledge models. In: NeurIPS (2021)"},{"key":"32_CR77","doi-asserted-by":"crossref","unstructured":"Zhang, C., Gao, F., Jia, B., Zhu, Y., Zhu, S.C.: Raven: a dataset for relational and analogical visual reasoning. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00546"},{"key":"32_CR78","doi-asserted-by":"crossref","unstructured":"Zhang, H., Huo, Y., Zhao, X., Song, Y., Roth, D.: Learning contextual causality from time-consecutive images. In: CVPR Workshops (2021)","DOI":"10.1109\/CVPRW53098.2021.00193"},{"key":"32_CR79","doi-asserted-by":"crossref","unstructured":"Zhu, Y., Groth, O., Bernstein, M., Fei-Fei, L.: Visual7W: grounded question answering in images. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.540"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2022"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-20059-5_32","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,10,28]],"date-time":"2022-10-28T16:13:11Z","timestamp":1666973591000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-20059-5_32"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"ISBN":["9783031200588","9783031200595"],"references-count":79,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-20059-5_32","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022]]},"assertion":[{"value":"29 October 2022","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Tel Aviv","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Israel","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2022","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 October 2022","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 October 2022","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2022","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2022.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"CMT","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"5804","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"1645","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"28% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.21","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.91","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}