{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,5]],"date-time":"2025-11-05T11:25:09Z","timestamp":1762341909781,"version":"3.40.3"},"publisher-location":"Cham","reference-count":24,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031159336"},{"type":"electronic","value":"9783031159343"}],"license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022]]},"DOI":"10.1007\/978-3-031-15934-3_13","type":"book-chapter","created":{"date-parts":[[2022,9,6]],"date-time":"2022-09-06T00:02:53Z","timestamp":1662422573000},"page":"149-161","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Inter-subtask Consistent Representation Learning for\u00a0Visual Commonsense Reasoning"],"prefix":"10.1007","author":[{"given":"Kexin","family":"Liu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shaojuan","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaowang","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Song","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,9,15]]},"reference":[{"doi-asserted-by":"crossref","unstructured":"Anderson, P., et al.: Bottom-up and top-down attention for image captioning and visual question answering. In: CVPR, pp. 6077\u20136086 (2018)","key":"13_CR1","DOI":"10.1109\/CVPR.2018.00636"},{"doi-asserted-by":"crossref","unstructured":"Ben-Younes, H., Cadene, R., Cord, M., Thome, N.: Mutan: multimodal tucker fusion for visual question answering. In: ICCV, pp. 2612\u20132620 (2017)","key":"13_CR2","DOI":"10.1109\/ICCV.2017.285"},{"doi-asserted-by":"crossref","unstructured":"Bromley, J., et al.: Signature verification using a \u201csiamese\u201d time delay neural network. IJPRAI 7(04), 669\u2013688 (1993)","key":"13_CR3","DOI":"10.1142\/S0218001493000339"},{"doi-asserted-by":"crossref","unstructured":"Chen, Q., Zhu, X., Ling, Z., Wei, S., Jiang, H., Inkpen, D.: Enhanced lstm for natural language inference. In: ACL, pp. 1657\u20131668 (2017)","key":"13_CR4","DOI":"10.18653\/v1\/P17-1152"},{"key":"13_CR5","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"104","DOI":"10.1007\/978-3-030-58577-8_7","volume-title":"Computer Vision \u2013 ECCV 2020","author":"Y-C Chen","year":"2020","unstructured":"Chen, Y.-C., et al.: UNITER: UNiversal image-TExt representation learning. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12375, pp. 104\u2013120. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58577-8_7"},{"unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: Bert: Pre-training of deep bidirectional transformers for language understanding. In: NAACL, pp. 4171\u20134186 (2018)","key":"13_CR6"},{"doi-asserted-by":"crossref","unstructured":"Hadsell, R., Chopra, S., LeCun, Y.: Dimensionality reduction by learning an invariant mapping. In: CVPR, vol. 2, pp. 1735\u20131742 (2006)","key":"13_CR7","DOI":"10.1109\/CVPR.2006.100"},{"doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: CVPR, pp. 770\u2013778 (2016)","key":"13_CR8","DOI":"10.1109\/CVPR.2016.90"},{"key":"13_CR9","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"727","DOI":"10.1007\/978-3-319-46484-8_44","volume-title":"Computer Vision \u2013 ECCV 2016","author":"A Jabri","year":"2016","unstructured":"Jabri, A., Joulin, A., van der Maaten, L.: Revisiting visual question answering baselines. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9912, pp. 727\u2013739. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46484-8_44"},{"unstructured":"Kim, J.H., On, K.W., Lim, W., Kim, J., Ha, J.W., Zhang, B.T.: Hadamard product for low-rank bilinear pooling. In: ICLR (2017)","key":"13_CR10"},{"unstructured":"Kingma, D.P., Ba, J.: Adam: A method for stochastic optimization. In: ICLR (2015)","key":"13_CR11"},{"doi-asserted-by":"crossref","unstructured":"Lee, J., Kim, I.: Vision-language-knowledge co-embedding for visual commonsense reasoning. Sensors 21(9), 2911 (2021)","key":"13_CR12","DOI":"10.3390\/s21092911"},{"unstructured":"Lin, J., Jain, U., Schwing, A.: Tab-vcr: tags and attributes based vcr baselines. In: NIPS, pp. 15589\u201315602 (2019)","key":"13_CR13"},{"issue":"6","key":"13_CR14","doi-asserted-by":"publisher","first-page":"1137","DOI":"10.1109\/TPAMI.2016.2577031","volume":"39","author":"S Ren","year":"2016","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster r-cnn: towards real-time object detection with region proposal networks. TPAMI 39(6), 1137\u20131149 (2016)","journal-title":"TPAMI"},{"doi-asserted-by":"crossref","unstructured":"Song, D., Ma, S., Sun, Z., Yang, S., Liao, L.: Kvl-bert: knowledge enhanced visual-and-linguistic bert for visual commonsense reasoning. Knowl.-Based Syst. 230, 107408 (2021)","key":"13_CR15","DOI":"10.1016\/j.knosys.2021.107408"},{"doi-asserted-by":"crossref","unstructured":"Wang, T., Huang, J., Zhang, H., Sun, Q.: Visual commonsense r-cnn. In: CVPR, pp. 10760\u201310770 (2020)","key":"13_CR16","DOI":"10.1109\/CVPR42600.2020.01077"},{"doi-asserted-by":"crossref","unstructured":"Wang, Z., et al.: Sgeitl: scene graph enhanced image-text learning for visual commonsense reasoning. arXiv (2021)","key":"13_CR17","DOI":"10.1609\/aaai.v36i5.20536"},{"issue":"3","key":"13_CR18","first-page":"1042","volume":"31","author":"Z Wen","year":"2020","unstructured":"Wen, Z., Peng, Y.: Multi-level knowledge injecting for visual commonsense reasoning. TCSVT 31(3), 1042\u20131054 (2020)","journal-title":"TCSVT"},{"unstructured":"Wu, A., Zhu, L., Han, Y., Yang, Y.: Connective cognition network for directional visual commonsense reasoning. In: NIPS, pp. 5669\u20135679 (2019)","key":"13_CR19"},{"doi-asserted-by":"crossref","unstructured":"Ye, K., Kovashka, A.: A case study of the shortcut effects in visual commonsense reasoning. In: AAAI, pp. 3181\u20133189 (2021)","key":"13_CR20","DOI":"10.1609\/aaai.v35i4.16428"},{"unstructured":"Yu, W., Zhou, J., Yu, W., Liang, X., Xiao, N.: Heterogeneous graph learning for visual commonsense reasoning. In: NIPS, pp. 2765\u20132775 (2019)","key":"13_CR21"},{"doi-asserted-by":"crossref","unstructured":"Zamir, A.R., et al.: Robust learning through cross-task consistency. In: CVPR, pp. 11197\u201311206 (2020)","key":"13_CR22","DOI":"10.1109\/CVPR42600.2020.01121"},{"doi-asserted-by":"crossref","unstructured":"Zellers, R., Bisk, Y., Farhadi, A., Choi, Y.: From recognition to cognition: visual commonsense reasoning. In: CVPR, pp. 6720\u20136731 (2019)","key":"13_CR23","DOI":"10.1109\/CVPR.2019.00688"},{"doi-asserted-by":"crossref","unstructured":"Zhang, X., Zhang, F., Xu, C.: Multi-level counterfactual contrast for visual commonsense reasoning. In: MM, pp. 1793\u20131802 (2021)","key":"13_CR24","DOI":"10.1145\/3474085.3475328"}],"container-title":["Lecture Notes in Computer Science","Artificial Neural Networks and Machine Learning \u2013 ICANN 2022"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-15934-3_13","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,3]],"date-time":"2024-10-03T07:39:00Z","timestamp":1727941140000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-15934-3_13"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"ISBN":["9783031159336","9783031159343"],"references-count":24,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-15934-3_13","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2022]]},"assertion":[{"value":"15 September 2022","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICANN","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Artificial Neural Networks","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Bristol","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"United Kingdom","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2022","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"6 September 2022","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"9 September 2022","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"31","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icann2022","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/e-nns.org\/icann2022\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Single-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"EasyChair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"561","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"255","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"4","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"45% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"No","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}