{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T17:41:59Z","timestamp":1784310119213,"version":"3.55.0"},"publisher-location":"Cham","reference-count":29,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031164484","type":"print"},{"value":"9783031164491","type":"electronic"}],"license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022]]},"DOI":"10.1007\/978-3-031-16449-1_4","type":"book-chapter","created":{"date-parts":[[2022,9,16]],"date-time":"2022-09-16T08:04:54Z","timestamp":1663315494000},"page":"33-43","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":41,"title":["Surgical-VQA: Visual Question Answering in\u00a0Surgical Scenes Using Transformer"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0103-1234","authenticated-orcid":false,"given":"Lalithkumar","family":"Seenivasan","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7162-2822","authenticated-orcid":false,"given":"Mobarakol","family":"Islam","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2284-703X","authenticated-orcid":false,"given":"Adithya K","family":"Krishna","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6488-1551","authenticated-orcid":false,"given":"Hongliang","family":"Ren","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2022,9,17]]},"reference":[{"key":"4_CR1","unstructured":"Abacha, A.B., Hasan, S.A., Datla, V.V., Liu, J., Demner-Fushman, D., M\u00fcller, H.: VQA-med: overview of the medical visual question answering task at imageclef 2019. clef2019 working notes. In: CEUR Workshop Proceedings, pp. 9\u201312. CEUR-WS.org $$<$$http:\/\/ceur-ws.org$$>$$. September"},{"issue":"3","key":"4_CR2","doi-asserted-by":"publisher","first-page":"43","DOI":"10.1109\/38.55152","volume":"10","author":"L Adams","year":"1990","unstructured":"Adams, L., et al.: Computer-assisted surgery. IEEE Comput. Graph. Appl. 10(3), 43\u201351 (1990)","journal-title":"IEEE Comput. Graph. Appl."},{"key":"4_CR3","unstructured":"Allan, M., et al.: 2018 robotic scene segmentation challenge. arXiv preprint arXiv:2001.11190 (2020)"},{"key":"4_CR4","unstructured":"Banerjee, S., Lavie, A.: Meteor: an automatic metric for MT evaluation with improved correlation with human judgments. In: Proceedings of the ACL Workshop on Intrinsic and Extrinsic Evaluation Measures for Machine Translation and\/or Summarization, pp. 65\u201372 (2005)"},{"key":"4_CR5","doi-asserted-by":"publisher","first-page":"325","DOI":"10.1016\/j.patrec.2021.09.008","volume":"151","author":"S Barra","year":"2021","unstructured":"Barra, S., Bisogni, C., De Marsico, M., Ricciardi, S.: Visual question answering: which investigated applications? Pattern Recogn. Lett. 151, 325\u2013331 (2021)","journal-title":"Pattern Recogn. Lett."},{"key":"4_CR6","doi-asserted-by":"crossref","unstructured":"Bates, D.W., Gawande, A.A.: Error in medicine: what have we learned? (2000)","DOI":"10.1007\/978-1-349-15068-7_16"},{"key":"4_CR7","doi-asserted-by":"crossref","unstructured":"Cornia, M., Stefanini, M., Baraldi, L., Cucchiara, R.: Meshed-memory transformer for image captioning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10578\u201310587 (2020)","DOI":"10.1109\/CVPR42600.2020.01059"},{"key":"4_CR8","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.J., Li, K., Fei-Fei, L.: Imagenet: a large-scale hierarchical image database. In: 2009 IEEE Conference on Computer Vision and Pattern Recognition, pp. 248\u2013255. IEEE (2009)","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"4_CR9","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: Bert: pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)"},{"key":"4_CR10","doi-asserted-by":"crossref","unstructured":"Hara, K., Kataoka, H., Satoh, Y.: Learning spatio-temporal features with 3D residual networks for action recognition. In: Proceedings of the IEEE International Conference on Computer Vision Workshops, pp. 3154\u20133160 (2017)","DOI":"10.1109\/ICCVW.2017.373"},{"key":"4_CR11","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"4_CR12","unstructured":"Hendrycks, D., Gimpel, K.: Gaussian error linear units (gelus). arXiv preprint arXiv:1606.08415 (2016)"},{"key":"4_CR13","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"627","DOI":"10.1007\/978-3-030-59716-0_60","volume-title":"Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2020","author":"M Islam","year":"2020","unstructured":"Islam, M., Seenivasan, L., Ming, L.C., Ren, H.: Learning and reasoning with the graph structure representation in robotic surgery. In: Martel, A.L., et al. (eds.) MICCAI 2020. LNCS, vol. 12263, pp. 627\u2013636. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-59716-0_60"},{"issue":"3","key":"4_CR14","doi-asserted-by":"publisher","first-page":"267","DOI":"10.1046\/j.1365-2923.2003.01440.x","volume":"37","author":"R Kneebone","year":"2003","unstructured":"Kneebone, R.: Simulation in surgical training: educational issues and practical implications. Med. Educ. 37(3), 267\u2013277 (2003)","journal-title":"Med. Educ."},{"key":"4_CR15","unstructured":"Li, L.H., Yatskar, M., Yin, D., Hsieh, C.J., Chang, K.W.: Visualbert: a simple and performant baseline for vision and language. arXiv preprint arXiv:1908.03557 (2019)"},{"key":"4_CR16","doi-asserted-by":"crossref","unstructured":"Papineni, K., Roukos, S., Ward, T., Zhu, W.J.: Bleu: a method for automatic evaluation of machine translation. In: Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics. pp. 311\u2013318 (2002)","DOI":"10.3115\/1073083.1073135"},{"issue":"6","key":"4_CR17","doi-asserted-by":"publisher","first-page":"508","DOI":"10.1016\/S0002-9610(98)00087-7","volume":"175","author":"DA Rogers","year":"1998","unstructured":"Rogers, D.A., Yeh, K.A., Howdieshell, T.R.: Computer-assisted learning versus a lecture and feedback seminar for teaching a basic surgical technical skill. Am. J. Surg. 175(6), 508\u2013510 (1998)","journal-title":"Am. J. Surg."},{"issue":"12","key":"4_CR18","doi-asserted-by":"publisher","first-page":"2120","DOI":"10.1111\/j.1742-1241.2007.01435.x","volume":"61","author":"S Sarker","year":"2007","unstructured":"Sarker, S., Patel, B.: Simulation and surgical training. Int. J. Clin. Pract. 61(12), 2120\u20132125 (2007)","journal-title":"Int. J. Clin. Pract."},{"key":"4_CR19","doi-asserted-by":"publisher","first-page":"3858","DOI":"10.1109\/LRA.2022.3146544","volume":"7","author":"L Seenivasan","year":"2022","unstructured":"Seenivasan, L., Mitheran, S., Islam, M., Ren, H.: Global-reasoned multi-task learning model for surgical scene understanding. IEEE Robot. Autom. Lett. 7, 3858\u20133865 (2022)","journal-title":"IEEE Robot. Autom. Lett."},{"issue":"1","key":"4_CR20","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1038\/s41598-021-98390-1","volume":"11","author":"D Sharma","year":"2021","unstructured":"Sharma, D., Purushotham, S., Reddy, C.K.: Medfusenet: an attention-based multimodal deep learning model for visual question answering in the medical domain. Sci. Rep. 11(1), 1\u201318 (2021)","journal-title":"Sci. Rep."},{"key":"4_CR21","doi-asserted-by":"crossref","unstructured":"Sharma, H., Jalal, A.S.: Image captioning improved visual question answering. Multimedia Tools Appl. 1\u201322 (2021)","DOI":"10.1007\/s11042-021-11276-2"},{"key":"4_CR22","doi-asserted-by":"publisher","DOI":"10.1016\/j.imavis.2021.104165","volume":"110","author":"H Sharma","year":"2021","unstructured":"Sharma, H., Jalal, A.S.: Visual question answering model based on graph neural network and contextual attention. Image Vis. Comput. 110, 104165 (2021)","journal-title":"Image Vis. Comput."},{"key":"4_CR23","unstructured":"Sheng, S., et al.: Human-adversarial visual question answering. In: Advances in Neural Information Processing Systems, vol. 34 (2021)"},{"key":"4_CR24","doi-asserted-by":"crossref","unstructured":"Touvron, H., et al.: Resmlp: feedforward networks for image classification with data-efficient training. arXiv preprint arXiv:2105.03404 (2021)","DOI":"10.1109\/TPAMI.2022.3206148"},{"issue":"1","key":"4_CR25","doi-asserted-by":"publisher","first-page":"86","DOI":"10.1109\/TMI.2016.2593957","volume":"36","author":"AP Twinanda","year":"2016","unstructured":"Twinanda, A.P., Shehata, S., Mutter, D., Marescaux, J., De Mathelin, M., Padoy, N.: Endonet: a deep architecture for recognition tasks on laparoscopic videos. IEEE Trans. Med. Imaging 36(1), 86\u201397 (2016)","journal-title":"IEEE Trans. Med. Imaging"},{"key":"4_CR26","doi-asserted-by":"crossref","unstructured":"Vedantam, R., Lawrence Zitnick, C., Parikh, D.: Cider: consensus-based image description evaluation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4566\u20134575 (2015)","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"4_CR27","unstructured":"Wang, Z., Yu, J., Yu, A.W., Dai, Z., Tsvetkov, Y., Cao, Y.: SIMVLM: simple visual language model pretraining with weak supervision. arXiv preprint arXiv:2108.10904 (2021)"},{"key":"4_CR28","doi-asserted-by":"crossref","unstructured":"Wiseman, S., Rush, A.M.: Sequence-to-sequence learning as beam-search optimization. arXiv preprint arXiv:1606.02960 (2016)","DOI":"10.18653\/v1\/D16-1137"},{"key":"4_CR29","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.inffus.2021.02.022","volume":"73","author":"S Zhang","year":"2021","unstructured":"Zhang, S., Chen, M., Chen, J., Zou, F., Li, Y.F., Lu, P.: Multimodal feature-wise co-attention method for visual question answering. Inf. Fusion 73, 1\u201310 (2021)","journal-title":"Inf. Fusion"}],"container-title":["Lecture Notes in Computer Science","Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2022"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-16449-1_4","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,7]],"date-time":"2024-03-07T16:52:37Z","timestamp":1709830357000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-16449-1_4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"ISBN":["9783031164484","9783031164491"],"references-count":29,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-16449-1_4","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022]]},"assertion":[{"value":"17 September 2022","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"MICCAI","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Medical Image Computing and Computer-Assisted Intervention","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Singapore","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Singapore","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2022","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18 September 2022","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 September 2022","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"25","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"miccai2022","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Microsoft Conference","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"1831","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"574","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"31% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"5","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}