{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,5]],"date-time":"2026-03-05T15:28:35Z","timestamp":1772724515193,"version":"3.50.1"},"publisher-location":"Cham","reference-count":28,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031439957","type":"print"},{"value":"9783031439964","type":"electronic"}],"license":[{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023]]},"DOI":"10.1007\/978-3-031-43996-4_37","type":"book-chapter","created":{"date-parts":[[2023,9,30]],"date-time":"2023-09-30T23:07:48Z","timestamp":1696115268000},"page":"386-396","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":7,"title":["$$\\textsf{GLSFormer}$$: Gated - Long, Short Sequence Transformer for\u00a0Step Recognition in\u00a0Surgical Videos"],"prefix":"10.1007","author":[{"given":"Nisarg A.","family":"Shah","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shameema","family":"Sikder","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"S. Swaroop","family":"Vedula","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Vishal M.","family":"Patel","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,10,1]]},"reference":[{"key":"37_CR1","doi-asserted-by":"crossref","unstructured":"Arnab, A., Dehghani, M., Heigold, G., Sun, C., Lu\u010di\u0107, M., Schmid, C.: Vivit: a video vision transformer. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6836\u20136846 (2021)","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"37_CR2","unstructured":"Bertasius, G., Wang, H., Torresani, L.: Is space-time attention all you need for video understanding? In: ICML, vol. 2, p. 4 (2021)"},{"key":"37_CR3","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"400","DOI":"10.1007\/978-3-642-15711-0_50","volume-title":"Medical Image Computing and Computer-Assisted Intervention \u2013 MICCAI 2010","author":"T Blum","year":"2010","unstructured":"Blum, T., Feu\u00dfner, H., Navab, N.: Modeling and segmentation of surgical workflow from laparoscopic video. In: Jiang, T., Navab, N., Pluim, J.P.W., Viergever, M.A. (eds.) MICCAI 2010. LNCS, vol. 6363, pp. 400\u2013407. Springer, Heidelberg (2010). https:\/\/doi.org\/10.1007\/978-3-642-15711-0_50"},{"key":"37_CR4","doi-asserted-by":"crossref","unstructured":"Bricon-Souf, N., Newman, C.R.: Context awareness in health care: a review. Int. J. Med. Inform. 76(1), 2\u201312 (2007)","DOI":"10.1016\/j.ijmedinf.2006.01.003"},{"key":"37_CR5","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"343","DOI":"10.1007\/978-3-030-59716-0_33","volume-title":"Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2020","author":"T Czempiel","year":"2020","unstructured":"Czempiel, T., et al.: TeCNO: surgical phase recognition with multi-stage temporal convolutional networks. In: Martel, A.L., et al. (eds.) MICCAI 2020. LNCS, vol. 12263, pp. 343\u2013352. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-59716-0_33"},{"issue":"6","key":"37_CR6","doi-asserted-by":"publisher","first-page":"1081","DOI":"10.1007\/s11548-016-1371-x","volume":"11","author":"O Dergachyova","year":"2016","unstructured":"Dergachyova, O., Bouget, D., Huaulm\u00e9, A., Morandi, X., Jannin, P.: Automatic data-driven real-time segmentation and recognition of surgical workflow. Int. J. Comput. Assist. Radiol. Surg. 11(6), 1081\u20131089 (2016). https:\/\/doi.org\/10.1007\/s11548-016-1371-x","journal-title":"Int. J. Comput. Assist. Radiol. Surg."},{"key":"37_CR7","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)"},{"key":"37_CR8","unstructured":"Dosovitskiy, A., et al.: An image is worth $$16 \\times 16$$ words: transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)"},{"key":"37_CR9","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Fan, H., Malik, J., He, K.: Slowfast networks for video recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6202\u20136211 (2019)","DOI":"10.1109\/ICCV.2019.00630"},{"key":"37_CR10","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"467","DOI":"10.1007\/978-3-030-32254-0_52","volume-title":"Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2019","author":"I Funke","year":"2019","unstructured":"Funke, I., Bodenstedt, S., Oehme, F., von Bechtolsheim, F., Weitz, J., Speidel, S.: Using 3D convolutional neural networks to learn spatiotemporal features for automatic surgical gesture recognition in video. In: Shen, D., et al. (eds.) MICCAI 2019. LNCS, vol. 11768, pp. 467\u2013475. Springer, Cham (2019). https:\/\/doi.org\/10.1007\/978-3-030-32254-0_52"},{"key":"37_CR11","doi-asserted-by":"publisher","first-page":"1217","DOI":"10.1007\/s11548-019-01995-1","volume":"14","author":"I Funke","year":"2019","unstructured":"Funke, I., Mees, S.T., Weitz, J., Speidel, S.: Video-based surgical skill assessment using 3D convolutional neural networks. Int. J. Comput. Assist. Radiol. Surg. 14, 1217\u20131225 (2019)","journal-title":"Int. J. Comput. Assist. Radiol. Surg."},{"key":"37_CR12","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"593","DOI":"10.1007\/978-3-030-87202-1_57","volume-title":"Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2021","author":"X Gao","year":"2021","unstructured":"Gao, X., Jin, Y., Long, Y., Dou, Q., Heng, P.-A.: Trans-SVNet: accurate phase recognition from surgical videos via\u00a0hybrid embedding aggregation transformer. In: de Bruijne, M., et al. (eds.) MICCAI 2021. LNCS, vol. 12904, pp. 593\u2013603. Springer, Cham (2021). https:\/\/doi.org\/10.1007\/978-3-030-87202-1_57"},{"key":"37_CR13","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"37_CR14","doi-asserted-by":"publisher","DOI":"10.1016\/j.artmed.2020.101837","volume":"104","author":"A Huaulm\u00e9","year":"2020","unstructured":"Huaulm\u00e9, A., Jannin, P., Reche, F., Faucheron, J.L., Moreau-Gaudry, A., Voros, S.: Offline identification of surgical deviations in laparoscopic rectopexy. Artif. Intell. Med. 104, 101837 (2020)","journal-title":"Artif. Intell. Med."},{"issue":"5","key":"37_CR15","doi-asserted-by":"publisher","first-page":"1114","DOI":"10.1109\/TMI.2017.2787657","volume":"37","author":"Y Jin","year":"2017","unstructured":"Jin, Y., et al.: SV-RCNet: workflow recognition from surgical videos using recurrent convolutional network. IEEE Trans. Med. Imaging 37(5), 1114\u20131126 (2017)","journal-title":"IEEE Trans. Med. Imaging"},{"issue":"7","key":"37_CR16","doi-asserted-by":"publisher","first-page":"1911","DOI":"10.1109\/TMI.2021.3069471","volume":"40","author":"Y Jin","year":"2021","unstructured":"Jin, Y., Long, Y., Chen, C., Zhao, Z., Dou, Q., Heng, P.A.: Temporal memory relation network for workflow recognition from surgical video. IEEE Trans. Med. Imaging 40(7), 1911\u20131923 (2021)","journal-title":"IEEE Trans. Med. Imaging"},{"key":"37_CR17","unstructured":"Kay, W., et al.: The kinetics human action video dataset. arXiv preprint arXiv:1705.06950 (2017)"},{"key":"37_CR18","doi-asserted-by":"publisher","first-page":"39","DOI":"10.1007\/s11548-012-0685-6","volume":"8","author":"F Lalys","year":"2013","unstructured":"Lalys, F., Bouget, D., Riffaud, L., Jannin, P.: Automatic knowledge-based recognition of low-level tasks in ophthalmological procedures. Int. J. Comput. Assist. Radiol. Surg. 8, 39\u201349 (2013)","journal-title":"Int. J. Comput. Assist. Radiol. Surg."},{"key":"37_CR19","doi-asserted-by":"crossref","unstructured":"Lea, C., Hager, G.D., Vidal, R.: An improved model for segmentation and recognition of fine-grained activities with application to surgical training tasks. In: 2015 IEEE Winter Conference on Applications of Computer Vision, pp. 1123\u20131129. IEEE (2015)","DOI":"10.1109\/WACV.2015.154"},{"issue":"4","key":"37_CR20","doi-asserted-by":"publisher","first-page":"673","DOI":"10.1007\/s11548-019-02108-8","volume":"15","author":"G Lecuyer","year":"2020","unstructured":"Lecuyer, G., Ragot, M., Martin, N., Launay, L., Jannin, P.: Assisted phase and step annotation for surgical videos. Int. J. Comput. Assist. Radiol. Surg. 15(4), 673\u2013680 (2020). https:\/\/doi.org\/10.1007\/s11548-019-02108-8","journal-title":"Int. J. Comput. Assist. Radiol. Surg."},{"issue":"2","key":"37_CR21","doi-asserted-by":"publisher","first-page":"82","DOI":"10.1080\/13645706.2019.1584116","volume":"28","author":"N Padoy","year":"2019","unstructured":"Padoy, N.: Machine and deep learning for workflow recognition during surgery. Minim. Invasive Therapy Allied Technol. 28(2), 82\u201390 (2019)","journal-title":"Minim. Invasive Therapy Allied Technol."},{"key":"37_CR22","doi-asserted-by":"crossref","unstructured":"Schoeffmann, K., Taschwer, M., Sarny, S., M\u00fcnzer, B., Primus, M.J., Putzgruber, D.: Cataract-101: video dataset of 101 cataract surgeries. In: Proceedings of the 9th ACM Multimedia Systems Conference, pp. 421\u2013425 (2018)","DOI":"10.1145\/3204949.3208137"},{"key":"37_CR23","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"339","DOI":"10.1007\/978-3-642-40760-4_43","volume-title":"Medical Image Computing and Computer-Assisted Intervention \u2013 MICCAI 2013","author":"L Tao","year":"2013","unstructured":"Tao, L., Zappella, L., Hager, G.D., Vidal, R.: Surgical gesture segmentation and recognition. In: Mori, K., Sakuma, I., Sato, Y., Barillot, C., Navab, N. (eds.) MICCAI 2013. LNCS, vol. 8151, pp. 339\u2013346. Springer, Heidelberg (2013). https:\/\/doi.org\/10.1007\/978-3-642-40760-4_43"},{"issue":"1","key":"37_CR24","doi-asserted-by":"publisher","first-page":"86","DOI":"10.1109\/TMI.2016.2593957","volume":"36","author":"AP Twinanda","year":"2016","unstructured":"Twinanda, A.P., Shehata, S., Mutter, D., Marescaux, J., De Mathelin, M., Padoy, N.: EndoNet: a deep architecture for recognition tasks on laparoscopic videos. IEEE Trans. Med. Imaging 36(1), 86\u201397 (2016)","journal-title":"IEEE Trans. Med. Imaging"},{"key":"37_CR25","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"449","DOI":"10.1007\/978-3-030-32254-0_50","volume-title":"Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2019","author":"F Yi","year":"2019","unstructured":"Yi, F., Jiang, T.: Hard frame detection and online mapping for surgical phase recognition. In: Shen, D., et al. (eds.) MICCAI 2019. LNCS, vol. 11768, pp. 449\u2013457. Springer, Cham (2019). https:\/\/doi.org\/10.1007\/978-3-030-32254-0_50"},{"issue":"4","key":"37_CR26","doi-asserted-by":"publisher","first-page":"e191860","DOI":"10.1001\/jamanetworkopen.2019.1860","volume":"2","author":"F Yu","year":"2019","unstructured":"Yu, F., et al.: Assessment of automated identification of phases in videos of cataract surgery using machine learning and deep learning techniques. JAMA Netw. Open 2(4), e191860\u2013e191860 (2019)","journal-title":"JAMA Netw. Open"},{"key":"37_CR27","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"409","DOI":"10.1007\/978-3-030-59716-0_39","volume-title":"Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2020","author":"J Zhang","year":"2020","unstructured":"Zhang, J., et al.: Symmetric dilated convolution for surgical gesture recognition. In: Martel, A.L., et al. (eds.) MICCAI 2020. LNCS, vol. 12263, pp. 409\u2013418. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-59716-0_39"},{"key":"37_CR28","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"265","DOI":"10.1007\/978-3-030-00937-3_31","volume-title":"Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2018","author":"O Zisimopoulos","year":"2018","unstructured":"Zisimopoulos, O., et al.: DeepPhase: surgical phase recognition in CATARACTS videos. In: Frangi, A.F., Schnabel, J.A., Davatzikos, C., Alberola-L\u00f3pez, C., Fichtinger, G. (eds.) MICCAI 2018. LNCS, vol. 11073, pp. 265\u2013272. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-00937-3_31"}],"container-title":["Lecture Notes in Computer Science","Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2023"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-43996-4_37","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,7,4]],"date-time":"2024-07-04T16:07:31Z","timestamp":1720109251000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-43996-4_37"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023]]},"ISBN":["9783031439957","9783031439964"],"references-count":28,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-43996-4_37","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023]]},"assertion":[{"value":"1 October 2023","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"MICCAI","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Medical Image Computing and Computer-Assisted Intervention","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Vancouver, BC","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Canada","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2023","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"8 October 2023","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"12 October 2023","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"miccai2023","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/conferences.miccai.org\/2023\/en\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"CMT","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"2250","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"730","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"32% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"5","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}