{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,28]],"date-time":"2026-07-28T01:04:46Z","timestamp":1785200686191,"version":"3.55.0"},"publisher-location":"Cham","reference-count":33,"publisher":"Springer International Publishing","isbn-type":[{"value":"9783030872014","type":"print"},{"value":"9783030872021","type":"electronic"}],"license":[{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021]]},"DOI":"10.1007\/978-3-030-87202-1_57","type":"book-chapter","created":{"date-parts":[[2021,9,22]],"date-time":"2021-09-22T19:03:23Z","timestamp":1632337403000},"page":"593-603","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":113,"title":["Trans-SVNet: Accurate Phase Recognition from Surgical Videos via\u00a0Hybrid Embedding Aggregation Transformer"],"prefix":"10.1007","author":[{"given":"Xiaojie","family":"Gao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yueming","family":"Jin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yonghao","family":"Long","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qi","family":"Dou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Pheng-Ann","family":"Heng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2021,9,21]]},"reference":[{"key":"57_CR1","unstructured":"Ba, J.L., Kiros, J.R., Hinton, G.E.: Layer normalization. arXiv preprint arXiv:1607.06450 (2016)"},{"issue":"1","key":"57_CR2","doi-asserted-by":"publisher","first-page":"2","DOI":"10.1016\/j.ijmedinf.2006.01.003","volume":"76","author":"N Bricon-Souf","year":"2007","unstructured":"Bricon-Souf, N., Newman, C.R.: Context awareness in health care: a review. Int. J. Med. Informatics 76(1), 2\u201312 (2007)","journal-title":"Int. J. Med. Informatics"},{"key":"57_CR3","doi-asserted-by":"publisher","unstructured":"Charri\u00e8re, K., et al.: Real-time analysis of cataract surgery videos using statistical models. Multimedia Tools Appl., 1\u201319 (2017). https:\/\/doi.org\/10.1007\/s11042-017-4793-8","DOI":"10.1007\/s11042-017-4793-8"},{"key":"57_CR4","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"343","DOI":"10.1007\/978-3-030-59716-0_33","volume-title":"Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2020","author":"T Czempiel","year":"2020","unstructured":"Czempiel, T., et al.: TeCNO: surgical phase recognition with multi-stage temporal convolutional networks. In: Martel, A.L., et al. (eds.) MICCAI 2020. LNCS, vol. 12263, pp. 343\u2013352. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-59716-0_33"},{"issue":"6","key":"57_CR5","doi-asserted-by":"publisher","first-page":"1081","DOI":"10.1007\/s11548-016-1371-x","volume":"11","author":"O Dergachyova","year":"2016","unstructured":"Dergachyova, O., Bouget, D., Huaulm\u00e9, A., Morandi, X., Jannin, P.: Automatic data-driven real-time segmentation and recognition of surgical workflow. Int. J. Comput. Assist. Radiol. Surg. 11(6), 1081\u20131089 (2016). https:\/\/doi.org\/10.1007\/s11548-016-1371-x","journal-title":"Int. J. Comput. Assist. Radiol. Surg."},{"key":"57_CR6","doi-asserted-by":"crossref","unstructured":"Farha, Y.A., Gall, J.: MS-TCN: multi-stage temporal convolutional network for action segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3575\u20133584 (2019)","DOI":"10.1109\/CVPR.2019.00369"},{"key":"57_CR7","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"467","DOI":"10.1007\/978-3-030-32254-0_52","volume-title":"Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2019","author":"I Funke","year":"2019","unstructured":"Funke, I., et al.: Using 3D convolutional neural networks to learn spatiotemporal features for automatic surgical gesture recognition in video. In: Shen, D., et al. (eds.) MICCAI 2019. LNCS, vol. 11768, pp. 467\u2013475. Springer, Cham (2019). https:\/\/doi.org\/10.1007\/978-3-030-32254-0_52"},{"key":"57_CR8","doi-asserted-by":"crossref","unstructured":"Gao, X., Jin, Y., Dou, Q., Heng, P.A.: Automatic gesture recognition in robot-assisted surgery with reinforcement learning and tree search. In: IEEE International Conference on Robotics and Automation, pp. 8440\u20138446. IEEE (2020)","DOI":"10.1109\/ICRA40945.2020.9196674"},{"key":"57_CR9","unstructured":"Han, K., et al.: A survey on visual transformer. arXiv preprint arXiv:2012.12556 (2020)"},{"key":"57_CR10","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"issue":"8","key":"57_CR11","doi-asserted-by":"publisher","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter, S., Schmidhuber, J.: Long short-term memory. Neural Comput. 9(8), 1735\u20131780 (1997)","journal-title":"Neural Comput."},{"key":"57_CR12","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"440","DOI":"10.1007\/978-3-030-32254-0_49","volume-title":"Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2019","author":"Y Jin","year":"2019","unstructured":"Jin, Y., Cheng, K., Dou, Q., Heng, P.-A.: Incorporating temporal prior from motion flow for instrument segmentation in minimally invasive surgery video. In: Shen, D., et al. (eds.) MICCAI 2019. LNCS, vol. 11768, pp. 440\u2013448. Springer, Cham (2019). https:\/\/doi.org\/10.1007\/978-3-030-32254-0_49"},{"issue":"5","key":"57_CR13","doi-asserted-by":"publisher","first-page":"1114","DOI":"10.1109\/TMI.2017.2787657","volume":"37","author":"Y Jin","year":"2018","unstructured":"Jin, Y., et al.: SV-RCNet: workflow recognition from surgical videos using recurrent convolutional network. IEEE Trans. Med. Imaging 37(5), 1114\u20131126 (2018)","journal-title":"IEEE Trans. Med. Imaging"},{"key":"57_CR14","doi-asserted-by":"crossref","unstructured":"Jin, Y., et al.: Multi-task recurrent convolutional network with correlation loss for surgical video analysis. Med. Image Anal. 59, 101572 (2020)","DOI":"10.1016\/j.media.2019.101572"},{"key":"57_CR15","doi-asserted-by":"crossref","unstructured":"Jin, Y., Long, Y., Chen, C., Zhao, Z., Dou, Q., Heng, P.A.: Temporal memory relation network for workflow recognition from surgical video. IEEE Trans. Med. Imaging (2021)","DOI":"10.1109\/TMI.2021.3069471"},{"key":"57_CR16","doi-asserted-by":"crossref","unstructured":"Khan, S., Naseer, M., Hayat, M., Zamir, S.W., Khan, F.S., Shah, M.: Transformers in vision: a survey. arXiv preprint arXiv:2101.01169 (2021)","DOI":"10.1145\/3505244"},{"key":"57_CR17","doi-asserted-by":"crossref","unstructured":"Lea, C., Flynn, M.D., Vidal, R., Reiter, A., Hager, G.D.: Temporal convolutional networks for action segmentation and detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 156\u2013165 (2017)","DOI":"10.1109\/CVPR.2017.113"},{"key":"57_CR18","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"47","DOI":"10.1007\/978-3-319-49409-8_7","volume-title":"Computer Vision \u2013 ECCV 2016 Workshops","author":"C Lea","year":"2016","unstructured":"Lea, C., Vidal, R., Reiter, A., Hager, G.D.: Temporal convolutional networks: a unified approach to action segmentation. In: Hua, G., J\u00e9gou, H. (eds.) ECCV 2016. LNCS, vol. 9915, pp. 47\u201354. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-49409-8_7"},{"key":"57_CR19","doi-asserted-by":"crossref","unstructured":"Maier-Hein, L., et al.: Surgical data science for next-generation interventions. Nat. Biomed. Eng. (2017)","DOI":"10.1038\/s41551-017-0132-7"},{"key":"57_CR20","unstructured":"Mikolov, T., Sutskever, I., Chen, K., Corrado, G., Dean, J.: Distributed representations of words and phrases and their compositionality. In: Advances in Neural Information Processing Systems (2013)"},{"issue":"2","key":"57_CR21","doi-asserted-by":"publisher","first-page":"82","DOI":"10.1080\/13645706.2019.1584116","volume":"28","author":"N Padoy","year":"2019","unstructured":"Padoy, N.: Machine and deep learning for workflow recognition during surgery. Minimally Invasive Therapy Allied Technol. 28(2), 82\u201390 (2019)","journal-title":"Minimally Invasive Therapy Allied Technol."},{"key":"57_CR22","unstructured":"Padoy, N., Blum, T., Feussner, H., Berger, M.O., Navab, N.: On-line recognition of surgical activity for monitoring in the operating room. In: Proceedings of the AAAI Conference on Artificial Intelligence, pp. 1718\u20131724 (2008)"},{"issue":"12","key":"57_CR23","doi-asserted-by":"publisher","first-page":"2352","DOI":"10.1109\/TMI.2014.2340473","volume":"33","author":"G Quellec","year":"2014","unstructured":"Quellec, G., Lamard, M., Cochener, B., Cazuguel, G.: Real-time segmentation and recognition of surgical tasks in cataract surgery videos. IEEE Trans. Med. Imaging 33(12), 2352\u20132360 (2014)","journal-title":"IEEE Trans. Med. Imaging"},{"key":"57_CR24","unstructured":"Twinanda, A.P., Mutter, D., Marescaux, J., de Mathelin, M., Padoy, N.: Single-and multi-task architectures for surgical workflow challenge at M2CAI 2016. arXiv preprint arXiv:1610.08844 (2016)"},{"key":"57_CR25","unstructured":"Twinanda, A.P., Shehata, S., Mutter, D., Marescaux, J., De Mathelin, M., Padoy, N.: MICCAI modeling and monitoring of computer assisted interventions challenge. http:\/\/camma.u-strasbg.fr\/m2cai2016\/"},{"issue":"1","key":"57_CR26","doi-asserted-by":"publisher","first-page":"86","DOI":"10.1109\/TMI.2016.2593957","volume":"36","author":"AP Twinanda","year":"2017","unstructured":"Twinanda, A.P., Shehata, S., Mutter, D., Marescaux, J., De Mathelin, M., Padoy, N.: EndoNet: a deep architecture for recognition tasks on laparoscopic videos. IEEE Trans. Med. Imaging 36(1), 86\u201397 (2017)","journal-title":"IEEE Trans. Med. Imaging"},{"key":"57_CR27","unstructured":"Twinanda, A.P.: Vision-based approaches for surgical activity recognition using laparoscopic and RBGD videos. Ph.D. thesis, Strasbourg (2017)"},{"key":"57_CR28","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Advances in Neural Information Processing Systems, pp. 5998\u20136008 (2017)"},{"key":"57_CR29","doi-asserted-by":"crossref","unstructured":"Wang, Y., Solomon, J.M.: Deep closest point: learning representations for point cloud registration. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3523\u20133532 (2019)","DOI":"10.1109\/ICCV.2019.00362"},{"key":"57_CR30","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"449","DOI":"10.1007\/978-3-030-32254-0_50","volume-title":"Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2019","author":"F Yi","year":"2019","unstructured":"Yi, F., Jiang, T.: Hard frame detection and online mapping for surgical phase recognition. In: Shen, D., et al. (eds.) MICCAI 2019. LNCS, vol. 11768, pp. 449\u2013457. Springer, Cham (2019). https:\/\/doi.org\/10.1007\/978-3-030-32254-0_50"},{"key":"57_CR31","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"409","DOI":"10.1007\/978-3-030-59716-0_39","volume-title":"Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2020","author":"J Zhang","year":"2020","unstructured":"Zhang, J., et al.: Symmetric dilated convolution for surgical gesture recognition. In: Martel, A.L., et al. (eds.) MICCAI 2020. LNCS, vol. 12263, pp. 409\u2013418. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-59716-0_39"},{"key":"57_CR32","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"679","DOI":"10.1007\/978-3-030-59716-0_65","volume-title":"Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2020","author":"Z Zhao","year":"2020","unstructured":"Zhao, Z., Jin, Y., Gao, X., Dou, Q., Heng, P.-A.: Learning motion flows for\u00a0semi-supervised instrument segmentation from robotic surgical video. In: Martel, A.L., et al. (eds.) MICCAI 2020. LNCS, vol. 12263, pp. 679\u2013689. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-59716-0_65"},{"key":"57_CR33","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"265","DOI":"10.1007\/978-3-030-00937-3_31","volume-title":"Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2018","author":"O Zisimopoulos","year":"2018","unstructured":"Zisimopoulos, O., et al.: DeepPhase: surgical phase recognition in CATARACTS videos. In: Frangi, A.F., Schnabel, J.A., Davatzikos, C., Alberola-L\u00f3pez, C., Fichtinger, G. (eds.) MICCAI 2018. LNCS, vol. 11073, pp. 265\u2013272. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-00937-3_31"}],"container-title":["Lecture Notes in Computer Science","Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2021"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-87202-1_57","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,1,9]],"date-time":"2023-01-09T23:14:16Z","timestamp":1673306056000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-030-87202-1_57"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021]]},"ISBN":["9783030872014","9783030872021"],"references-count":33,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-87202-1_57","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021]]},"assertion":[{"value":"21 September 2021","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"MICCAI","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Medical Image Computing and Computer-Assisted Intervention","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Strasbourg","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"France","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2021","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 September 2021","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"1 October 2021","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"24","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"miccai2021","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/miccai2021.org\/en\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Microsoft CMT","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"1622","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"531","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"33% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"4","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"The conference was held virtually.","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}