{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,23]],"date-time":"2026-06-23T01:08:56Z","timestamp":1782176936281,"version":"3.54.5"},"publisher-location":"Cham","reference-count":80,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031732348","type":"print"},{"value":"9783031732355","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,9,30]],"date-time":"2024-09-30T00:00:00Z","timestamp":1727654400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,9,30]],"date-time":"2024-09-30T00:00:00Z","timestamp":1727654400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-73235-5_27","type":"book-chapter","created":{"date-parts":[[2024,9,29]],"date-time":"2024-09-29T06:01:53Z","timestamp":1727589713000},"page":"481-500","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":20,"title":["OphNet: A Large-Scale Video Benchmark for\u00a0Ophthalmic Surgical Workflow Understanding"],"prefix":"10.1007","author":[{"given":"Ming","family":"Hu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Peng","family":"Xia","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lin","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Siyuan","family":"Yan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Feilong","family":"Tang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhongxing","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yimin","family":"Luo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kaimin","family":"Song","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jurgen","family":"Leitner","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xuelian","family":"Cheng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jun","family":"Cheng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chi","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kaijing","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zongyuan","family":"Ge","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,9,30]]},"reference":[{"key":"27_CR1","unstructured":"Fair use on Youtube. https:\/\/support.google.com\/youtube\/answer\/9783148?hl=en#:~:text=If%20the%20use%20of%20copyright,copyright%20removal%20request%20to%20YouTube"},{"key":"27_CR2","unstructured":"Youtube\u2019s copyright exception policy. https:\/\/www.youtube.com\/howyoutubeworks\/policies\/copyright\/#copyright-exceptions"},{"key":"27_CR3","unstructured":"Adrito, D., et al.: PitVis: workflow recognition in endoscopic pituitary surgery"},{"key":"27_CR4","doi-asserted-by":"publisher","unstructured":"Al Hajj, H., et al.: CATARACTS: challenge on automatic tool annotation for cataract surgery. Med. Image Anal. 52, 24\u201341 (2019). https:\/\/doi.org\/10.1016\/j.media.2018.11.008. https:\/\/www.sciencedirect.com\/science\/article\/pii\/S136184151830865X","DOI":"10.1016\/j.media.2018.11.008"},{"key":"27_CR5","doi-asserted-by":"crossref","unstructured":"Alwassel, H., Giancola, S., Ghanem, B.: TSP: temporally-sensitive pretraining of video encoders for localization tasks. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision Workshops, pp. 3173\u20133183 (2021)","DOI":"10.1109\/ICCVW54120.2021.00356"},{"issue":"1","key":"27_CR6","doi-asserted-by":"publisher","first-page":"22208","DOI":"10.1038\/s41598-020-79173-6","volume":"10","author":"O Bar","year":"2020","unstructured":"Bar, O., et al.: Impact of data on generalization of AI for surgical intelligence applications. Sci. Rep. 10(1), 22208 (2020). https:\/\/doi.org\/10.1038\/s41598-020-79173-6","journal-title":"Sci. Rep."},{"key":"27_CR7","doi-asserted-by":"publisher","unstructured":"Bernal, J., S\u00e1nchez, F.J., Fern\u00e1ndez-Esparrach, G., Gil, D., Rodr\u00edguez, C., Vilari\u00f1o, F.: WM-DOVA maps for accurate polyp highlighting in colonoscopy: validation vs. saliency maps from physicians. Comput. Med. Imag. Graph. 43, 99\u2013111 (2015). https:\/\/doi.org\/10.1016\/j.compmedimag.2015.02.007. https:\/\/www.sciencedirect.com\/science\/article\/pii\/S0895611115000567","DOI":"10.1016\/j.compmedimag.2015.02.007"},{"key":"27_CR8","unstructured":"Bodenstedt, S., et al.: Unsupervised temporal context learning using convolutional neural networks for laparoscopic workflow analysis. arXiv preprint arXiv:1702.03684 (2017)"},{"key":"27_CR9","doi-asserted-by":"publisher","unstructured":"Borgli, H., et al.: HyperKvasir, a comprehensive multi-class image and video dataset for gastrointestinal endoscopy. Sci. Data 7(1), 283 (2020). https:\/\/doi.org\/10.1038\/s41597-020-00622-y","DOI":"10.1038\/s41597-020-00622-y"},{"key":"27_CR10","doi-asserted-by":"crossref","unstructured":"Carreira, J., Zisserman, A.: Quo Vadis, action recognition? A new model and the kinetics dataset. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6299\u20136308 (2017)","DOI":"10.1109\/CVPR.2017.502"},{"key":"27_CR11","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"343","DOI":"10.1007\/978-3-030-59716-0_33","volume-title":"Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2020","author":"T Czempiel","year":"2020","unstructured":"Czempiel, T., et al.: TeCNO: surgical phase recognition with multi-stage temporal convolutional networks. In: Martel, A.L., et al. (eds.) MICCAI 2020. LNCS, vol. 12263, pp. 343\u2013352. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-59716-0_33"},{"key":"27_CR12","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"604","DOI":"10.1007\/978-3-030-87202-1_58","volume-title":"Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2021","author":"T Czempiel","year":"2021","unstructured":"Czempiel, T., Paschali, M., Ostler, D., Kim, S.T., Busam, B., Navab, N.: OperA: attention-regularized transformers for surgical phase recognition. In: de Bruijne, M., et al. (eds.) MICCAI 2021, Part IV 24. LNCS, vol. 12904, pp. 604\u2013614. Springer, Cham (2021). https:\/\/doi.org\/10.1007\/978-3-030-87202-1_58"},{"issue":"4","key":"27_CR13","doi-asserted-by":"publisher","first-page":"364","DOI":"10.1093\/comjnl\/20.4.364","volume":"20","author":"D Defays","year":"1977","unstructured":"Defays, D.: An efficient algorithm for a complete link method. Comput. J. 20(4), 364\u2013366 (1977). https:\/\/doi.org\/10.1093\/comjnl\/20.4.364","journal-title":"Comput. J."},{"key":"27_CR14","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.J., Li, K., Fei-Fei, L.: ImageNet: a large-scale hierarchical image database. In: 2009 IEEE Conference on Computer Vision and Pattern Recognition, pp. 248\u2013255. IEEE (2009)","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"27_CR15","doi-asserted-by":"crossref","unstructured":"Dong, S., Hu, H., Lian, D., Luo, W., Qian, Y., Gao, S.: Weakly supervised video representation learning with unaligned text for sequential videos. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2437\u20132447 (2023)","DOI":"10.1109\/CVPR52729.2023.00241"},{"key":"27_CR16","doi-asserted-by":"publisher","unstructured":"Duong, H.T., Le, V.T., Hoang, V.T.: Deep learning-based anomaly detection in video surveillance: a survey. Sensors 23(11) (2023). https:\/\/doi.org\/10.3390\/s23115024. https:\/\/www.mdpi.com\/1424-8220\/23\/11\/5024","DOI":"10.3390\/s23115024"},{"key":"27_CR17","doi-asserted-by":"crossref","unstructured":"Heilbron, F.C., Escorcia, V., Ghanem, B., Niebles, J.C.: ActivityNet: a large-scale video benchmark for human activity understanding. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 961\u2013970 (2015)","DOI":"10.1109\/CVPR.2015.7298698"},{"key":"27_CR18","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C.: X3D: expanding architectures for efficient video recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 203\u2013213 (2020)","DOI":"10.1109\/CVPR42600.2020.00028"},{"key":"27_CR19","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Fan, H., Malik, J., He, K.: SlowFast networks for video recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6202\u20136211 (2019)","DOI":"10.1109\/ICCV.2019.00630"},{"key":"27_CR20","doi-asserted-by":"crossref","unstructured":"Forslund\u00a0Jacobsen, M., Konge, L., Alberti, M., la\u00a0Cour, M., Park, Y.S., Thomsen, A.S.S.: Robot-assisted vitreoretinal surgery improves surgical accuracy compared with manual surgery: a randomized trial in a simulated setting. Retina 40(11), 2091\u20132098 (2020)","DOI":"10.1097\/IAE.0000000000002720"},{"key":"27_CR21","unstructured":"Ghamsarian, N., et al.: Cataract-1K: cataract surgery dataset for scene segmentation, phase recognition, and irregularity detection. arXiv preprint arXiv:2312.06295 (2023)"},{"key":"27_CR22","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"76","DOI":"10.1007\/978-3-030-87237-3_8","volume-title":"Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2021","author":"N Ghamsarian","year":"2021","unstructured":"Ghamsarian, N., Taschwer, M., Putzgruber-Adamitsch, D., Sarny, S., El-Shabrawi, Y., Schoeffmann, K.: LensID: a CNN-RNN-based framework towards lens irregularity detection in cataract surgery videos. In: de Bruijne, M., et al. (eds.) MICCAI 2021. LNCS, vol. 12908, pp. 76\u201386. Springer, Cham (2021). https:\/\/doi.org\/10.1007\/978-3-030-87237-3_8"},{"key":"27_CR23","doi-asserted-by":"publisher","unstructured":"Ghamsarian, N., Taschwer, M., Putzgruber-Adamitsch, D., Sarny, S., Schoeffmann, K.: Relevance detection in cataract surgery videos by spatio-temporal action localization. In: 25th International Conference on Pattern Recognition, ICPR 2020, Virtual Event, Milan, Italy, 10\u201315 January 2021, pp. 10720\u201310727. IEEE (2020). https:\/\/doi.org\/10.1109\/ICPR48806.2021.9412525","DOI":"10.1109\/ICPR48806.2021.9412525"},{"key":"27_CR24","unstructured":"Grammatikopoulou, M., et al.: CaDIS: cataract dataset for image segmentation. arXiv preprint arXiv:1906.11586 (2019)"},{"key":"27_CR25","doi-asserted-by":"crossref","unstructured":"Gu, C., et al.: AVA: a video dataset of spatio-temporally localized atomic visual actions (2018)","DOI":"10.1109\/CVPR.2018.00633"},{"issue":"1\u20133","key":"27_CR26","doi-asserted-by":"publisher","first-page":"185","DOI":"10.1016\/0004-3702(81)90024-2","volume":"17","author":"BK Horn","year":"1981","unstructured":"Horn, B.K., Schunck, B.G.: Determining optical flow. Artif. Intell. 17(1\u20133), 185\u2013203 (1981)","journal-title":"Artif. Intell."},{"issue":"6","key":"27_CR27","doi-asserted-by":"publisher","first-page":"531","DOI":"10.1007\/s11633-022-1371-y","volume":"19","author":"GP Ji","year":"2022","unstructured":"Ji, G.P., et al.: Video polyp segmentation: a deep learning perspective. Mach. Intell. Res. 19(6), 531\u2013549 (2022). https:\/\/doi.org\/10.1007\/s11633-022-1371-y","journal-title":"Mach. Intell. Res."},{"issue":"5","key":"27_CR28","doi-asserted-by":"publisher","first-page":"1114","DOI":"10.1109\/TMI.2017.2787657","volume":"37","author":"Y Jin","year":"2018","unstructured":"Jin, Y., et al.: SV-RCNet: workflow recognition from surgical videos using recurrent convolutional network. IEEE Trans. Med. Imaging 37(5), 1114\u20131126 (2018)","journal-title":"IEEE Trans. Med. Imaging"},{"key":"27_CR29","unstructured":"Kay, W., et\u00a0al.: The kinetics human action video dataset. arXiv preprint arXiv:1705.06950 (2017)"},{"key":"27_CR30","doi-asserted-by":"publisher","unstructured":"Li, J., et al.: Imitation learning from expert video data for dissection trajectory prediction in endoscopic surgical procedure. In: Greenspan, H., et al. (eds.) Medical Image Computing and Computer Assisted Intervention, MICCAI 2023. LNCS, vol. 14228, pp. 494\u2013504. Springer, Cham (2023). https:\/\/doi.org\/10.1007\/978-3-031-43996-4_47","DOI":"10.1007\/978-3-031-43996-4_47"},{"key":"27_CR31","doi-asserted-by":"crossref","unstructured":"Li, Y., et al.: MViTv2: improved multiscale vision transformers for classification and detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4804\u20134814 (2022)","DOI":"10.1109\/CVPR52688.2022.00476"},{"key":"27_CR32","doi-asserted-by":"publisher","unstructured":"Lin, S., et al.: Semantic-super: a semantic-aware surgical perception framework for endoscopic tissue identification, reconstruction, and tracking. In: 2023 IEEE International Conference on Robotics and Automation (ICRA), pp. 4739\u20134746 (2023). https:\/\/doi.org\/10.1109\/ICRA48891.2023.10160746","DOI":"10.1109\/ICRA48891.2023.10160746"},{"key":"27_CR33","doi-asserted-by":"crossref","unstructured":"Lin, T., Liu, X., Li, X., Ding, E., Wen, S.: BMN: boundary-matching network for temporal action proposal generation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3889\u20133898 (2019)","DOI":"10.1109\/ICCV.2019.00399"},{"key":"27_CR34","unstructured":"Liu, Z., et al.: Video swin transformer. arXiv preprint arXiv:2106.13230 (2021)"},{"issue":"7","key":"27_CR35","doi-asserted-by":"publisher","first-page":"1704","DOI":"10.1007\/s11263-023-01779-w","volume":"131","author":"F Long","year":"2023","unstructured":"Long, F., Yao, T., Qiu, Z., Tian, X., Luo, J., Mei, T.: Bi-calibration networks for weakly-supervised video representation learning. Int. J. Comput. Vis. 131(7), 1704\u20131721 (2023). https:\/\/doi.org\/10.1007\/s11263-023-01779-w","journal-title":"Int. J. Comput. Vis."},{"issue":"2","key":"27_CR36","doi-asserted-by":"publisher","first-page":"553","DOI":"10.1007\/s00464-017-5878-1","volume":"32","author":"C Loukas","year":"2018","unstructured":"Loukas, C.: Video content analysis of surgical procedures. Surg. Endosc. 32(2), 553\u2013568 (2018). https:\/\/doi.org\/10.1007\/s00464-017-5878-1","journal-title":"Surg. Endosc."},{"key":"27_CR37","unstructured":"Lucas, B.D., Kanade, T., et\u00a0al.: An iterative image registration technique with an application to stereo vision, vol.\u00a081, Vancouver (1981)"},{"key":"27_CR38","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"387","DOI":"10.1007\/978-3-030-87240-3_37","volume-title":"Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2021","author":"Y Ma","year":"2021","unstructured":"Ma, Y., Chen, X., Cheng, K., Li, Y., Sun, B.: LDPolypVideo benchmark: a large-scale colonoscopy video dataset of diverse polyps. In: de Bruijne, M., et al. (eds.) MICCAI 2021. LNCS, vol. 12905, pp. 387\u2013396. Springer, Cham (2021). https:\/\/doi.org\/10.1007\/978-3-030-87240-3_37"},{"issue":"1","key":"27_CR39","doi-asserted-by":"publisher","first-page":"101","DOI":"10.1038\/s41597-021-00882-2","volume":"8","author":"L Maier-Hein","year":"2021","unstructured":"Maier-Hein, L., et al.: Heidelberg colorectal data set for surgical data science in the sensor operating room. Sci. Data 8(1), 101 (2021)","journal-title":"Sci. Data"},{"issue":"9","key":"27_CR40","doi-asserted-by":"publisher","first-page":"2051","DOI":"10.1109\/TMI.2016.2547947","volume":"35","author":"P Mesejo","year":"2016","unstructured":"Mesejo, P., et al.: Computer-aided classification of gastrointestinal lesions in regular colonoscopy. IEEE Trans. Med. Imaging 35(9), 2051\u20132063 (2016). https:\/\/doi.org\/10.1109\/TMI.2016.2547947","journal-title":"IEEE Trans. Med. Imaging"},{"key":"27_CR41","unstructured":"Ming, H., et al.: NurViD: a large expert-level video database for nursing procedure activity understanding. In: Thirty-Seventh Conference on Neural Information Processing Systems Datasets and Benchmarks Track (2023)"},{"key":"27_CR42","doi-asserted-by":"crossref","unstructured":"Ni, B., et al.: Expanding language-image pretrained models for general video recognition (2022)","DOI":"10.1007\/978-3-031-19772-7_1"},{"key":"27_CR43","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"139","DOI":"10.1007\/978-3-030-36711-4_13","volume-title":"Neural Information Processing","author":"Z-L Ni","year":"2019","unstructured":"Ni, Z.-L., et al.: RAUNet: residual attention U-Net for semantic segmentation of cataract surgical instruments. In: Gedeon, T., Wong, K.W., Lee, M. (eds.) ICONIP 2019. LNCS, vol. 11954, pp. 139\u2013149. Springer, Cham (2019). https:\/\/doi.org\/10.1007\/978-3-030-36711-4_13"},{"key":"27_CR44","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"364","DOI":"10.1007\/978-3-030-59716-0_35","volume-title":"Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2020","author":"CI Nwoye","year":"2020","unstructured":"Nwoye, C.I., et al.: Recognition of instrument-tissue interactions in endoscopic videos via action triplets. In: Martel, A.L., et al. (eds.) MICCAI 2020, Part III 23. LNCS, vol. 12263, pp. 364\u2013374. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-59716-0_35"},{"key":"27_CR45","unstructured":"Nwoye, C.I., Padoy, N.: Data splits and metrics for benchmarking methods on surgical action triplet datasets. arXiv preprint arXiv:2204.05235 (2022)"},{"key":"27_CR46","doi-asserted-by":"publisher","first-page":"102433","DOI":"10.1016\/j.media.2022.102433","volume":"78","author":"CI Nwoye","year":"2022","unstructured":"Nwoye, C.I., et al.: Rendezvous: attention mechanisms for the recognition of surgical action triplets in endoscopic videos. Med. Image Anal. 78, 102433 (2022)","journal-title":"Med. Image Anal."},{"issue":"1","key":"27_CR47","doi-asserted-by":"publisher","first-page":"139","DOI":"10.1007\/s11548-022-02785-y","volume":"18","author":"X Pan","year":"2023","unstructured":"Pan, X., Gao, X., Wang, H., Zhang, W., Mu, Y., He, X.: Temporal-based Swin Transformer network for workflow recognition of surgical video. Int. J. Comput. Assist. Radiol. Surg. 18(1), 139\u2013147 (2023). https:\/\/doi.org\/10.1007\/s11548-022-02785-y","journal-title":"Int. J. Comput. Assist. Radiol. Surg."},{"key":"27_CR48","unstructured":"Radford, A., et\u00a0al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PMLR (2021)"},{"key":"27_CR49","doi-asserted-by":"crossref","unstructured":"Rasheed, H., Khattak, M.U., Maaz, M., Khan, S., Khan, F.S.: Fine-tuned CLIP models are efficient video learners. In: The IEEE\/CVF Conference on Computer Vision and Pattern Recognition (2023)","DOI":"10.1109\/CVPR52729.2023.00633"},{"key":"27_CR50","doi-asserted-by":"publisher","first-page":"101920","DOI":"10.1016\/j.media.2020.101920","volume":"70","author":"T Ross","year":"2021","unstructured":"Ross, T., et al.: Comparative validation of multi-instance instrument segmentation in endoscopy: results of the ROBUST-MIS 2019 challenge. Med. Image Anal. 70, 101920 (2021)","journal-title":"Med. Image Anal."},{"key":"27_CR51","doi-asserted-by":"publisher","first-page":"925","DOI":"10.1007\/s11548-018-1772-0","volume":"13","author":"T Ross","year":"2018","unstructured":"Ross, T., et al.: Exploiting the potential of unlabeled endoscopic video data with self-supervised learning. Int. J. Comput. Assist. Radiol. Surg. 13, 925\u2013933 (2018)","journal-title":"Int. J. Comput. Assist. Radiol. Surg."},{"key":"27_CR52","doi-asserted-by":"publisher","unstructured":"Ro\u00df, T., et al.: Comparative validation of multi-instance instrument segmentation in endoscopy: results of the ROBUST-MIS 2019 challenge. Med. Image Anal. 70, 101920 (2021). https:\/\/doi.org\/10.1016\/j.media.2020.101920. https:\/\/www.sciencedirect.com\/science\/article\/pii\/S136184152030284X","DOI":"10.1016\/j.media.2020.101920"},{"key":"27_CR53","doi-asserted-by":"crossref","unstructured":"Sato, F., Hachiuma, R., Sekii, T.: Prompt-guided zero-shot anomaly action recognition using pretrained deep skeleton features. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6471\u20136480 (2023)","DOI":"10.1109\/CVPR52729.2023.00626"},{"key":"27_CR54","doi-asserted-by":"publisher","unstructured":"Schoeffmann, K., Husslein, H., Kletz, S., Petscharnig, S., M\u00fcnzer, B., Beecks, C.: Video retrieval in laparoscopic video recordings with dynamic content descriptors. Multim. Tools Appl. 77(13), 16813\u201316832 (2018). https:\/\/doi.org\/10.1007\/s11042-017-5252-2","DOI":"10.1007\/s11042-017-5252-2"},{"key":"27_CR55","doi-asserted-by":"publisher","unstructured":"Schoeffmann, K., Taschwer, M., Sarny, S., M\u00fcnzer, B., Primus, M.J., Putzgruber, D.: Cataract-101: video dataset of 101 cataract surgeries. In: C\u00e9sar, P., Zink, M., Murray, N. (eds.) Proceedings of the 9th ACM Multimedia Systems Conference, MMSys 2018, Amsterdam, The Netherlands, 12\u201315 June 2018, pp. 421\u2013425. ACM (2018). https:\/\/doi.org\/10.1145\/3204949.3208137","DOI":"10.1145\/3204949.3208137"},{"key":"27_CR56","doi-asserted-by":"crossref","unstructured":"Shi, D., Zhong, Y., Cao, Q., Ma, L., Li, J., Tao, D.: TriDet: temporal action detection with relative boundary modeling. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18857\u201318866 (2023)","DOI":"10.1109\/CVPR52729.2023.01808"},{"issue":"9","key":"27_CR57","doi-asserted-by":"publisher","first-page":"1573","DOI":"10.1007\/s11548-020-02198-9","volume":"15","author":"X Shi","year":"2020","unstructured":"Shi, X., Jin, Y., Dou, Q., Heng, P.A.: LRTD: long-range temporal dependency based active learning for surgical workflow recognition. Int. J. Comput. Assist. Radiol. Surg. 15(9), 1573\u20131584 (2020)","journal-title":"Int. J. Comput. Assist. Radiol. Surg."},{"key":"27_CR58","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"510","DOI":"10.1007\/978-3-319-46448-0_31","volume-title":"Computer Vision \u2013 ECCV 2016","author":"GA Sigurdsson","year":"2016","unstructured":"Sigurdsson, G.A., Varol, G., Wang, X., Farhadi, A., Laptev, I., Gupta, A.: Hollywood in homes: crowdsourcing data collection for activity understanding. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9905, pp. 510\u2013526. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46448-0_31"},{"issue":"1","key":"27_CR59","doi-asserted-by":"publisher","first-page":"142","DOI":"10.1038\/s41597-021-00920-z","volume":"8","author":"PH Smedsrud","year":"2021","unstructured":"Smedsrud, P.H., et al.: Kvasir-Capsule, a video capsule endoscopy dataset. Sci. Data 8(1), 142 (2021). https:\/\/doi.org\/10.1038\/s41597-021-00920-z","journal-title":"Sci. Data"},{"key":"27_CR60","unstructured":"Soomro, K., Zamir, A.R., Shah, M.: UCF101: a dataset of 101 human actions classes from videos in the wild (2012)"},{"key":"27_CR61","unstructured":"Spaeth, G., Danesh-Meyer, H., Goldberg, I., Kampik, A.: Ophthalmic Surgery: Principles and Practice. Elsevier Health Sciences (2011). E-Book. https:\/\/books.google.com.hk\/books?id=wHWMUGH-5csC"},{"key":"27_CR62","doi-asserted-by":"crossref","unstructured":"Stauder, R., Ostler, D., Kranzfelder, M., Koller, S., Feu\u00dfner, H., Navab, N.: The TUM LapChole dataset for the M2CAI 2016 workflow challenge. arXiv preprint arXiv:1610.09278 (2016)","DOI":"10.1515\/iss-2017-0035"},{"key":"27_CR63","doi-asserted-by":"publisher","unstructured":"Tian, Y., et al.: Contrastive transformer-based multiple instance learning for weakly supervised polyp frame detection. In: Wang, L., Dou, Q., Fletcher, P.T., Speidel, S., Li, S. (eds.) Medical Image Computing and Computer Assisted Intervention, MICCAI 2022. LNCS, vol. 13433, pp. 88\u201398. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-16437-8_9","DOI":"10.1007\/978-3-031-16437-8_9"},{"key":"27_CR64","doi-asserted-by":"crossref","unstructured":"Tran, D., Wang, H., Torresani, L., Feiszli, M.: Video classification with channel-separated convolutional networks (2019)","DOI":"10.1109\/ICCV.2019.00565"},{"key":"27_CR65","doi-asserted-by":"crossref","unstructured":"Twinanda, A.P., Shehata, S., Mutter, D., Marescaux, J., de\u00a0Mathelin, M., Padoy, N.: EndoNet: a deep architecture for recognition tasks on laparoscopic videos. IEEE Trans. Med. Imaging (TMI) 36, 86\u201397 (2016). https:\/\/api.semanticscholar.org\/CorpusID:5633749","DOI":"10.1109\/TMI.2016.2593957"},{"issue":"4","key":"27_CR66","doi-asserted-by":"publisher","first-page":"1069","DOI":"10.1109\/TMI.2018.2878055","volume":"38","author":"AP Twinanda","year":"2019","unstructured":"Twinanda, A.P., Yengera, G., Mutter, D., Marescaux, J., Padoy, N.: RSDNet: learning to predict remaining surgery duration from laparoscopic videos without manual annotations. IEEE Trans. Med. Imaging 38(4), 1069\u20131078 (2019). https:\/\/doi.org\/10.1109\/TMI.2018.2878055","journal-title":"IEEE Trans. Med. Imaging"},{"key":"27_CR67","doi-asserted-by":"publisher","first-page":"102770","DOI":"10.1016\/j.media.2023.102770","volume":"86","author":"M Wagner","year":"2023","unstructured":"Wagner, M., et al.: Comparative validation of machine learning algorithms for surgical workflow and skill analysis with the HeiChole benchmark. Med. Image Anal. 86, 102770 (2023)","journal-title":"Med. Image Anal."},{"key":"27_CR68","doi-asserted-by":"publisher","unstructured":"Wang, T., Li, H., Pu, T., Yang, L.: Microsurgery robots: applications, design, and development. Sensors 23(20) (2023). https:\/\/doi.org\/10.3390\/s23208503. https:\/\/www.mdpi.com\/1424-8220\/23\/20\/8503","DOI":"10.3390\/s23208503"},{"key":"27_CR69","doi-asserted-by":"crossref","unstructured":"Wang, X., Zhang, S., Qing, Z., Shao, Y., Gao, C., Sang, N.: Self-supervised learning for semi-supervised temporal action proposal. In: CVPR (2021)","DOI":"10.1109\/CVPR46437.2021.00194"},{"key":"27_CR70","doi-asserted-by":"publisher","unstructured":"Wang, Z., et al.: AutoLaparo: a new dataset of integrated multi-tasks for image-guided surgical automation in laparoscopic hysterectomy. In: Wang, L., Dou, Q., Fletcher, P.T., Speidel, S., Li, S. (eds.) Medical Image Computing and Computer Assisted Intervention, MICCAI 2022. LNCS, vol. 13437, pp. 486\u2013496. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-16449-1_46","DOI":"10.1007\/978-3-031-16449-1_46"},{"key":"27_CR71","doi-asserted-by":"crossref","unstructured":"Wu, W., Sun, Z., Ouyang, W.: Revisiting classifier: transferring vision-language models for video recognition (2023)","DOI":"10.1609\/aaai.v37i3.25386"},{"key":"27_CR72","doi-asserted-by":"crossref","unstructured":"Wu, W., Wang, X., Luo, H., Wang, J., Yang, Y., Ouyang, W.: Bidirectional cross-modal knowledge exploration for video recognition with pre-trained vision-language models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (2023)","DOI":"10.1109\/CVPR52729.2023.00640"},{"key":"27_CR73","unstructured":"Yengera, G., Mutter, D., Marescaux, J., Padoy, N.: Less is more: surgical phase recognition with less annotations through self-supervised pre-training of CNN-LSTM networks. arXiv preprint arXiv:1805.08569 (2018)"},{"key":"27_CR74","doi-asserted-by":"publisher","unstructured":"Yu, F., et al.: Assessment of automated identification of phases in videos of cataract surgery using machine learning and deep learning techniques. JAMA Netw. Open 2(4), e191860\u2013e191860 (2019). https:\/\/doi.org\/10.1001\/jamanetworkopen.2019.1860","DOI":"10.1001\/jamanetworkopen.2019.1860"},{"key":"27_CR75","unstructured":"Yu, T., Mutter, D., Marescaux, J., Padoy, N.: Learning from a tiny dataset of manual annotations: a teacher\/student approach for surgical phase recognition. arXiv preprint arXiv:1812.00033 (2018)"},{"key":"27_CR76","doi-asserted-by":"crossref","unstructured":"Yuan, K., Srivastav, V., Navab, N., Padoy, N.: HecVL: hierarchical video-language pretraining for zero-shot surgical phase recognition. arXiv preprint arXiv:2405.10075 (2024)","DOI":"10.1007\/978-3-031-72089-5_29"},{"key":"27_CR77","unstructured":"Yuan, K., et al.: Learning multi-modal representations by watching hundreds of surgical video lectures (2023)"},{"key":"27_CR78","doi-asserted-by":"publisher","unstructured":"Zha, R., Cheng, X., Li, H., Harandi, M., Ge, Z.: EndoSurf: neural surface reconstruction of deformable tissues with stereo endoscope videos. In: Greenspan, H., et al. (eds.) Medical Image Computing and Computer Assisted Intervention, MICCAI 2023. LNCS, vol. 14228, pp. 13\u201323. Springer, Cham (2023). https:\/\/doi.org\/10.1007\/978-3-031-43996-4_2","DOI":"10.1007\/978-3-031-43996-4_2"},{"key":"27_CR79","doi-asserted-by":"publisher","unstructured":"Zhang, CL., Wu, J., Li, Y.: ActionFormer: localizing moments of actions with transformers. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) Computer Vision, ECCV 2022. LNCS, vol. 13664, pp. 492\u2013510. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19772-7_29","DOI":"10.1007\/978-3-031-19772-7_29"},{"key":"27_CR80","doi-asserted-by":"crossref","unstructured":"Zhou, L., Xu, C., Corso, J.J.: Towards automatic learning of procedures from web instructional videos. In: AAAI Conference on Artificial Intelligence, pp. 7590\u20137598 (2018). https:\/\/www.aaai.org\/ocs\/index.php\/AAAI\/AAAI18\/paper\/view\/17344","DOI":"10.1609\/aaai.v32i1.12342"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-73235-5_27","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,28]],"date-time":"2024-11-28T21:16:03Z","timestamp":1732828563000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-73235-5_27"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,9,30]]},"ISBN":["9783031732348","9783031732355"],"references-count":80,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-73235-5_27","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,9,30]]},"assertion":[{"value":"30 September 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}