{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,23]],"date-time":"2025-10-23T05:48:05Z","timestamp":1761198485875,"version":"3.44.0"},"publisher-location":"Cham","reference-count":33,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783032051134"},{"type":"electronic","value":"9783032051141"}],"license":[{"start":{"date-parts":[[2025,9,21]],"date-time":"2025-09-21T00:00:00Z","timestamp":1758412800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,9,21]],"date-time":"2025-09-21T00:00:00Z","timestamp":1758412800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-05114-1_53","type":"book-chapter","created":{"date-parts":[[2025,9,20]],"date-time":"2025-09-20T14:11:18Z","timestamp":1758377478000},"page":"552-562","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["StepAL: Step-Aware Active Learning for\u00a0Cataract Surgical Videos"],"prefix":"10.1007","author":[{"given":"Nisarg A.","family":"Shah","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bardia","family":"Safaei","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shameema","family":"Sikder","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"S. Swaroop","family":"Vedula","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Vishal M.","family":"Patel","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,9,21]]},"reference":[{"key":"53_CR1","doi-asserted-by":"crossref","unstructured":"Arnab, A., Dehghani, M., Heigold, G., Sun, C., Lu\u010di\u0107, M., Schmid, C.: ViViT: a video vision transformer. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6836\u20136846 (2021)","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"53_CR2","unstructured":"Ash, J.T., Zhang, C., Krishnamurthy, A., Langford, J., Agarwal, A.: Deep batch active learning by diverse, uncertain gradient lower bounds. arXiv preprint arXiv:1906.03671 (2019)"},{"key":"53_CR3","series-title":"Lecture Notes in Computer Science (Lecture Notes in Artificial Intelligence)","doi-asserted-by":"publisher","first-page":"35","DOI":"10.1007\/978-3-540-72927-3_5","volume-title":"Learning Theory","author":"M-F Balcan","year":"2007","unstructured":"Balcan, M.-F., Broder, A., Zhang, T.: Margin based active learning. In: Bshouty, N.H., Gentile, C. (eds.) COLT 2007. LNCS (LNAI), vol. 4539, pp. 35\u201350. Springer, Heidelberg (2007). https:\/\/doi.org\/10.1007\/978-3-540-72927-3_5"},{"key":"53_CR4","doi-asserted-by":"crossref","unstructured":"Caramalau, R., Bhattarai, B., Kim, T.K.: Sequential graph convolutional network for active learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9583\u20139592 (2021)","DOI":"10.1109\/CVPR46437.2021.00946"},{"key":"53_CR5","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"343","DOI":"10.1007\/978-3-030-59716-0_33","volume-title":"Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2020","author":"T Czempiel","year":"2020","unstructured":"Czempiel, T., et al.: TeCNO: surgical phase recognition with multi-stage temporal convolutional networks. In: Martel, A.L., et al. (eds.) MICCAI 2020. LNCS, vol. 12263, pp. 343\u2013352. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-59716-0_33"},{"key":"53_CR6","doi-asserted-by":"publisher","first-page":"133","DOI":"10.1023\/A:1007330508534","volume":"28","author":"Y Freund","year":"1997","unstructured":"Freund, Y., Seung, H.S., Shamir, E., Tishby, N.: Selective sampling using the query by committee algorithm. Mach. Learn. 28, 133\u2013168 (1997)","journal-title":"Mach. Learn."},{"key":"53_CR7","doi-asserted-by":"crossref","unstructured":"Funke, I., Mees, S.T., Weitz, J., Speidel, S.: Video-based surgical skill assessment using 3D convolutional neural networks. IJCARS (2019)","DOI":"10.1007\/s11548-019-01995-1"},{"key":"53_CR8","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"593","DOI":"10.1007\/978-3-030-87202-1_57","volume-title":"Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2021","author":"X Gao","year":"2021","unstructured":"Gao, X., Jin, Y., Long, Y., Dou, Q., Heng, P.-A.: Trans-SVNet: accurate phase recognition from surgical videos via\u00a0hybrid embedding aggregation transformer. In: de Bruijne, M., et al. (eds.) MICCAI 2021. LNCS, vol. 12904, pp. 593\u2013603. Springer, Cham (2021). https:\/\/doi.org\/10.1007\/978-3-030-87202-1_57"},{"issue":"1","key":"53_CR9","doi-asserted-by":"publisher","first-page":"373","DOI":"10.1038\/s41597-024-03193-4","volume":"11","author":"N Ghamsarian","year":"2024","unstructured":"Ghamsarian, N., et al.: Cataract-1K dataset for deep-learning-assisted analysis of cataract surgery videos. Sci. Data 11(1), 373 (2024)","journal-title":"Sci. Data"},{"issue":"1","key":"53_CR10","doi-asserted-by":"publisher","first-page":"81","DOI":"10.1007\/s41884-022-00081-x","volume":"6","author":"H Hino","year":"2023","unstructured":"Hino, H., Eguchi, S.: Active learning by query by committee with robust divergences. Inf. Geom 6(1), 81\u2013106 (2023)","journal-title":"Inf. Geom"},{"key":"53_CR11","unstructured":"Kay, W., et\u00a0al.: The kinetics human action video dataset. arXiv preprint arXiv:1705.06950 (2017)"},{"key":"53_CR12","unstructured":"Kingma, D.P., Ba, J.: Adam: a method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)"},{"key":"53_CR13","doi-asserted-by":"publisher","unstructured":"Ma, S., Du, H., Curran, K.M., Lawlor, A., Dong, R.: Adaptive curriculum query strategy for active learning in medical image classification. In: Linguraru, M.G., et al. (eds.) MICCAI 2024. LNCS, vol. 15011, pp. 48\u201357. Springer, Cham (2024). https:\/\/doi.org\/10.1007\/978-3-031-72120-5_5","DOI":"10.1007\/978-3-031-72120-5_5"},{"issue":"9","key":"53_CR14","doi-asserted-by":"publisher","first-page":"691","DOI":"10.1038\/s41551-017-0132-7","volume":"1","author":"L Maier-Hein","year":"2017","unstructured":"Maier-Hein, L., et al.: Surgical data science for next-generation interventions. Nat. Biomed. Eng. 1(9), 691\u2013696 (2017)","journal-title":"Nat. Biomed. Eng."},{"issue":"2","key":"53_CR15","doi-asserted-by":"publisher","first-page":"82","DOI":"10.1080\/13645706.2019.1584116","volume":"28","author":"N Padoy","year":"2019","unstructured":"Padoy, N.: Machine and deep learning for workflow recognition during surgery. Minim. Invasive Therapy Allied Technol. 28(2), 82\u201390 (2019)","journal-title":"Minim. Invasive Therapy Allied Technol."},{"issue":"9","key":"53_CR16","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3472291","volume":"54","author":"P Ren","year":"2021","unstructured":"Ren, P., et al.: A survey of deep active learning. ACM Comput. Surv. (CSUR) 54(9), 1\u201340 (2021)","journal-title":"ACM Comput. Surv. (CSUR)"},{"key":"53_CR17","doi-asserted-by":"crossref","unstructured":"Safaei, B., Patel, V.M.: Active learning for vision language models. In: Proceedings of the Winter Conference on Applications of Computer Vision (WACV), pp. 4902\u20134912, February 2025","DOI":"10.1109\/WACV61041.2025.00480"},{"key":"53_CR18","doi-asserted-by":"crossref","unstructured":"Safaei, B., Siddiqui, F., Xu, J., Patel, V.M., Lo, S.Y.: Filter images first, generate instructions later: pre-instruction data selection for visual instruction tuning. In: Proceedings of the Computer Vision and Pattern Recognition Conference, pp. 14247\u201314256 (2025)","DOI":"10.1109\/CVPR52734.2025.01329"},{"key":"53_CR19","doi-asserted-by":"crossref","unstructured":"Safaei, B., Vibashan, V., de Melo, C.M., Patel, V.M.: Entropic open-set active learning. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 38, no. 5, pp. 4686\u20134694 (2024)","DOI":"10.1609\/aaai.v38i5.28269"},{"key":"53_CR20","doi-asserted-by":"crossref","unstructured":"Safaei, B., VS, V., Patel, V.M.: Certainty and uncertainty guided active domain adaptation. arXiv preprint arXiv:2505.19421 (2025)","DOI":"10.1109\/ICIP55913.2025.11084455"},{"key":"53_CR21","doi-asserted-by":"publisher","unstructured":"Schoeffmann, K., Taschwer, M., Sarny, S., M\u00fcnzer, B., Primus, M.J., Putzgruber, D.: Cataract-101: video dataset of 101 cataract surgeries. In: Proceedings of the 9th ACM Multimedia Systems Conference, MMSys 2018, pp. 421\u2013425. Association for Computing Machinery, New York, NY, USA (2018). https:\/\/doi.org\/10.1145\/3204949.3208137","DOI":"10.1145\/3204949.3208137"},{"key":"53_CR22","unstructured":"Sener, O., Savarese, S.: Active learning for convolutional neural networks: a core-set approach. arXiv preprint arXiv:1708.00489 (2017)"},{"key":"53_CR23","doi-asserted-by":"crossref","unstructured":"Shah, N.A., Bandara, C., Skider, S., Vedula, S.S., Patel, V.M.: CSMAE: cataract surgical masked autoencoder (MAE) based pre-training. In: Proceedings of the International Symposium on Biomedical Imaging (ISBI) (2025)","DOI":"10.1109\/ISBI60581.2025.10981288"},{"key":"53_CR24","doi-asserted-by":"publisher","unstructured":"Shah, N.A., Sikder, S., Vedula, S.S., Patel, V.M.: GLSFormer: gated-long, short sequence transformer for step recognition in surgical videos. In: Greenspan, H., et al. (eds.) MICCAI 2023. LNCS, vol. 14228. Springer, Cham (2023). https:\/\/doi.org\/10.1007\/978-3-031-43996-4_37","DOI":"10.1007\/978-3-031-43996-4_37"},{"key":"53_CR25","doi-asserted-by":"crossref","unstructured":"Shah, N.A., Sikder, S., Vedula, S.S., Patel, V.M.: Step detection in cataract surgery videos. In: 2025 IEEE 22nd International Symposium on Biomedical Imaging (ISBI), pp.\u00a01\u20135. IEEE (2025)","DOI":"10.1109\/ISBI60581.2025.10980885"},{"key":"53_CR26","unstructured":"Shah, N.A., Xia, M., Vijay, S., Sikder, S., Vedula, S.S., Patel, V.M.: A vision foundation model for cataract surgery using joint-embedding predictive architecture. In: Medical Imaging with Deep Learning (2025)"},{"key":"53_CR27","doi-asserted-by":"crossref","unstructured":"Sinha, S., Ebrahimi, S., Darrell, T.: Variational adversarial active learning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 5972\u20135981 (2019)","DOI":"10.1109\/ICCV.2019.00607"},{"key":"53_CR28","doi-asserted-by":"crossref","unstructured":"Taketsugu, H., Ukita, N.: Active transfer learning for efficient video-specific human pose estimation. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 1880\u20131890 (2024)","DOI":"10.1109\/WACV57701.2024.00189"},{"issue":"1","key":"53_CR29","doi-asserted-by":"publisher","first-page":"86","DOI":"10.1109\/TMI.2016.2593957","volume":"36","author":"AP Twinanda","year":"2016","unstructured":"Twinanda, A.P., Shehata, S., Mutter, D., Marescaux, J., Mathelin, M., Padoy, N.: EndoNet: a deep architecture for recognition tasks on laparoscopic videos. IEEE Trans. Med. Imaging 36(1), 86\u201397 (2016)","journal-title":"IEEE Trans. Med. Imaging"},{"key":"53_CR30","doi-asserted-by":"crossref","unstructured":"Wang, D., Shang, Y.: A new active labeling method for deep learning. In: 2014 International Joint Conference on Neural Networks (IJCNN), pp. 112\u2013119. IEEE (2014)","DOI":"10.1109\/IJCNN.2014.6889457"},{"key":"53_CR31","doi-asserted-by":"crossref","unstructured":"Wu, J., Chen, J., Huang, D.: Entropy-based active learning for object detection with progressive diversity constraint. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9397\u20139406 (2022)","DOI":"10.1109\/CVPR52688.2022.00918"},{"issue":"4","key":"53_CR32","doi-asserted-by":"publisher","first-page":"e191860","DOI":"10.1001\/jamanetworkopen.2019.1860","volume":"2","author":"F Yu","year":"2019","unstructured":"Yu, F., et al.: Assessment of automated identification of phases in videos of cataract surgery using machine learning and deep learning techniques. JAMA Netw. Open 2(4), e191860\u2013e191860 (2019)","journal-title":"JAMA Netw. Open"},{"key":"53_CR33","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"265","DOI":"10.1007\/978-3-030-00937-3_31","volume-title":"Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2018","author":"O Zisimopoulos","year":"2018","unstructured":"Zisimopoulos, O., et al.: DeepPhase: surgical phase recognition in CATARACTS videos. In: Frangi, A.F., Schnabel, J.A., Davatzikos, C., Alberola-L\u00f3pez, C., Fichtinger, G. (eds.) MICCAI 2018. LNCS, vol. 11073, pp. 265\u2013272. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-00937-3_31"}],"container-title":["Lecture Notes in Computer Science","Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2025"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-05114-1_53","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,20]],"date-time":"2025-09-20T14:11:27Z","timestamp":1758377487000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-05114-1_53"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,9,21]]},"ISBN":["9783032051134","9783032051141"],"references-count":33,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-05114-1_53","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2025,9,21]]},"assertion":[{"value":"21 September 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"MICCAI","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Medical Image Computing and Computer-Assisted Intervention","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Daejeon","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Korea (Republic of)","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 September 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 September 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"miccai2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/conferences.miccai.org\/2025\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}