{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,21]],"date-time":"2026-08-21T12:07:14Z","timestamp":1787314034844,"version":"build-2736575974"},"publisher-location":"Cham","reference-count":31,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031832420","type":"print"},{"value":"9783031832437","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-83243-7_10","type":"book-chapter","created":{"date-parts":[[2025,3,1]],"date-time":"2025-03-01T06:28:46Z","timestamp":1740810526000},"page":"106-117","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":11,"title":["SANGRIA: Surgical Video Scene Graph Optimization for\u00a0Surgical Workflow Prediction"],"prefix":"10.1007","author":[{"given":"\u00c7a\u011fhan","family":"K\u00f6ksal","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ghazal","family":"Ghazaei","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Felix","family":"Holm","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Azade","family":"Farshad","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Nassir","family":"Navab","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,3,1]]},"reference":[{"key":"10_CR1","doi-asserted-by":"crossref","unstructured":"Aflalo, A., Bagon, S., Kashti, T., Eldar, Y.: Deepcut: unsupervised segmentation using graph neural networks clustering. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 32\u201341 (2023)","DOI":"10.1109\/ICCVW60793.2023.00010"},{"key":"10_CR2","doi-asserted-by":"publisher","first-page":"24","DOI":"10.1016\/j.media.2018.11.008","volume":"52","author":"H Al Hajj","year":"2019","unstructured":"Al Hajj, H., et al.: Cataracts: challenge on automatic tool annotation for cataract surgery. Med. Image Anal. 52, 24\u201341 (2019)","journal-title":"Med. Image Anal."},{"key":"10_CR3","unstructured":"Bertasius, G., Wang, H., Torresani, L.: Is space-time attention all you need for video understanding? In: ICML, vol.\u00a02, p.\u00a04 (2021)"},{"key":"10_CR4","unstructured":"Bianchi, F.M., Grattarola, D., Alippi, C.: Spectral clustering with graph neural networks for graph pooling. In: International Conference on Machine Learning, pp. 874\u2013883. PMLR (2020)"},{"key":"10_CR5","doi-asserted-by":"crossref","unstructured":"Caron, M., et al.: Emerging properties in self-supervised vision transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9650\u20139660 (2021)","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"10_CR6","doi-asserted-by":"crossref","unstructured":"Carreira, J., Zisserman, A.: Quo vadis, action recognition? a new model and the kinetics dataset. In: proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6299\u20136308 (2017)","DOI":"10.1109\/CVPR.2017.502"},{"key":"10_CR7","unstructured":"Dosovitskiy, A., et\u00a0al.: An image is worth 16$$\\,\\times \\,$$16 words: transformers for image recognition at scale. In: International Conference on Learning Representations (2020)"},{"key":"10_CR8","doi-asserted-by":"crossref","unstructured":"Farshad, A., Yeganeh, Y., Chi, Y., Shen, C., Ommer, B., Navab, N.: Scenegenie: scene graph guided diffusion models for image synthesis. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 88\u201398 (2023)","DOI":"10.1109\/ICCVW60793.2023.00016"},{"key":"10_CR9","unstructured":"Francesco, L., et al.: Object-centric learning with slot attention. In: Advances in Neural Information Processing Systems, vol. 33, pp. 11525\u201311538 (2020)"},{"key":"10_CR10","doi-asserted-by":"crossref","unstructured":"Holm, F., Ghazaei, G., Czempiel, T., \u00d6zsoy, E., Saur, S., Navab, N.: Dynamic scene graph representation for surgical video. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 81\u201387 (2023)","DOI":"10.1109\/ICCVW60793.2023.00015"},{"key":"10_CR11","unstructured":"Kipf, T.N., Welling, M.: Semi-supervised classification with graph convolutional networks. arXiv preprint arXiv:1609.02907 (2016)"},{"key":"10_CR12","unstructured":"Kirillov, A., et\u00a0al.: Segment anything. arXiv preprint arXiv:2304.02643 (2023)"},{"key":"10_CR13","unstructured":"Li, S.J., AbuFarha, Y., Liu, Y., Cheng, M.M., Gall, J.: MS-TCN++: multi-stage temporal convolutional network for action segmentation. IEEE Trans. Pattern Anal. Mach. Intell. (2020)"},{"key":"10_CR14","doi-asserted-by":"crossref","unstructured":"Lindenberger, P., Sarlin, P.E., Pollefeys, M.: Lightglue: local feature matching at light speed. arXiv preprint arXiv:2306.13643 (2023)","DOI":"10.1109\/ICCV51070.2023.01616"},{"key":"10_CR15","doi-asserted-by":"crossref","unstructured":"Ma, M., Shao, M., Zhao, X., Fu, Y.: Prototype based feature learning for face image set classification. In: 2013 10th IEEE International Conference and Workshops on Automatic Face and Gesture Recognition (FG), pp.\u00a01\u20136. IEEE (2013)","DOI":"10.1109\/FG.2013.6553697"},{"key":"10_CR16","unstructured":"Murali, A., et al.: Latent graph representations for critical view of safety assessment. arXiv preprint arXiv:2212.04155 (2022)"},{"key":"10_CR17","doi-asserted-by":"crossref","unstructured":"Murali, A., et al.: Encoding surgical videos as latent spatiotemporal graphs for object and anatomy-driven reasoning. In: International Conference on Medical Image Computing and Computer-Assisted Intervention, pp. 647\u2013657. Springer (2023)","DOI":"10.1007\/978-3-031-43996-4_62"},{"key":"10_CR18","unstructured":"Oquab, M., et\u00a0al.: Dinov2: learning robust visual features without supervision. arXiv preprint arXiv:2304.07193 (2023)"},{"key":"10_CR19","doi-asserted-by":"crossref","unstructured":"Pissas, T., Ravasio, C.S., Da\u00a0Cruz, L., Bergeles, C.: Effective semantic segmentation in cataract surgery: What matters most? In: Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2021, pp. 509\u2013518 (2021)","DOI":"10.1007\/978-3-030-87202-1_49"},{"key":"10_CR20","doi-asserted-by":"crossref","unstructured":"Schoeffmann, K., Taschwer, M., Sarny, S., M\u00fcnzer, B., Primus, M.J., Putzgruber, D.: Cataract-101: video dataset of 101 cataract surgeries. In: Proceedings of the 9th ACM Multimedia Systems Conference, pp. 421\u2013425 (2018)","DOI":"10.1145\/3204949.3208137"},{"key":"10_CR21","doi-asserted-by":"publisher","first-page":"102751","DOI":"10.1016\/j.media.2023.102751","volume":"85","author":"L Sestini","year":"2023","unstructured":"Sestini, L., Rosa, B., De Momi, E., Ferrigno, G., Padoy, N.: Fun-sis: a fully unsupervised approach for surgical instrument segmentation. Med. Image Anal. 85, 102751 (2023)","journal-title":"Med. Image Anal."},{"key":"10_CR22","doi-asserted-by":"crossref","unstructured":"Shah, N.A., Sikder, S., Vedula, S.S., Patel, V.M.: Glsformer: gated-long, short sequence transformer for step recognition in surgical videos. In: International Conference on Medical Image Computing and Computer-Assisted Intervention, pp. 386\u2013396. Springer (2023)","DOI":"10.1007\/978-3-031-43996-4_37"},{"key":"10_CR23","doi-asserted-by":"crossref","unstructured":"Sharma, S., Nwoye, C.I., Mutter, D., Padoy, N.: Rendezvous in time: an attention-based temporal fusion approach for surgical triplet recognition. Int. J. Comput. Assist. Radiol. Surg. 1\u20137 (2023)","DOI":"10.1007\/s11548-023-02914-1"},{"key":"10_CR24","doi-asserted-by":"crossref","unstructured":"Sharma, S., Nwoye, C.I., Mutter, D., Padoy, N.: Surgical action triplet detection by mixed supervised learning of instrument-tissue interactions. In: International Conference on Medical Image Computing and Computer-Assisted Intervention, pp. 505\u2013514. Springer (2023)","DOI":"10.1007\/978-3-031-43996-4_48"},{"issue":"8","key":"10_CR25","doi-asserted-by":"publisher","first-page":"888","DOI":"10.1109\/34.868688","volume":"22","author":"J Shi","year":"2000","unstructured":"Shi, J., Malik, J.: Normalized cuts and image segmentation. IEEE Trans. Pattern Anal. Mach. Intell. 22(8), 888\u2013905 (2000)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"127","key":"10_CR26","first-page":"1","volume":"24","author":"A Tsitsulin","year":"2023","unstructured":"Tsitsulin, A., Palowitch, J., Perozzi, B., M\u00fcller, E.: Graph clustering with graph neural networks. J. Mach. Learn. Res. 24(127), 1\u201321 (2023)","journal-title":"J. Mach. Learn. Res."},{"issue":"1","key":"10_CR27","doi-asserted-by":"publisher","first-page":"86","DOI":"10.1109\/TMI.2016.2593957","volume":"36","author":"AP Twinanda","year":"2016","unstructured":"Twinanda, A.P., Shehata, S., Mutter, D., Marescaux, J., De Mathelin, M., Padoy, N.: Endonet: a deep architecture for recognition tasks on laparoscopic videos. IEEE Trans. Med. Imaging 36(1), 86\u201397 (2016)","journal-title":"IEEE Trans. Med. Imaging"},{"key":"10_CR28","doi-asserted-by":"crossref","unstructured":"Wang, X., Girdhar, R., Yu, S.X., Misra, I.: Cut and learn for unsupervised object detection and instance segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3124\u20133134 (2023)","DOI":"10.1109\/CVPR52729.2023.00305"},{"key":"10_CR29","doi-asserted-by":"crossref","unstructured":"Wang, X., Misra, I., Zeng, Z., Girdhar, R., Darrell, T.: Videocutler: surprisingly simple unsupervised video instance segmentation. arXiv preprint arXiv:2308.14710 (2023)","DOI":"10.1109\/CVPR52733.2024.02147"},{"key":"10_CR30","doi-asserted-by":"crossref","unstructured":"Wang, Y., et al.: Tokencut: segmenting objects in images and videos with self-supervised transformer and normalized cut. IEEE Trans. Pattern Anal. Mach. Intell. (2023)","DOI":"10.1109\/TPAMI.2023.3305122"},{"key":"10_CR31","doi-asserted-by":"crossref","unstructured":"Zhao, Z., Jin, Y., Gao, X., Dou, Q., Heng, P.A.: Learning motion flows for semi-supervised instrument segmentation from robotic surgical video. In: Medical Image Computing and Computer Assisted Intervention\u2013MICCAI 2020: 23rd International Conference, Lima, Peru, October 4\u20138, 2020, Proceedings, Part III 23, pp. 679\u2013689. Springer (2020)","DOI":"10.1007\/978-3-030-59716-0_65"}],"container-title":["Lecture Notes in Computer Science","Graphs in Biomedical Image Analysis"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-83243-7_10","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,3,1]],"date-time":"2025-03-01T06:28:56Z","timestamp":1740810536000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-83243-7_10"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9783031832420","9783031832437"],"references-count":31,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-83243-7_10","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"1 March 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"GRAIL","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Workshop on Graphs in Biomedical Image Analysis","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Marrakesh","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Morocco","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"6 October 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"6 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"6","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"grail2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/grail-miccai.github.io\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}