{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T11:26:57Z","timestamp":1783423617561,"version":"3.54.6"},"publisher-location":"Cham","reference-count":25,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032051134","type":"print"},{"value":"9783032051141","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,9,21]],"date-time":"2025-09-21T00:00:00Z","timestamp":1758412800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,9,21]],"date-time":"2025-09-21T00:00:00Z","timestamp":1758412800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-05114-1_30","type":"book-chapter","created":{"date-parts":[[2025,9,20]],"date-time":"2025-09-20T14:10:18Z","timestamp":1758377418000},"page":"310-319","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["HieraSurg: Hierarchy-Aware Diffusion Model for\u00a0Surgical Video Generation"],"prefix":"10.1007","author":[{"given":"Diego","family":"Biagini","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Nassir","family":"Navab","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Azade","family":"Farshad","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,9,21]]},"reference":[{"key":"30_CR1","unstructured":"Blattmann, A., et al.: Stable video diffusion: scaling latent video diffusion models to large datasets, arXiv preprint arXiv:2311.15127 (2023)"},{"key":"30_CR2","unstructured":"Yang, Z., et al.: CogVideoX: text-to-video diffusion models with an expert transformer. In: The Thirteenth International Conference on Learning Representations (2025)"},{"key":"30_CR3","unstructured":"Kong, W., et al.: HunyuanVideo: a systematic framework for large video generative models (2025)"},{"key":"30_CR4","unstructured":"Kang, B., et al.: How far is video generation from world model: a physical law perspective, arXiv preprint arXiv:2411.02385 (2024)"},{"issue":"8","key":"30_CR5","doi-asserted-by":"publisher","first-page":"1615","DOI":"10.1007\/s11548-024-03213-z","volume":"19","author":"DG Saragih","year":"2024","unstructured":"Saragih, D.G., Hibi, A., Tyrrell, P.N.: Using diffusion models to generate synthetic labeled data for medical image segmentation. Int. J. Comput. Assist. Radiol. Surg. 19(8), 1615\u20131625 (2024)","journal-title":"Int. J. Comput. Assist. Radiol. Surg."},{"key":"30_CR6","doi-asserted-by":"publisher","first-page":"230","DOI":"10.1007\/978-3-031-72089-5_22","volume-title":"Medical Image Computing and Computer Assisted Intervention MICCAI 2024","author":"C Li","year":"2024","unstructured":"Li, C., et al.: Endora: video generation models as\u0103endoscopy simulators. In: Linguraru, M.G., et al. (eds.) MICCAI 2024. LNCS, pp. 230\u2013240. Springer, Cham (2024). https:\/\/doi.org\/10.1007\/978-3-031-72089-5_22"},{"key":"30_CR7","doi-asserted-by":"crossref","unstructured":"Caron, M., et al.: Emerging properties in self-supervised vision transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9650\u20139660 (2021)","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"30_CR8","doi-asserted-by":"publisher","first-page":"146","DOI":"10.1007\/978-3-031-77610-6_14","volume-title":"Medical Image Computing and Computer Assisted Intervention MICCAI 2024 Workshops","author":"Y Yeganeh","year":"2025","unstructured":"Yeganeh, Y., et al.: VISAGE: video synthesis using action graphs for\u0103surgery. In: Celebi, M.E., Reyes, M., Chen, Z., Li, X. (eds.) MICCAI 2024. LNCS, pp. 146\u2013156. Springer, Cham (2025). https:\/\/doi.org\/10.1007\/978-3-031-77610-6_14"},{"key":"30_CR9","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PmLR (2021)"},{"key":"30_CR10","unstructured":"Ravi, N., et al.: SAM 2: segment anything in images and videos (2024)"},{"key":"30_CR11","doi-asserted-by":"crossref","unstructured":"Twinanda, A.P., Shehata, S., Mutter, D., Marescaux, J., de Mathelin, M., Padoy, N.: EndoNet: a deep architecture for recognition tasks on laparoscopic videos. IEEE Trans. Med. Imaging 36(1), 86\u201397 (2017). Conference Name: IEEE Transactions on Medical Imaging","DOI":"10.1109\/TMI.2016.2593957"},{"key":"30_CR12","doi-asserted-by":"crossref","unstructured":"Nwoye, C.I., et al.: Rendezvous: attention mechanisms for the recognition of surgical action triplets in endoscopic videos. In: Medical Image Analysis, vol. 78, p. 102\u2013433 (2022)","DOI":"10.1016\/j.media.2022.102433"},{"key":"30_CR13","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., Ommer, B.: Highresolution image synthesis with latent diffusion models. Presented at the Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10 684\u201310 695 (2022)"},{"key":"30_CR14","unstructured":"Ho, J., Jain, A., Abbeel, P.: Denoising diffusion probabilistic models. In: Advances in Neural Information Processing Systems, vol. 33, pp. 6840\u20136851, Curran Associates, Inc., (2020)"},{"key":"30_CR15","unstructured":"Hong, W.-Y., Kao, C.-L., Kuo, Y.-H., Wang, J.-R., Chang, W.-L., Shih, C.-S.: CholecSeg8k: a semantic segmentation dataset for laparoscopic cholecystectomy based on cholec80 (2020)"},{"key":"30_CR16","unstructured":"Murali, A., et al.: The endoscapes dataset for surgical scene segmentation, object detection, and critical view of safety assessment: Official splits and benchmark, arXiv preprint arXiv:2312.12429 (2023)"},{"key":"30_CR17","doi-asserted-by":"crossref","unstructured":"Li, Y., Ling, H., Ramakrishnan, I.V., Prasanna, P., Sasson, A., Gupta, H.: Critical view of safety assessment in laparoscopic cholecystectomy via segment anything model. In: 2024 46th Annual International Conference of the IEEE Engineering in Medicine and Biology Society (EMBC), pp. 1\u20136 (2024)","DOI":"10.1109\/EMBC53108.2024.10781674"},{"key":"30_CR18","doi-asserted-by":"crossref","unstructured":"Kirillov, A., et al.: Segment anything. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 4015\u20134026 (2023)","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"30_CR19","doi-asserted-by":"crossref","unstructured":"Ranzinger, M., Heinrich, G., Kautz, J., Molchanov, P.: AM-RADIO: agglomerative vision foundation model reduce all domains into one. Presented at the Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12 490\u201312 500 (2024)","DOI":"10.1109\/CVPR52733.2024.01187"},{"key":"30_CR20","doi-asserted-by":"crossref","unstructured":"Peebles, W., Xie, S.: Scalable diffusion models with transformers. Presented at the Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 4195\u20134205 (2023)","DOI":"10.1109\/ICCV51070.2023.00387"},{"key":"30_CR21","unstructured":"Yuan, K., Srivastav, V., Navab, N., Padoy, N.: Procedure-aware surgical video-language pretraining with hierarchical knowledge augmentation. Presented at the The Thirty-eighth Annual Conference on Neural Information Processing Systems (2024)"},{"key":"30_CR22","unstructured":"Unterthiner, T., Steenkiste, S.V., Kurach, K., Marinier, R., Michalski, M., Gelly, S.: FVD: a new metric for video generation (2019)"},{"key":"30_CR23","unstructured":"Heusel, M., Ramsauer, H., Unterthiner, T., Nessler, B., Hochreiter, S.: GANs trained by a two time-scale update rule converge to a local nash equilibrium. In: Advances in Neural Information Processing Systems, vol. 30 (2017)"},{"key":"30_CR24","doi-asserted-by":"crossref","unstructured":"Varghese, R., Sambath, M.: YOLOv8: a novel object detection algorithm with enhanced performance and robustness. In: 2024 International Conference on Advances in Data Engineering and Intelligent Computing Systems (ADICS), pp. 1\u20136 (2024)","DOI":"10.1109\/ADICS58448.2024.10533619"},{"key":"30_CR25","doi-asserted-by":"crossref","unstructured":"Zhang, L., Rao, A., Agrawala, M.: Adding conditional control to textto- image diffusion models. Presented at the Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3836\u20133847 (2023)","DOI":"10.1109\/ICCV51070.2023.00355"}],"container-title":["Lecture Notes in Computer Science","Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2025"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-05114-1_30","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,20]],"date-time":"2025-09-20T14:10:26Z","timestamp":1758377426000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-05114-1_30"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,9,21]]},"ISBN":["9783032051134","9783032051141"],"references-count":25,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-05114-1_30","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,9,21]]},"assertion":[{"value":"21 September 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"MICCAI","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Medical Image Computing and Computer-Assisted Intervention","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Daejeon","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Korea (Republic of)","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 September 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 September 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"28","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"miccai2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/conferences.miccai.org\/2025\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}