{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,19]],"date-time":"2026-02-19T22:02:02Z","timestamp":1771538522426,"version":"3.50.1"},"reference-count":55,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"InnoHK Program and in part by the National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U2441251"],"award-info":[{"award-number":["U2441251"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"InnoHK Program and in part by the National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U1836217"],"award-info":[{"award-number":["U1836217"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. on Image Process."],"published-print":{"date-parts":[[2026]]},"DOI":"10.1109\/tip.2026.3659752","type":"journal-article","created":{"date-parts":[[2026,2,6]],"date-time":"2026-02-06T20:52:47Z","timestamp":1770411167000},"page":"1966-1976","source":"Crossref","is-referenced-by-count":0,"title":["Procedure-Aware Hierarchical Alignment for Open Surgery Video-Language Pretraining"],"prefix":"10.1109","volume":"35","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1465-5553","authenticated-orcid":false,"given":"Boqiang","family":"Xu","sequence":"first","affiliation":[{"name":"State Key Laboratory of Multimodal Artificial Intelligence Systems (MAIS), Institute of Automation, Chinese Academy of Sciences (CASIA), Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jinlin","family":"Wu","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Multimodal Artificial Intelligence Systems (MAIS), Institute of Automation, Chinese Academy of Sciences (CASIA), Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3890-1894","authenticated-orcid":false,"given":"Jian","family":"Liang","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Multimodal Artificial Intelligence Systems (MAIS), Institute of Automation, Chinese Academy of Sciences (CASIA), Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4029-9935","authenticated-orcid":false,"given":"Zhenan","family":"Sun","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Multimodal Artificial Intelligence Systems (MAIS), Institute of Automation, Chinese Academy of Sciences (CASIA), Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4315-7556","authenticated-orcid":false,"given":"Hongbin","family":"Liu","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Multimodal Artificial Intelligence Systems (MAIS), Institute of Automation, Chinese Academy of Sciences (CASIA), Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4516-9729","authenticated-orcid":false,"given":"Jiebo","family":"Luo","sequence":"additional","affiliation":[{"name":"Hong Kong Institute of Science and Innovation, Hong Kong, SAR, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0791-189X","authenticated-orcid":false,"given":"Zhen","family":"Lei","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Multimodal Artificial Intelligence Systems (MAIS), Institute of Automation, Chinese Academy of Sciences (CASIA), Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00175"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1038\/s41598-020-79173-6"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-59716-0_33"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-87202-1_58"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1016\/j.media.2025.103716"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1016\/j.ijsu.2020.07.017"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1038\/s41746-020-00376-2"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00028"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00630"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA40945.2020.9196674"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-87202-1_57"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1007\/s11548-022-02559-6"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1001\/jamasurg.2023.6262"},{"key":"ref15","article-title":"CaDIS: Cataract dataset for image segmentation","author":"Grammatikopoulou","year":"2019","journal-title":"arXiv:1906.11586"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01092"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73235-5_27"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1097\/IAE.0000000000002720"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/TMI.2017.2787657"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/TMI.2021.3069471"},{"key":"ref22","article-title":"The kinetics human action video dataset","author":"Kay","year":"2017","journal-title":"arXiv:1705.06950"},{"key":"ref23","article-title":"Adam: A method for stochastic optimization","author":"Kingma","year":"2014","journal-title":"arXiv:1412.6980"},{"key":"ref24","first-page":"9694","article-title":"Align before fuse: Vision and language representation learning with momentum distillation","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Li"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ICRA48891.2023.10160746"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1007\/s00464-017-5878-1"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1038\/s41597-021-00882-2"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/TASSP.1980.1163491"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v28i1.8939"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19772-7_1"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-36711-4_13"},{"key":"ref32","article-title":"Data splits and metrics for method benchmarking on surgical action triplet datasets","author":"Nwoye","year":"2022","journal-title":"arXiv:2204.05235"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1016\/j.media.2022.102433"},{"key":"ref34","article-title":"Representation learning with contrastive predictive coding","author":"van den Oord","year":"2018","journal-title":"arXiv:1807.03748"},{"key":"ref35","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Radford"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00633"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1001\/jamasurg.2014.1791"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/MPUL.2011.942929"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1007\/s11042-017-5252-2"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01808"},{"key":"ref41","article-title":"TemporalMaxer: Maximize temporal context with only max pooling for temporal action localization","author":"Tang","year":"2023","journal-title":"arXiv:2303.09055"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref43","article-title":"Qwen2-VL: Enhancing vision-language model\u2019s perception of the world at any resolution","author":"Wang","year":"2024","journal-title":"arXiv:2409.12191"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.3390\/s23208503"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-16449-1_46"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.544"},{"key":"ref47","first-page":"53688","article-title":"Learning fine-grained view-invariant representations from unpaired ego-exo videos via temporal alignment","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Xue"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72952-2_18"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1001\/jamanetworkopen.2019.1860"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.52202\/079017-3907"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1016\/j.media.2025.103644"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00692"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-43996-4_2"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19772-7_29"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-24553-9_53"}],"container-title":["IEEE Transactions on Image Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/83\/11355710\/11373258.pdf?arnumber=11373258","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,19]],"date-time":"2026-02-19T21:01:16Z","timestamp":1771534876000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11373258\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"references-count":55,"URL":"https:\/\/doi.org\/10.1109\/tip.2026.3659752","relation":{},"ISSN":["1057-7149","1941-0042"],"issn-type":[{"value":"1057-7149","type":"print"},{"value":"1941-0042","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]}}}