{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T15:47:06Z","timestamp":1778082426270,"version":"3.51.4"},"reference-count":79,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100023015","name":"ARPA-H","doi-asserted-by":"publisher","award":["1AY2AX000062"],"award-info":[{"award-number":["1AY2AX000062"]}],"id":[{"id":"10.13039\/100023015","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"NSF","doi-asserted-by":"publisher","award":["IIS-2115110"],"award-info":[{"award-number":["IIS-2115110"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000006","name":"ONR","doi-asserted-by":"publisher","award":["N000142512287"],"award-info":[{"award-number":["N000142512287"]}],"id":[{"id":"10.13039\/100000006","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000183","name":"ARO","doi-asserted-by":"publisher","award":["W911NF2110276"],"award-info":[{"award-number":["W911NF2110276"]}],"id":[{"id":"10.13039\/100000183","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,10,19]]},"DOI":"10.1109\/iccv51701.2025.01309","type":"proceedings-article","created":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T19:45:49Z","timestamp":1777491949000},"page":"14106-14116","source":"Crossref","is-referenced-by-count":3,"title":["Multi-Modal Few-Shot Temporal Action Segmentation"],"prefix":"10.1109","author":[{"given":"Zijia","family":"Lu","sequence":"first","affiliation":[{"name":"Northeastern University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ehsan","family":"Elhamifar","sequence":"additional","affiliation":[{"name":"Northeastern University"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01599"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.495"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19833-5_4"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01766"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2023\/77"},{"key":"ref6","article-title":"Dwivedi and Xavier Bresson. A generalization of transformer networks to graphs","author":"Prakash","year":"2020","journal-title":"arXiv preprint arXiv"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58520-4_33"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00644"},{"key":"ref9","first-page":"3575","article-title":"Farha and Jurgen Gall. Ms-tcn: Multi-stage temporal convolutional network for action segmentation","volume-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","author":"Abu"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2011.5995444"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00058"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.231"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ICME57554.2024.10687535"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.52202\/079017-2855"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-021-01453-z"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1503.02531"},{"key":"ref17","article-title":"Videograph: Recognizing minutes-long human activities in videos","volume-title":"ICCV Workshop on Scene Graph Representation and Learning","author":"Hussein"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52733.2024.01300"},{"key":"ref19","volume-title":"Multi-modal prompting for low-shot temporal action localization","author":"Ju","year":"2023"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00851"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.105"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/WACV.2016.7477701"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01234"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.113"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00937"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.00933"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01765"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr42600.2020.01083"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00968"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00634"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01926"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2020.3021756"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00826"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.342"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01852"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00930"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00798"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01928"},{"key":"ref39","volume-title":"Bit: Bi-level temporal modeling for efficient supervised action segmentation","author":"Lu","year":"2023"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01721"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.02241"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01758"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.5244\/C.35.101"},{"key":"ref44","article-title":"Multi-modal few-shot temporal action detection via visionlanguage meta-adaptation","author":"Nag","year":"2022","journal-title":"arXiv preprint arXiv"},{"key":"ref45","volume-title":"Activity graph transformer for temporal action localization","author":"Nawhal","year":"2021"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00054"},{"key":"ref47","article-title":"Towards understanding knowledge distillation","volume-title":"International Conference on Machine learning","author":"Phuong"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19772-7_17"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1410"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00627"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00771"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00334"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01722"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01781"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01002"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW63382.2024.00214"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.599"},{"key":"ref58","article-title":"Coarse to fine multi-resolution temporal convolutional network","volume":"abs\/2105.10859","author":"Singhania","year":"2021","journal-title":"CoRR"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.5555\/3294996.3295163"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2021.3089127"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72970-6_10"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00131"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00130"},{"key":"ref64","volume-title":"Graph attention networks. In International Conference on Learning Representations","author":"Veli\u010dkovi\u0107"},{"key":"ref65","article-title":"Matching networks for one shot learning","author":"Vinyals","year":"2016","journal-title":"Neural Information Processing Systems"},{"key":"ref66","article-title":"Frustratingly simple few-shot object detection","volume-title":"International Conference on Machine learning","author":"Wang"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1145\/3460426.3463643"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01253"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-017-1013-y"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.5244\/C.35.49"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02664"},{"key":"ref72","article-title":"Graph transformer networks","volume":"32","author":"Yun","year":"2019","journal-title":"Advances in neural information processing systems"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.01079"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58558-7_31"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1007\/s10489-023-05259-z"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.52202\/079017-1881"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01424"},{"key":"ref78","volume-title":"Languagebind: Extending video-language pretraining to n modality by language-based semantic alignment","author":"Zhu","year":"2023"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00365"}],"event":{"name":"2025 IEEE\/CVF International Conference on Computer Vision (ICCV)","location":"Honolulu, HI, USA","start":{"date-parts":[[2025,10,19]]},"end":{"date-parts":[[2025,10,25]]}},"container-title":["2025 IEEE\/CVF International Conference on Computer Vision (ICCV)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11443115\/11443287\/11444985.pdf?arnumber=11444985","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T06:37:47Z","timestamp":1777531067000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11444985\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,19]]},"references-count":79,"URL":"https:\/\/doi.org\/10.1109\/iccv51701.2025.01309","relation":{},"subject":[],"published":{"date-parts":[[2025,10,19]]}}}