{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T16:21:48Z","timestamp":1778084508877,"version":"3.51.4"},"publisher-location":"Cham","reference-count":136,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031729690","type":"print"},{"value":"9783031729706","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,11,23]],"date-time":"2024-11-23T00:00:00Z","timestamp":1732320000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,23]],"date-time":"2024-11-23T00:00:00Z","timestamp":1732320000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72970-6_10","type":"book-chapter","created":{"date-parts":[[2024,11,22]],"date-time":"2024-11-22T10:51:59Z","timestamp":1732272719000},"page":"161-183","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":8,"title":["Two-Stage Active Learning for\u00a0Efficient Temporal Action Segmentation"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-4368-4265","authenticated-orcid":false,"given":"Yuhao","family":"Su","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5517-548X","authenticated-orcid":false,"given":"Ehsan","family":"Elhamifar","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,11,23]]},"reference":[{"key":"10_CR1","doi-asserted-by":"crossref","unstructured":"Aakur, S.N., Sarkar, S.: A perceptual prediction framework for self supervised event segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1197\u20131206 (2019)","DOI":"10.1109\/CVPR.2019.00129"},{"key":"10_CR2","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"137","DOI":"10.1007\/978-3-030-58517-4_9","volume-title":"Computer Vision \u2013 ECCV 2020","author":"S Agarwal","year":"2020","unstructured":"Agarwal, S., Arora, H., Anand, S., Arora, C.: Contextual diversity for active learning. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12361, pp. 137\u2013153. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58517-4_9"},{"key":"10_CR3","doi-asserted-by":"crossref","unstructured":"Angluin, D.: Queries and concept learning. Mach. Learn. 2, 319\u2013342 (1988)","DOI":"10.1007\/BF00116828"},{"key":"10_CR4","doi-asserted-by":"crossref","unstructured":"Aziere, N., Todorovic, S.: Multistage temporal convolution transformer for action segmentation. Image Vis. Comput. 128, 104567 (2022)","DOI":"10.1016\/j.imavis.2022.104567"},{"key":"10_CR5","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"657","DOI":"10.1007\/978-3-031-19778-9_38","volume-title":"Computer Vision \u2013 ECCV 2022","author":"S Bansal","year":"2022","unstructured":"Bansal, S., Arora, C., Jawahar, C.: My view is the best view: procedure learning from egocentric videos. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13673, pp. 657\u2013675. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19778-9_38"},{"key":"10_CR6","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"52","DOI":"10.1007\/978-3-031-19833-5_4","volume-title":"Computer Vision \u2013 ECCV 2022","author":"N Behrmann","year":"2022","unstructured":"Behrmann, N., Golestaneh, S.A., Kolter, Z., Gall, J., Noroozi, M.: Unified fully and timestamp supervised temporal action segmentation via sequence to sequence translation. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13695, pp. 52\u201368. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19833-5_4"},{"key":"10_CR7","doi-asserted-by":"crossref","unstructured":"Beluch, W.H., Genewein, T., N\u00fcrnberger, A., K\u00f6hler, J.M.: The power of ensembles for active learning in image classification. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 9368\u20139377 (2018)","DOI":"10.1109\/CVPR.2018.00976"},{"key":"10_CR8","doi-asserted-by":"crossref","unstructured":"Bueno-Benito, E., Vecino, B.T., Dimiccoli, M.: Leveraging triplet loss for unsupervised action segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) Workshops, pp. 4922\u20134930 (2023)","DOI":"10.1109\/CVPRW59228.2023.00520"},{"key":"10_CR9","doi-asserted-by":"crossref","unstructured":"Cabannes, V., Bottou, L., Lecun, Y., Balestriero, R.: Active self-supervised learning: a few low-cost relationships are all you need. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 16274\u201316283 (2023)","DOI":"10.1109\/ICCV51070.2023.01491"},{"key":"10_CR10","doi-asserted-by":"crossref","unstructured":"Cao, K., Ji, J., Cao, Z., Chang, C.Y., Niebles, J.C.: Few-shot video classification via temporal alignment. In: IEEE Conference on Computer Vision and Pattern Recognition (2020)","DOI":"10.1109\/CVPR42600.2020.01063"},{"key":"10_CR11","doi-asserted-by":"crossref","unstructured":"Cao, Y.T., Shi, Y., Yu, B., Wang, J., Tao, D.: Knowledge-aware federated active learning with non-IID data. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 22279\u201322289 (2023)","DOI":"10.1109\/ICCV51070.2023.02036"},{"key":"10_CR12","doi-asserted-by":"crossref","unstructured":"Carreira, J., Zisserman, A.: Quo vadis, action recognition? A new model and the kinetics dataset. In: IEEE Conference on Computer Vision and Pattern Recognition (2017)","DOI":"10.1109\/CVPR.2017.502"},{"key":"10_CR13","doi-asserted-by":"crossref","unstructured":"Chang, C.Y., Huang, D.A., Sui, Y., Fei-Fei, L., Niebles, J.C.: D3TW: discriminative differentiable dynamic time warping for weakly supervised action alignment and segmentation. In: IEEE Conference on Computer Vision and Pattern Recognition (2019)","DOI":"10.1109\/CVPR.2019.00366"},{"key":"10_CR14","doi-asserted-by":"crossref","unstructured":"Chang, X., Tung, F., Mori, G.: Learning discriminative prototypes with dynamic time warping. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8395\u20138404 (2021)","DOI":"10.1109\/CVPR46437.2021.00829"},{"key":"10_CR15","unstructured":"Chen, T., Kornblith, S., Norouzi, M., Hinton, G.E.: A simple framework for contrastive learning of visual representations. In: International Conference on Machine Learning (2020)"},{"key":"10_CR16","unstructured":"Cuturi, M., Blondel, M.: Soft-DTW: a differentiable loss function for time-series. In: International Conference on Machine Learning (2017)"},{"key":"10_CR17","doi-asserted-by":"crossref","unstructured":"Ding, G., Sener, F., Yao, A.: Temporal action segmentation: an analysis of modern techniques. IEEE Trans. Pattern Anal. Mach. Intell. (2023)","DOI":"10.1109\/TPAMI.2023.3327284"},{"key":"10_CR18","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"17","DOI":"10.1007\/978-3-031-19833-5_2","volume-title":"Computer Vision \u2013 ECCV 2022","author":"G Ding","year":"2022","unstructured":"Ding, G., Yao, A.: Leveraging action affinity and continuity for semi-supervised temporal action segmentation. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13695, pp. 17\u201332. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19833-5_2"},{"key":"10_CR19","doi-asserted-by":"crossref","unstructured":"Ding, G., Yao, A.: Temporal action segmentation with high-level complex activity labels. IEEE Trans. Multimed. (2022)","DOI":"10.1109\/TMM.2022.3231099"},{"key":"10_CR20","unstructured":"Ding, L., Xu, C.: TricorNet: a hybrid temporal convolutional and recurrent network for video action segmentation. arXiv preprint arXiv:1705.07818 (2017)"},{"key":"10_CR21","unstructured":"Ding, L., Xu, C.: Weakly-supervised action segmentation with iterative soft boundary assignment. In: IEEE Conference on Computer Vision and Pattern Recognition (2018)"},{"key":"10_CR22","doi-asserted-by":"crossref","unstructured":"Donahue, G., Elhamifar, E.: Learning to predict activity progress by self-supervised video alignment. In: IEEE Conference on Computer Vision and Pattern Recognition (2024)","DOI":"10.1109\/CVPR52733.2024.01766"},{"key":"10_CR23","doi-asserted-by":"crossref","unstructured":"Du, D., Su, B., Li, Y., Qi, Z., Si, L., Shan, Y.: Do we really need temporal convolutions in action segmentation? In: 2023 IEEE International Conference on Multimedia and Expo (ICME), pp. 1014\u20131019. IEEE (2023)","DOI":"10.1109\/ICME55011.2023.00178"},{"key":"10_CR24","doi-asserted-by":"crossref","unstructured":"Du, X., et al.: Be consistent! Improving procedural text comprehension using label consistency. In: Annual Meeting of the North American Association for Computational Linguistics (2019)","DOI":"10.18653\/v1\/N19-1244"},{"key":"10_CR25","doi-asserted-by":"crossref","unstructured":"Du, Z., Wang, Q.: Dilated transformer with feature aggregation module for action segmentation. Neural Process. Lett. 1\u201317 (2022)","DOI":"10.1007\/s11063-022-11133-9"},{"key":"10_CR26","doi-asserted-by":"crossref","unstructured":"Du, Z., Wang, X., Zhou, G., Wang, Q.: Fast and unsupervised action boundary detection for action segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3323\u20133332 (2022)","DOI":"10.1109\/CVPR52688.2022.00332"},{"key":"10_CR27","unstructured":"Dvornik, N., Hadji, I., Derpanis, K.G., Garg, A., Jepson, A.D.: Drop-DTW: aligning common signal between sequences while dropping outliers. Neural Inf. Process. Syst. (2021)"},{"key":"10_CR28","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"319","DOI":"10.1007\/978-3-031-19833-5_19","volume-title":"Computer Vision \u2013 ECCV 2022","author":"N Dvornik","year":"2022","unstructured":"Dvornik, N., et al.: Flow graph to video grounding for weakly-supervised multi-step localization. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13695, pp. 319\u2013335. Springer, Cham (2022)"},{"key":"10_CR29","doi-asserted-by":"crossref","unstructured":"Dvornik, N., Hadji, I., Zhang, R., Derpanis, K.G., Wildes, R.P., Jepson, A.D.: StepFormer: self-supervised step discovery and localization in instructional videos. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18952\u201318961 (2023)","DOI":"10.1109\/CVPR52729.2023.01817"},{"key":"10_CR30","doi-asserted-by":"crossref","unstructured":"Dwibedi, D., Aytar, Y., Tompson, J., Sermanet, P., Zisserman, A.: Temporal cycle-consistency learning. In: IEEE Conference on Computer Vision and Pattern Recognition (2019)","DOI":"10.1109\/CVPR.2019.00190"},{"key":"10_CR31","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"557","DOI":"10.1007\/978-3-030-58520-4_33","volume-title":"Computer Vision \u2013 ECCV 2020","author":"E Elhamifar","year":"2020","unstructured":"Elhamifar, E., Huynh, D.: Self-supervised multi-task procedure learning from instructional videos. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12362, pp. 557\u2013573. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58520-4_33"},{"key":"10_CR32","doi-asserted-by":"crossref","unstructured":"Fang, M., Li, Y., Cohn, T.: Learning how to active learn: a deep reinforcement learning approach. arXiv preprint arXiv:1708.02383 (2017)","DOI":"10.18653\/v1\/D17-1063"},{"key":"10_CR33","doi-asserted-by":"crossref","unstructured":"Farha, Y.A., Gall, J.: MS-TCN: multi-stage temporal convolutional network for action segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3575\u20133584 (2019)","DOI":"10.1109\/CVPR.2019.00369"},{"key":"10_CR34","doi-asserted-by":"crossref","unstructured":"Fathi, A., Ren, X., Rehg, J.M.: Learning to recognize objects in egocentric activities. In IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2011)","DOI":"10.1109\/CVPR.2011.5995444"},{"key":"10_CR35","unstructured":"Fathi, A., Ren, X., Rehg, J.M.: Learning to recognize objects in egocentric activities. In: IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2021)"},{"key":"10_CR36","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"562","DOI":"10.1007\/978-3-319-10593-2_37","volume-title":"Computer Vision \u2013 ECCV 2014","author":"A Freytag","year":"2014","unstructured":"Freytag, A., Rodner, E., Denzler, J.: Selecting influential examples: active learning with expected model output changes. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8692, pp. 562\u2013577. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10593-2_37"},{"key":"10_CR37","doi-asserted-by":"crossref","unstructured":"Fried, D., Alayrac, J.B., Blunsom, P., Dyer, C., Clark, S., Nematzadeh, A.: Learning to segment actions from observation and narration. In: Annual Meeting of the Association for Computational Linguistics (2020)","DOI":"10.18653\/v1\/2020.acl-main.231"},{"key":"10_CR38","doi-asserted-by":"crossref","unstructured":"Gabrys, R., Yaakobi, E., Milenkovic, O.: Codes in the Damerau distance for deletion and adjacent transposition correction. IEEE Trans. Inf. Theory (2017)","DOI":"10.1109\/TIT.2017.2778143"},{"key":"10_CR39","unstructured":"Gal, Y., Ghahramani, Z.: Dropout as a Bayesian approximation: representing model uncertainty in deep learning. In: International Conference on Machine Learning, pp. 1050\u20131059. PMLR (2016)"},{"key":"10_CR40","doi-asserted-by":"crossref","unstructured":"Gao, S.H., Han, Q., Li, Z.Y., Peng, P., Wang, L., Cheng, M.M.: Global2Local: efficient structure search for video action segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16805\u201316814 (2021)","DOI":"10.1109\/CVPR46437.2021.01653"},{"key":"10_CR41","unstructured":"Goel, K., Brunskill, E.: Learning procedural abstractions and evaluating discrete latent temporal structure. In: International Conference on Learning Representation (2019)"},{"key":"10_CR42","doi-asserted-by":"crossref","unstructured":"Hadji, I., Derpanis, K.G., Jepson, A.D.: Representation learning via global temporal alignment and cycle-consistency. In: IEEE Conference on Computer Vision and Pattern Recognition (2021)","DOI":"10.1109\/CVPR46437.2021.01092"},{"key":"10_CR43","doi-asserted-by":"crossref","unstructured":"Hampiholi, B., Jarvers, C., Mader, W., Neumann, H.: Depthwise separable temporal convolutional network for action segmentation. In: 2020 International Conference on 3D Vision (3DV), pp. 633\u2013641. IEEE (2020)","DOI":"10.1109\/3DV50981.2020.00073"},{"key":"10_CR44","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"137","DOI":"10.1007\/978-3-319-46493-0_9","volume-title":"Computer Vision \u2013 ECCV 2016","author":"D-A Huang","year":"2016","unstructured":"Huang, D.-A., Fei-Fei, L., Niebles, J.C.: Connectionist temporal modeling for weakly supervised action labeling. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9908, pp. 137\u2013153. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46493-0_9"},{"key":"10_CR45","doi-asserted-by":"crossref","unstructured":"Huang, S., Wang, T., Xiong, H., Huan, J., Dou, D.: Semi-supervised active learning with temporal output discrepancy. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3447\u20133456 (2021)","DOI":"10.1109\/ICCV48922.2021.00343"},{"key":"10_CR46","doi-asserted-by":"crossref","unstructured":"Huang, Y., Sugano, Y., Sato, Y.: Improving action segmentation via graph-based temporal reasoning. In: 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2020)","DOI":"10.1109\/CVPR42600.2020.01404"},{"key":"10_CR47","doi-asserted-by":"crossref","unstructured":"Ishikawa, Y., Kasai, S., Aoki, Y., Kataoka, H.: Alleviating over-segmentation errors by detecting action boundaries. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV), pp. 2322\u20132331 (2021)","DOI":"10.1109\/WACV48630.2021.00237"},{"key":"10_CR48","doi-asserted-by":"crossref","unstructured":"Ji, W., et al.: Are binary annotations sufficient? Video moment retrieval via hierarchical uncertainty-based active learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 23013\u201323022 (2023)","DOI":"10.1109\/CVPR52729.2023.02204"},{"key":"10_CR49","doi-asserted-by":"crossref","unstructured":"Joshi, A.J., Porikli, F., Papanikolopoulos, N.: Multi-class active learning for image classification. In: 2009 IEEE Conference on Computer Vision and Pattern Recognition, pp. 2372\u20132379. IEEE (2009)","DOI":"10.1109\/CVPRW.2009.5206627"},{"key":"10_CR50","doi-asserted-by":"crossref","unstructured":"Khan, H., et al.: Timestamp-supervised action segmentation with graph convolutional networks. In: 2022 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS), pp. 10619\u201310626. IEEE (2022)","DOI":"10.1109\/IROS47612.2022.9981351"},{"key":"10_CR51","unstructured":"Koide, S., Kawano, K., Kutsuna, T.: Neural edit operations for biological sequences. Adv. Neural Inf. Process. Syst. (2018)"},{"key":"10_CR52","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/978-3-031-19839-7_1","volume-title":"Computer Vision \u2013 ECCV 2022","author":"S Kothawade","year":"2022","unstructured":"Kothawade, S., Ghosh, S., Shekhar, S., Xiang, Y., Iyer, R.: Talisman: targeted active learning for object detection with rare classes and slices using submodular mutual information. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13698, pp. 1\u201316. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19839-7_1"},{"issue":"6","key":"10_CR53","doi-asserted-by":"publisher","first-page":"1382","DOI":"10.1109\/TSP.2002.1003062","volume":"50","author":"V Krishnamurthy","year":"2002","unstructured":"Krishnamurthy, V.: Algorithms for optimal scheduling and management of hidden Markov model sensors. IEEE Trans. Signal Process. 50(6), 1382\u20131397 (2002)","journal-title":"IEEE Trans. Signal Process."},{"key":"10_CR54","doi-asserted-by":"crossref","unstructured":"Kuehne, H., Arslan, A., Serre, T.: The language of actions: recovering the syntax and semantics of goal-directed human. In: IEEE Conference on Computer Vision and Pattern Recognition (2014)","DOI":"10.1109\/CVPR.2014.105"},{"key":"10_CR55","doi-asserted-by":"crossref","unstructured":"Kuehne, H., Richard, A., Gall, J.: Weakly supervised learning of actions from transcripts. Comput. Vision Image Underst. J. (2017)","DOI":"10.1016\/j.cviu.2017.06.004"},{"key":"10_CR56","doi-asserted-by":"crossref","unstructured":"Kukleva, A., Kuehne, H., Sener, F., Gall, J.: Unsupervised learning of action classes with continuous temporal embedding. In: IEEE Conference on Computer Vision and Pattern Recognition (2019)","DOI":"10.1109\/CVPR.2019.01234"},{"key":"10_CR57","doi-asserted-by":"crossref","unstructured":"Kye, S.M., Choi, K., Byun, H., Chang, B.: TiDAL: learning training dynamics for active learning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 22335\u201322345 (2023)","DOI":"10.1109\/ICCV51070.2023.02041"},{"key":"10_CR58","doi-asserted-by":"crossref","unstructured":"Lea, C., Flynn, M.D., Vidal, R., Reiter, A., Hager, G.D.: Temporal convolutional networks for action segmentation and detection. In: IEEE Conference on Computer Vision and Pattern Recognition (2017)","DOI":"10.1109\/CVPR.2017.113"},{"key":"10_CR59","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"36","DOI":"10.1007\/978-3-319-46487-9_3","volume-title":"Computer Vision \u2013 ECCV 2016","author":"C Lea","year":"2016","unstructured":"Lea, C., Reiter, A., Vidal, R., Hager, G.D.: Segmental spatiotemporal CNNs for fine-grained action segmentation. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9907, pp. 36\u201352. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46487-9_3"},{"key":"10_CR60","doi-asserted-by":"crossref","unstructured":"Lee, S., Lu, Z., Zhang, Z., Hoai, M., Elhamifar, E.: Error detection in egocentric procedural task videos. In: IEEE Conference on Computer Vision and Pattern Recognition (2024)","DOI":"10.1109\/CVPR52733.2024.01765"},{"key":"10_CR61","doi-asserted-by":"publisher","unstructured":"Lei, P., Todorovic, S.: Temporal deformable residual networks for action segmentation in videos. In: 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (2018). https:\/\/doi.org\/10.1109\/CVPR.2018.00705","DOI":"10.1109\/CVPR.2018.00705"},{"key":"10_CR62","unstructured":"Levenshtein, V.I.: Binary codes capable of correcting deletions, insertions and reversals. Sov. Phys. Dokl. 10 (1966)"},{"key":"10_CR63","doi-asserted-by":"crossref","unstructured":"Li, J., Lei, P., Todorovic, S.: Weakly supervised energy-based learning for action segmentation. In: International Conference on Computer Vision (2019)","DOI":"10.1109\/ICCV.2019.00634"},{"key":"10_CR64","doi-asserted-by":"crossref","unstructured":"Li, R., Zhang, B., Liu, J., Liu, W., Zhao, J., Teng, Z.: Heterogeneous diversity driven active learning for multi-object tracking. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 9932\u20139941 (2023)","DOI":"10.1109\/ICCV51070.2023.00911"},{"key":"10_CR65","doi-asserted-by":"publisher","unstructured":"Li, S.J., AbuFarha, Y., Liu, Y., Cheng, M.M., Gall, J.: MS-TCN++: multi-stage temporal convolutional network for action segmentation. IEEE Trans. Pattern Anal. Mach. Intell. 1 (2020). https:\/\/doi.org\/10.1109\/TPAMI.2020.3021756","DOI":"10.1109\/TPAMI.2020.3021756"},{"key":"10_CR66","doi-asserted-by":"publisher","first-page":"373","DOI":"10.1016\/j.neucom.2021.04.121","volume":"454","author":"Y Li","year":"2021","unstructured":"Li, Y., et al.: Efficient two-step networks for temporal action segmentation. Neurocomputing 454, 373\u2013381 (2021)","journal-title":"Neurocomputing"},{"key":"10_CR67","doi-asserted-by":"crossref","unstructured":"Li, Z., Abu\u00a0Farha, Y., Gall, J.: Temporal action segmentation from timestamp supervision. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition (2021)","DOI":"10.1109\/CVPR46437.2021.00826"},{"key":"10_CR68","doi-asserted-by":"crossref","unstructured":"Liu, D., Li, Q., Dinh, A., Jiang, T., Shah, M., Xu, C.: Diffusion action segmentation. arXiv preprint arXiv:2303.17959 (2023)","DOI":"10.1109\/ICCV51070.2023.00930"},{"key":"10_CR69","doi-asserted-by":"crossref","unstructured":"Liu, K., Li, Y., Liu, S., Tan, C., Shao, Z.: Reducing the label bias for timestamp supervised temporal action segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 6503\u20136513 (June 2023)","DOI":"10.1109\/CVPR52729.2023.00629"},{"key":"10_CR70","unstructured":"Liu, Z., et al.: Temporal segment transformer for action segmentation. arXiv preprint arXiv:2302.13074 (2023)"},{"key":"10_CR71","doi-asserted-by":"crossref","unstructured":"Liu, Z., Ding, H., Zhong, H., Li, W., Dai, J., He, C.: Influence selection for active learning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9274\u20139283 (2021)","DOI":"10.1109\/ICCV48922.2021.00914"},{"key":"10_CR72","doi-asserted-by":"crossref","unstructured":"Liu, Z., Wang, J., Gong, S., Lu, H., Tao, D.: Deep reinforcement active learning for human-in-the-loop person re-identification. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6122\u20136131 (2019)","DOI":"10.1109\/ICCV.2019.00622"},{"key":"10_CR73","doi-asserted-by":"crossref","unstructured":"Lu, Z., Elhamifar, E.: Weakly-supervised action segmentation and alignment via transcript-aware union-of-subspaces learning. In: International Conference on Computer Vision (2021)","DOI":"10.1109\/ICCV48922.2021.00798"},{"key":"10_CR74","doi-asserted-by":"crossref","unstructured":"Lu, Z., Elhamifar, E.: Set-supervised action learning in procedural task videos via pairwise order consistency. In: IEEE Conference on Computer Vision and Pattern Recognition (2022)","DOI":"10.1109\/CVPR52688.2022.01928"},{"key":"10_CR75","doi-asserted-by":"crossref","unstructured":"Lu, Z., Elhamifar, E.: FACT: frame-action cross-attention temporal modeling for efficient action segmentation. In: IEEE Conference on Computer Vision and Pattern Recognition (2024)","DOI":"10.1109\/CVPR52733.2024.01721"},{"key":"10_CR76","unstructured":"Luo, W., Schwing, A., Urtasun, R.: Latent structured active learning. Adv. Neural Inf. Process. Syst. 26 (2013)"},{"key":"10_CR77","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"580","DOI":"10.1007\/978-3-030-00934-2_65","volume-title":"Medical Image Computing and Computer Assisted Intervention \u2013 MICCAI 2018","author":"D Mahapatra","year":"2018","unstructured":"Mahapatra, D., Bozorgtabar, B., Thiran, J.-P., Reyes, M.: Efficient active learning for image classification and segmentation using a sample selection and conditional generative adversarial network. In: Frangi, A.F., Schnabel, J.A., Davatzikos, C., Alberola-L\u00f3pez, C., Fichtinger, G. (eds.) MICCAI 2018. LNCS, vol. 11071, pp. 580\u2013588. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-00934-2_65"},{"key":"10_CR78","unstructured":"Mahmood, R., Fidler, S., Law, M.T.: Low budget active learning via wasserstein distance: an integer programming approach. arXiv preprint arXiv:2106.02968 (2021)"},{"key":"10_CR79","doi-asserted-by":"crossref","unstructured":"Mayer, C., Timofte, R.: Adversarial sampling for active learning. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 3071\u20133079 (2020)","DOI":"10.1109\/WACV45572.2020.9093556"},{"key":"10_CR80","doi-asserted-by":"crossref","unstructured":"Miech, A., Alayrac, J.B., Smaira, L., Laptev, I., Sivic, J., Zisserman, A.: End-to-end learning of visual representations from uncurated instructional videos. In: IEEE Conference on Computer Vision and Pattern Recognition (2020)","DOI":"10.1109\/CVPR42600.2020.00990"},{"key":"10_CR81","doi-asserted-by":"crossref","unstructured":"Miech, A., Zhukov, D., Alayrac, J.B., Tapaswi, M., Laptev, I., Sivic, J.: Howto100M: learning a text-video embedding by watching hundred million narrated video CLIPs. In: International Conference on Computer Vision (2019)","DOI":"10.1109\/ICCV.2019.00272"},{"key":"10_CR82","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-74048-3","volume-title":"Information Retrieval for Music and Motion","author":"M M\u00fcller","year":"2007","unstructured":"M\u00fcller, M.: Information Retrieval for Music and Motion, vol. 2. Springer, Cham (2007). https:\/\/doi.org\/10.1007\/978-3-540-74048-3"},{"key":"10_CR83","doi-asserted-by":"crossref","unstructured":"Narr, A., Triebel, R., Cremers, D.: Stream-based active learning for efficient and adaptive classification of 3D objects. In: 2016 IEEE International Conference on Robotics and Automation (ICRA), pp. 227\u2013233. IEEE (2016)","DOI":"10.1109\/ICRA.2016.7487138"},{"key":"10_CR84","doi-asserted-by":"publisher","first-page":"108764","DOI":"10.1016\/j.patcog.2022.108764","volume":"129","author":"J Park","year":"2022","unstructured":"Park, J., Kim, D., Huh, S., Jo, S.: Maximization and restoration: action segmentation through dilation passing and temporal reconstruction. Pattern Recogn. 129, 108764 (2022)","journal-title":"Pattern Recogn."},{"key":"10_CR85","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"279","DOI":"10.1007\/978-3-031-19772-7_17","volume-title":"Computer Vision \u2013 ECCV 2022","author":"R Rahaman","year":"2022","unstructured":"Rahaman, R., Singhania, D., Thiery, A., Yao, A.: A generalized and robust framework for timestamp supervision in temporal action segmentation. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13664, pp. 279\u2013296. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19772-7_17"},{"key":"10_CR86","first-page":"14358","volume":"35","author":"A Rana","year":"2022","unstructured":"Rana, A., Rawat, Y.: Are all frames equal? Active sparse labeling for video action detection. Adv. Neural. Inf. Process. Syst. 35, 14358\u201314373 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"10_CR87","doi-asserted-by":"crossref","unstructured":"Rana, A.J., Rawat, Y.S.: Hybrid active learning via deep clustering for video action detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18867\u201318877 (2023)","DOI":"10.1109\/CVPR52729.2023.01809"},{"issue":"9","key":"10_CR88","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3472291","volume":"54","author":"P Ren","year":"2021","unstructured":"Ren, P., et al.: A survey of deep active learning. ACM Comput. Surv. (CSUR) 54(9), 1\u201340 (2021)","journal-title":"ACM Comput. Surv. (CSUR)"},{"key":"10_CR89","doi-asserted-by":"crossref","unstructured":"Richard, A., Kuehne, H., Gall, J.: Weakly supervised action learning with RNN based fine-to-coarse modeling. In: IEEE Conference on Computer Vision and Pattern Recognition (2017)","DOI":"10.1109\/CVPR.2017.140"},{"key":"10_CR90","doi-asserted-by":"crossref","unstructured":"Richard, A., Kuehne, H., Gall, J.: Action sets: weakly supervised action segmentation without ordering constraints. In: IEEE Conference on Computer Vision and Pattern Recognition (2018)","DOI":"10.1109\/CVPR.2018.00627"},{"key":"10_CR91","doi-asserted-by":"crossref","unstructured":"Rochan, M., Wang, Y.: Video summarization by learning from unpaired data. In: IEEE Conference on Computer Vision and Pattern Recognition (2019)","DOI":"10.1109\/CVPR.2019.00809"},{"key":"10_CR92","doi-asserted-by":"crossref","unstructured":"Sakoe, H., Chiba, S.: Dynamic programming algorithm optimization for spoken word recognition. IEEE Trans. Acoust. Speech Signal Process. 26 (1978)","DOI":"10.1109\/TASSP.1978.1163055"},{"key":"10_CR93","doi-asserted-by":"crossref","unstructured":"Sarfraz, S., Murray, N., Sharma, V., Diba, A., Van\u00a0Gool, L., Stiefelhagen, R.: Temporally-weighted hierarchical clustering for unsupervised action segmentation. In: IEEE\/CVF Conference on Computer Vision and Pattern Recognition (2021)","DOI":"10.1109\/CVPR46437.2021.01107"},{"key":"10_CR94","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"154","DOI":"10.1007\/978-3-030-58517-4_10","volume-title":"Computer Vision \u2013 ECCV 2020","author":"F Sener","year":"2020","unstructured":"Sener, F., Singhania, D., Yao, A.: Temporal aggregate representations for long-range video understanding. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12361, pp. 154\u2013171. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58517-4_10"},{"key":"10_CR95","doi-asserted-by":"crossref","unstructured":"Sener, F., Yao, A.: Unsupervised learning and segmentation of complex activities from video. In: IEEE Conference on Computer Vision and Pattern Recognition (2018)","DOI":"10.1109\/CVPR.2018.00873"},{"key":"10_CR96","unstructured":"Sener, O., Savarese, S.: Active learning for convolutional neural networks: a core-set approach. In: International Conference on Learning Representations (2018). https:\/\/openreview.net\/forum?id=H1aIuk-RW"},{"key":"10_CR97","doi-asserted-by":"crossref","unstructured":"Sener, O., Zamir, A.R., Savarese, S., Saxena, A.: Unsupervised semantic parsing of video collections. In: IEEE International Conference on Computer Vision (2015)","DOI":"10.1109\/ICCV.2015.509"},{"key":"10_CR98","doi-asserted-by":"crossref","unstructured":"Shah, A., Lundell, B., Sawhney, H., Chellappa, R.: STEPs: self-supervised key step extraction and localization from unlabeled procedural videos. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 10375\u201310387 (2023)","DOI":"10.1109\/ICCV51070.2023.00952"},{"key":"10_CR99","doi-asserted-by":"crossref","unstructured":"Shen, Y., Elhamifar, E.: Semi-weakly-supervised learning of complex actions from instructional task videos. In: IEEE Conference on Computer Vision and Pattern Recognition (2022)","DOI":"10.1109\/CVPR52688.2022.00334"},{"key":"10_CR100","doi-asserted-by":"crossref","unstructured":"Shen, Y., Elhamifar, E.: Progress-aware online action segmentation for egocentric procedural task videos. In: IEEE Conference on Computer Vision and Pattern Recognition (2024)","DOI":"10.1109\/CVPR52733.2024.01722"},{"key":"10_CR101","doi-asserted-by":"crossref","unstructured":"Shen, Y., Wang, L., Elhamifar, E.: Learning to segment actions from visual and language instructions via differentiable weak sequence alignment. In: IEEE Conference on Computer Vision and Pattern Recognition (2021)","DOI":"10.1109\/CVPR46437.2021.01002"},{"key":"10_CR102","doi-asserted-by":"crossref","unstructured":"Singh, B., Marks, T.K., Jones, M., Tuzel, O., Shao, M.: A multi-stream bi-directional recurrent neural network for finegrained action detection. In: IEEE Conference on Computer Vision and Pattern Recognition (2016)","DOI":"10.1109\/CVPR.2016.216"},{"key":"10_CR103","unstructured":"Singhania, D., Rahaman, R., Yao, A.: Coarse to fine multi-resolution temporal convolutional network. CoRR abs\/2105.10859 (2021). https:\/\/arxiv.org\/abs\/2105.10859"},{"key":"10_CR104","doi-asserted-by":"crossref","unstructured":"Singhania, D., Rahaman, R., Yao, A.: Iterative contrast-classify for semi-supervised temporal action segmentation. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a036, pp. 2262\u20132270 (2022)","DOI":"10.1609\/aaai.v36i2.20124"},{"key":"10_CR105","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"282","DOI":"10.1007\/978-3-030-92659-5_18","volume-title":"Pattern Recognition","author":"Y Souri","year":"2021","unstructured":"Souri, Y., Farha, Y.A., Despinoy, F., Francesca, G., Gall, J.: FIFA: fast inference approximation for\u00a0action segmentation. In: Bauckhage, C., Gall, J., Schwing, A. (eds.) DAGM GCPR 2021. LNCS, vol. 13024, pp. 282\u2013296. Springer, Cham (2021). https:\/\/doi.org\/10.1007\/978-3-030-92659-5_18"},{"issue":"10","key":"10_CR106","doi-asserted-by":"publisher","first-page":"6196","DOI":"10.1109\/TPAMI.2021.3089127","volume":"44","author":"Y Souri","year":"2021","unstructured":"Souri, Y., Fayyaz, M., Minciullo, L., Francesca, G., Gall, J.: Fast weakly supervised action segmentation using mutual consistency. IEEE Trans. Pattern Anal. Mach. Intell. 44(10), 6196\u20136208 (2021)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10_CR107","doi-asserted-by":"crossref","unstructured":"Stein, S., McKenna, S.J.: Combining embedded accelerometers with computer vision for recognizing food preparation activities. In: Proceedings of the 2013 ACM International Joint Conference on Pervasive and Ubiquitous Computing (2013)","DOI":"10.1145\/2493432.2493482"},{"key":"10_CR108","doi-asserted-by":"crossref","unstructured":"Su, B., Hua, G.: Order-preserving wasserstein distance for sequence matching. In: IEEE Conference on Computer Vision and Pattern Recognition (2017)","DOI":"10.1109\/CVPR.2017.310"},{"key":"10_CR109","doi-asserted-by":"crossref","unstructured":"Tang, Y., Zhang, X., Ma, L., Wang, J., Chen, S., Jiang, Y.G.: Non-local netvlad encoding for video classification. In: Proceedings of the European Conference on Computer Vision (ECCV) Workshops (2018)","DOI":"10.1007\/978-3-030-11018-5_20"},{"issue":"2","key":"10_CR110","doi-asserted-by":"publisher","first-page":"615","DOI":"10.1007\/s00530-022-00998-4","volume":"29","author":"X Tian","year":"2023","unstructured":"Tian, X., Jin, Y., Tang, X.: Local-global transformer neural network for temporal action segmentation. Multimed. Syst. 29(2), 615\u2013626 (2023)","journal-title":"Multimed. Syst."},{"key":"10_CR111","unstructured":"Vaswani, A., et al.: Attention is all you need. Neural Inf. Process. Syst. (2017)"},{"key":"10_CR112","doi-asserted-by":"crossref","unstructured":"VidalMata, R.G., Scheirer, W.J., Kukleva, A., Cox, D., Kuehne, H.: Joint visual-temporal embedding for unsupervised learning of actions in untrimmed sequences. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 1238\u20131247 (2021)","DOI":"10.1109\/WACV48630.2021.00128"},{"key":"10_CR113","doi-asserted-by":"publisher","first-page":"63","DOI":"10.1016\/j.neucom.2020.03.066","volume":"407","author":"D Wang","year":"2020","unstructured":"Wang, D., Yuan, Y., Wang, Q.: Gated forward refinement network for action segmentation. Neurocomputing 407, 63\u201371 (2020)","journal-title":"Neurocomputing"},{"key":"10_CR114","doi-asserted-by":"crossref","unstructured":"Wang, J., Du, Z., Li, A., Wang, Y.: Atrous temporal convolutional network for video action segmentation. In: 2019 IEEE International Conference on Image Processing (ICIP), pp. 1585\u20131589. IEEE (2019)","DOI":"10.1109\/ICIP.2019.8803088"},{"key":"10_CR115","doi-asserted-by":"crossref","unstructured":"Wang, J., Wang, Z., Zhuang, S., Hao, Y., Wang, H.: Cross-enhancement transformer for action segmentation. Multimed. Tools Appl. 1\u201314 (2023)","DOI":"10.1007\/s11042-023-16041-1"},{"issue":"12","key":"10_CR116","doi-asserted-by":"publisher","first-page":"2591","DOI":"10.1109\/TCSVT.2016.2589879","volume":"27","author":"K Wang","year":"2016","unstructured":"Wang, K., Zhang, D., Li, Y., Zhang, R., Lin, L.: Cost-effective active learning for deep image classification. IEEE Trans. Circuits Syst. Video Technol. 27(12), 2591\u20132600 (2016)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10_CR117","doi-asserted-by":"crossref","unstructured":"Wang, X., Zhang, S., Qing, Z., Shao, Y., Gao, C., Sang, N.: Self-supervised learning for semi-supervised temporal action proposal. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1905\u20131914 (2021)","DOI":"10.1109\/CVPR46437.2021.00194"},{"key":"10_CR118","doi-asserted-by":"crossref","unstructured":"Wang, Z., et al.: SSCAP: self-supervised co-occurrence action parsing for unsupervised temporal action segmentation. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 1819\u20131828 (2022)","DOI":"10.1109\/WACV51458.2022.00025"},{"key":"10_CR119","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"34","DOI":"10.1007\/978-3-030-58595-2_3","volume-title":"Computer Vision \u2013 ECCV 2020","author":"Z Wang","year":"2020","unstructured":"Wang, Z., Gao, Z., Wang, L., Li, Z., Wu, G.: Boundary-aware cascade networks for temporal action segmentation. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12370, pp. 34\u201351. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58595-2_3"},{"key":"10_CR120","doi-asserted-by":"crossref","unstructured":"Wanyan, Y., Yang, X., Chen, C., Xu, C.: Active exploration of multimodal complementarity for few-shot action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 6492\u20136502 (2023)","DOI":"10.1109\/CVPR52729.2023.00628"},{"key":"10_CR121","unstructured":"Wei, K., Iyer, R., Bilmes, J.: Submodularity in data subset selection and active learning. In: International Conference on Machine Learning, pp. 1954\u20131963. PMLR (2015)"},{"key":"10_CR122","doi-asserted-by":"crossref","unstructured":"Xie, Y., Lu, H., Yan, J., Yang, X., Tomizuka, M., Zhan, W.: Active finetuning: Exploiting annotation budget in the pretraining-finetuning paradigm. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 23715\u201323724 (2023)","DOI":"10.1109\/CVPR52729.2023.02271"},{"key":"10_CR123","unstructured":"Xu, C., Elhamifar, E.: Deep supervised summarization: algorithm and application to learning instructions. Neural Inf. Process. Syst. (2019)"},{"key":"10_CR124","unstructured":"Yang, Y., Ma, J., Huang, S., Chen, L., Lin, X., Han, G., Chang, S.F.: TempCLR: temporal alignment representation with contrastive learning. arXiv preprint arXiv:2212.13738 (2022)"},{"key":"10_CR125","unstructured":"Yi, F., Wen, H., Jiang, T.: ASFormer: transformer for action segmentation. In: The British Machine Vision Conference (BMVC) (2021)"},{"key":"10_CR126","doi-asserted-by":"crossref","unstructured":"Yoo, D., Kweon, I.S.: Learning loss for active learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 93\u2013102 (2019)","DOI":"10.1109\/CVPR.2019.00018"},{"key":"10_CR127","doi-asserted-by":"crossref","unstructured":"Yuan, T., et al.: Multiple instance active learning for object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5330\u20135339 (2021)","DOI":"10.1109\/CVPR46437.2021.00529"},{"key":"10_CR128","unstructured":"Zhang, J., Tsai, P.H., Tsai, M.H.: Semantic2Graph: graph-based multi-modal feature fusion for action segmentation in videos (2022)"},{"key":"10_CR129","doi-asserted-by":"crossref","unstructured":"Zhang, K., Chao, W.L., Sha, F., Grauman, K.: Summary transfer: exemplar-based subset selection for video summarization. IEEE Conference on Computer Vision and Pattern Recognition (2016)","DOI":"10.1109\/CVPR.2016.120"},{"key":"10_CR130","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Ren, K., Zhang, C., Yan, T.: SG-TCN: semantic guidance temporal convolutional network for action segmentation. In: 2022 International Joint Conference on Neural Networks (IJCNN), pp.\u00a01\u20138. IEEE (2022)","DOI":"10.1109\/IJCNN55064.2022.9891932"},{"key":"10_CR131","unstructured":"Zhao, G., Dougherty, E., Yoon, B.J., Alexander, F., Qian, X.: Uncertainty-aware active learning for optimal Bayesian classifier. In: International Conference on Learning Representations (ICLR 2021) (2021)"},{"key":"10_CR132","unstructured":"Zhou, F., Torre, F.: Canonical time warping for alignment of human behavior. Adv. Neural Inf. Process. Syst. 22 (2009)"},{"key":"10_CR133","unstructured":"Zhu, J.J., Bento, J.: Generative adversarial active learning. arXiv preprint arXiv:1702.07956 (2017)"},{"key":"10_CR134","unstructured":"Zhu, X., Su, W., Lu, L., Li, B., Wang, X., Dai, J.: Deformable DETR: deformable transformers for end-to-end object detection. arXiv preprint arXiv:2010.04159 (2020)"},{"key":"10_CR135","doi-asserted-by":"crossref","unstructured":"Zhukov, D., Alayrac, J.B., Cinbis, R.G., Fouhey, D., Laptev, I., Sivic, J.: Cross-task weakly supervised learning from instructional videos. In: IEEE Conference on Computer Vision and Pattern Recognition (2019)","DOI":"10.1109\/CVPR.2019.00365"},{"key":"10_CR136","doi-asserted-by":"crossref","unstructured":"Zolfaghari\u00a0Bengar, J., et al.: Temporal coherence for active learning in videos. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision Workshops (2019)","DOI":"10.1109\/ICCVW.2019.00120"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72970-6_10","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,22]],"date-time":"2024-11-22T11:15:21Z","timestamp":1732274121000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72970-6_10"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,23]]},"ISBN":["9783031729690","9783031729706"],"references-count":136,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72970-6_10","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,23]]},"assertion":[{"value":"23 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}