{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T17:23:26Z","timestamp":1777569806796,"version":"3.51.4"},"publisher-location":"Cham","reference-count":81,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031729720","type":"print"},{"value":"9783031729737","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72973-7_9","type":"book-chapter","created":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T14:03:04Z","timestamp":1730383384000},"page":"141-161","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Learning by\u00a0Aligning 2D Skeleton Sequences and\u00a0Multi-modality Fusion"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1396-6544","authenticated-orcid":false,"given":"Quoc-Huy","family":"Tran","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-6064-8929","authenticated-orcid":false,"given":"Muhammad","family":"Ahmed","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-5286-3082","authenticated-orcid":false,"given":"Murad","family":"Popattia","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-2925-1984","authenticated-orcid":false,"given":"M. Hassan","family":"Ahmed","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1886-2402","authenticated-orcid":false,"given":"Andrey","family":"Konin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8221-2637","authenticated-orcid":false,"given":"M. Zeeshan","family":"Zia","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,11,1]]},"reference":[{"key":"9_CR1","unstructured":"Ahsan, U., Sun, C., Essa, I.: Discrimnet: Semi-supervised action recognition from videos using generative adversarial networks. arXiv preprint (2018)"},{"key":"9_CR2","doi-asserted-by":"crossref","unstructured":"Arnab, A., Dehghani, M., Heigold, G., Sun, C., Lu\u010di\u0107, M., Schmid, C.: Vivit: video vision transformer. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6836\u20136846 (2021)a","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"9_CR3","doi-asserted-by":"crossref","unstructured":"Asghari-Esfeden, S., Sznaier, M., Camps, O.: Dynamic motion representation for human action recognition. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 557\u2013566 (2020)","DOI":"10.1109\/WACV45572.2020.9093500"},{"key":"9_CR4","doi-asserted-by":"crossref","unstructured":"Ben-Shabat, Y., et al.: The ikea asm dataset: Understanding people assembling furniture through actions, objects, and pose. In: arXiv preprint (2020)","DOI":"10.1109\/WACV48630.2021.00089"},{"key":"9_CR5","doi-asserted-by":"crossref","unstructured":"Benaim, S., et al.: Speednet: learning the speediness in videos. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9922\u20139931 (2020)","DOI":"10.1109\/CVPR42600.2020.00994"},{"key":"9_CR6","doi-asserted-by":"crossref","unstructured":"Caetano, C., Sena, J., Br\u00e9mond, F., Dos\u00a0Santos, J.A., Schwartz, W.R.: Skelemotion: A new representation of skeleton joint sequences based on motion information for 3d action recognition. In: 2019 16th IEEE International Conference on Advanced Video and Signal Based Surveillance (AVSS), pp.\u00a01\u20138. IEEE (2019)","DOI":"10.1109\/AVSS.2019.8909840"},{"key":"9_CR7","doi-asserted-by":"crossref","unstructured":"Cai, J., Jiang, N., Han, X., Jia, K., Lu, J.: Jolo-gcn: mining joint-centered light-weight information for skeleton-based action recognition. In: Proceedings of the IEEE\/CVF Winter Conference on Applications Of Computer Vision, pp. 2735\u20132744 (2021)","DOI":"10.1109\/WACV48630.2021.00278"},{"issue":"1","key":"9_CR8","doi-asserted-by":"publisher","first-page":"172","DOI":"10.1109\/TPAMI.2019.2929257","volume":"43","author":"Z Cao","year":"2019","unstructured":"Cao, Z., Hidalgo, G., Simon, T., Wei, S.E., Sheikh, Y.: Openpose: realtime multi-person 2d pose estimation using part affinity fields. IEEE Trans. Pattern Anal. Mach. Intell. 43(1), 172\u2013186 (2019)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"9_CR9","doi-asserted-by":"crossref","unstructured":"Carlucci, F.M., D\u2019Innocente, A., Bucci, S., Caputo, B., Tommasi, T.: Domain generalization by solving jigsaw puzzles. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2229\u20132238 (2019)","DOI":"10.1109\/CVPR.2019.00233"},{"key":"9_CR10","doi-asserted-by":"crossref","unstructured":"Caron, M., Bojanowski, P., Joulin, A., Douze, M.: Deep clustering for unsupervised learning of visual features. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 132\u2013149 (2018)","DOI":"10.1007\/978-3-030-01264-9_9"},{"key":"9_CR11","doi-asserted-by":"crossref","unstructured":"Caron, M., Bojanowski, P., Mairal, J., Joulin, A.: Unsupervised pre-training of image features on non-curated data. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2959\u20132968 (2019)","DOI":"10.1109\/ICCV.2019.00305"},{"key":"9_CR12","first-page":"9912","volume":"33","author":"M Caron","year":"2020","unstructured":"Caron, M., Misra, I., Mairal, J., Goyal, P., Bojanowski, P., Joulin, A.: Unsupervised learning of visual features by contrasting cluster assignments. Adv. Neural. Inf. Process. Syst. 33, 9912\u20139924 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"9_CR13","unstructured":"Chen, T., Kornblith, S., Norouzi, M., Hinton, G.: A simple framework for contrastive learning of visual representations. In: International Conference on Machine Learning, pp. 1597\u20131607. PMLR (2020)"},{"key":"9_CR14","doi-asserted-by":"crossref","unstructured":"Chen, X., He, K.: Exploring simple siamese representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15750\u201315758 (2021)","DOI":"10.1109\/CVPR46437.2021.01549"},{"key":"9_CR15","doi-asserted-by":"crossref","unstructured":"Chen, Y., Zhang, Z., Yuan, C., Li, B., Deng, Y., Hu, W.: Channel-wise topology refinement graph convolution for skeleton-based action recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 13359\u201313368 (2021)","DOI":"10.1109\/ICCV48922.2021.01311"},{"key":"9_CR16","doi-asserted-by":"crossref","unstructured":"p Choutas, V., Weinzaepfel, P., Revaud, J., Schmid, C.: Potion: pose motion representation for action recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7024\u20137033 (2018)","DOI":"10.1109\/CVPR.2018.00734"},{"key":"9_CR17","unstructured":"Cuturi, M.: Sinkhorn distances: Lightspeed computation of optimal transport. In: Advances in Neural Information Processing Systems, vol. 26 (2013)"},{"key":"9_CR18","unstructured":"Cuturi, M., Blondel, M.: Soft-dtw: a differentiable loss function for time-series. In: International Conference on Machine Learning, pp. 894\u2013903. PMLR (2017)"},{"key":"9_CR19","doi-asserted-by":"publisher","first-page":"72","DOI":"10.1007\/978-3-030-58545-7_5","volume-title":"Computer Vision \u2013 ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part IX","author":"S Das","year":"2020","unstructured":"Das, S., Sharma, S., Dai, R., Br\u00e9mond, F., Thonnat, M.: VPN: learning video-pose embedding for activities of daily living. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) Computer Vision \u2013 ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part IX, pp. 72\u201390. Springer International Publishing, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58545-7_5"},{"key":"9_CR20","doi-asserted-by":"publisher","DOI":"10.1016\/j.cviu.2022.103406","volume":"219","author":"I Dave","year":"2022","unstructured":"Dave, I., Gupta, R., Rizve, M.N., Shah, M.: Tclr: temporal contrastive learning for video representation. Comput. Vis. Image Underst. 219, 103406 (2022)","journal-title":"Comput. Vis. Image Underst."},{"key":"9_CR21","doi-asserted-by":"crossref","unstructured":"Diba, A., Sharma, V., Gool, L.V., Stiefelhagen, R.: Dynamonet: Dynamic action and motion network. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 6192\u20136201 (2019)","DOI":"10.1109\/ICCV.2019.00629"},{"key":"9_CR22","doi-asserted-by":"crossref","unstructured":"Duan, H., Zhao, Y., Chen, K., Lin, D., Dai, B.: Revisiting skeleton-based action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2969\u20132978 (2022)","DOI":"10.1109\/CVPR52688.2022.00298"},{"key":"9_CR23","doi-asserted-by":"crossref","unstructured":"Dwibedi, D., Aytar, Y., Tompson, J., Sermanet, P., Zisserman, A.: Temporal cycle-consistency learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (June 2019)","DOI":"10.1109\/CVPR.2019.00190"},{"key":"9_CR24","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Fan, H., Xiong, B., Girshick, R., He, K.: A large-scale study on unsupervised spatiotemporal representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3299\u20133309 (2021)","DOI":"10.1109\/CVPR46437.2021.00331"},{"key":"9_CR25","doi-asserted-by":"crossref","unstructured":"Fernando, B., Bilen, H., Gavves, E., Gould, S.: Self-supervised video representation learning with odd-one-out networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3636\u20133645 (2017)","DOI":"10.1109\/CVPR.2017.607"},{"key":"9_CR26","unstructured":"Gidaris, S., Singh, P., Komodakis, N.: Unsupervised representation learning by predicting image rotations. In: International Conference on Learning Representations (2018). https:\/\/openreview.net\/forum?id=S1v4N2l0-"},{"key":"9_CR27","doi-asserted-by":"crossref","unstructured":"Goroshin, R., Bruna, J., Tompson, J., Eigen, D., LeCun, Y.: Unsupervised learning of spatiotemporally coherent metrics. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 4086\u20134093 (2015)","DOI":"10.1109\/ICCV.2015.465"},{"key":"9_CR28","first-page":"21271","volume":"33","author":"JB Grill","year":"2020","unstructured":"Grill, J.B., et al.: Bootstrap your own latent-a new approach to self-supervised learning. Adv. Neural. Inf. Process. Syst. 33, 21271\u201321284 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"issue":"7","key":"9_CR29","doi-asserted-by":"publisher","first-page":"2097","DOI":"10.1007\/s11263-021-01470-y","volume":"129","author":"P Gupta","year":"2021","unstructured":"Gupta, P., et al.: Quo vadis, skeleton action recognition? Int. J. Comput. Vision 129(7), 2097\u20132112 (2021)","journal-title":"Int. J. Comput. Vision"},{"key":"9_CR30","doi-asserted-by":"crossref","unstructured":"Haresh, S., et al.: Learning by aligning videos in time. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5548\u20135558 (2021)","DOI":"10.1109\/CVPR46437.2021.00550"},{"key":"9_CR31","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"9_CR32","doi-asserted-by":"crossref","unstructured":"Hernandez\u00a0Ruiz, A., Porzi, L., Rota\u00a0Bul\u00f2, S., Moreno-Noguer, F.: 3d cnns on distance matrices for human action recognition. In: Proceedings of the 25th ACM International Conference on Multimedia, pp. 1087\u20131095 (2017)","DOI":"10.1145\/3123266.3123299"},{"key":"9_CR33","doi-asserted-by":"crossref","unstructured":"Hu, K., Shao, J., Liu, Y., Raj, B., Savvides, M., Shen, Z.: Contrast and order representations for video self-supervised learning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7939\u20137949 (2021)","DOI":"10.1109\/ICCV48922.2021.00784"},{"key":"9_CR34","doi-asserted-by":"crossref","unstructured":"Hyder, S.W., et al.: Action segmentation using 2d skeleton heatmaps and multi-modality fusion. In: Proceedings of the IEEE International Conference on Robotics and Automation (ICRA) (2024)","DOI":"10.1109\/ICRA57147.2024.10610644"},{"key":"9_CR35","doi-asserted-by":"crossref","unstructured":"Jenni, S., Jin, H., Favaro, P.: Steering self-supervised feature learning beyond local pixel statistics. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6408\u20136417 (2020)","DOI":"10.1109\/CVPR42600.2020.00644"},{"key":"9_CR36","doi-asserted-by":"publisher","unstructured":"Kay, W., et al.: The kinetics human action video dataset. arXiv (2017). https:\/\/doi.org\/10.48550\/ARXIV.1705.06950, https:\/\/arxiv.org\/abs\/1705.06950","DOI":"10.48550\/ARXIV.1705.06950"},{"key":"9_CR37","doi-asserted-by":"crossref","unstructured":"Ke, Q., Bennamoun, M., An, S., Sohel, F., Boussaid, F.: A new representation of skeleton sequences for 3d action recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3288\u20133297 (2017)","DOI":"10.1109\/CVPR.2017.486"},{"key":"9_CR38","doi-asserted-by":"crossref","unstructured":"Khan, H., et al.: Timestamp-supervised action segmentation with graph convolutional networks. In: 2022 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS), pp. 10619\u201310626. IEEE (2022)","DOI":"10.1109\/IROS47612.2022.9981351"},{"key":"9_CR39","doi-asserted-by":"crossref","unstructured":"Kim, D., Cho, D., Kweon, I.S.: Self-supervised video representation learning with space-time cubic puzzles. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a033, pp. 8545\u20138552 (2019)","DOI":"10.1609\/aaai.v33i01.33018545"},{"key":"9_CR40","unstructured":"Kingma, D.P., Ba, J.: Adam: A method for stochastic optimization. arXiv preprint (2014)"},{"key":"9_CR41","doi-asserted-by":"crossref","unstructured":"Kumar, S., Haresh, S., Ahmed, A., Konin, A., Zia, M.Z., Tran, Q.H.: Unsupervised action segmentation by joint representation learning and online clustering. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 20174\u201320185 (2022)","DOI":"10.1109\/CVPR52688.2022.01954"},{"key":"9_CR42","doi-asserted-by":"crossref","unstructured":"Kwon, T., Tekin, B., St\u00fchmer, J., Bogo, F., Pollefeys, M.: H2o: Two hands manipulating objects for first person interaction recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 10138\u201310148 (October 2021)","DOI":"10.1109\/ICCV48922.2021.00998"},{"key":"9_CR43","doi-asserted-by":"crossref","unstructured":"Kwon, T., Tekin, B., Tang, S., Pollefeys, M.: Context-aware sequence alignment using 4d skeletal augmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8172\u20138182 (2022)","DOI":"10.1109\/CVPR52688.2022.00800"},{"key":"9_CR44","doi-asserted-by":"publisher","unstructured":"Larsson, G., Maire, M., Shakhnarovich, G.: Learning representations for automatic colorization. In: European Conference on Computer Vision, pp. 577\u2013593. Springer (2016). https:\/\/doi.org\/10.1007\/978-3-319-46493-0_35","DOI":"10.1007\/978-3-319-46493-0_35"},{"key":"9_CR45","doi-asserted-by":"crossref","unstructured":"Larsson, G., Maire, M., Shakhnarovich, G.: Colorization as a proxy task for visual understanding. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6874\u20136883 (2017)","DOI":"10.1109\/CVPR.2017.96"},{"key":"9_CR46","doi-asserted-by":"crossref","unstructured":"Lee, H.Y., Huang, J.B., Singh, M., Yang, M.H.: Unsupervised representation learning by sorting sequences. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 667\u2013676 (2017)","DOI":"10.1109\/ICCV.2017.79"},{"key":"9_CR47","doi-asserted-by":"crossref","unstructured":"Li, C., Zhong, Q., Xie, D., Pu, S.: Co-occurrence feature learning from skeleton data for action recognition and detection with hierarchical aggregation. arXiv preprint (2018)","DOI":"10.24963\/ijcai.2018\/109"},{"key":"9_CR48","doi-asserted-by":"crossref","unstructured":"Lin, L., Song, S., Yang, W., Liu, J.: Ms2l: multi-task self-supervised learning for skeleton based action recognition. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 2490\u20132498 (2020)","DOI":"10.1145\/3394171.3413548"},{"key":"9_CR49","doi-asserted-by":"crossref","unstructured":"Lin, Z., Zhang, W., Deng, X., Ma, C., Wang, H.: Image-based pose representation for action recognition and hand gesture recognition. In: 2020 15th IEEE International Conference on Automatic Face and Gesture Recognition (FG 2020), pp. 532\u2013539. IEEE (2020)","DOI":"10.1109\/FG47880.2020.00066"},{"key":"9_CR50","doi-asserted-by":"crossref","unstructured":"Liu, W., Tekin, B., Coskun, H., Vineet, V., Fua, P., Pollefeys, M.: Learning to align sequential actions in the wild. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2181\u20132191 (2022)","DOI":"10.1109\/CVPR52688.2022.00222"},{"key":"9_CR51","doi-asserted-by":"crossref","unstructured":"Liu, X., Van De\u00a0Weijer, J., Bagdanov, A.D.: Leveraging unlabeled data for crowd counting by learning to rank. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7661\u20137669 (2018)","DOI":"10.1109\/CVPR.2018.00799"},{"key":"9_CR52","doi-asserted-by":"crossref","unstructured":"Luvizon, D.C., Picard, D., Tabia, H.: 2d\/3d pose estimation and action recognition using multitask deep learning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5137\u20135146 (2018)","DOI":"10.1109\/CVPR.2018.00539"},{"key":"9_CR53","doi-asserted-by":"publisher","first-page":"527","DOI":"10.1007\/978-3-319-46448-0_32","volume-title":"Computer Vision \u2013 ECCV 2016: 14th European Conference, Amsterdam, The Netherlands, October 11\u201314, 2016, Proceedings, Part I","author":"I Misra","year":"2016","unstructured":"Misra, I., Zitnick, C.L., Hebert, M.: Shuffle and learn: unsupervised learning using temporal order verification. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) Computer Vision \u2013 ECCV 2016: 14th European Conference, Amsterdam, The Netherlands, October 11\u201314, 2016, Proceedings, Part I, pp. 527\u2013544. Springer International Publishing, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46448-0_32"},{"key":"9_CR54","doi-asserted-by":"crossref","unstructured":"Mobahi, H., Collobert, R., Weston, J.: Deep learning from temporal coherence in video. In: Proceedings of the 26th Annual International Conference on Machine Learning, pp. 737\u2013744 (2009)","DOI":"10.1145\/1553374.1553469"},{"key":"9_CR55","doi-asserted-by":"crossref","unstructured":"Noroozi, M., Pirsiavash, H., Favaro, P.: Representation learning by learning to count. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 5898\u20135906 (2017)","DOI":"10.1109\/ICCV.2017.628"},{"key":"9_CR56","unstructured":"Paszke, A., et al.: Automatic differentiation in pytorch (2017)"},{"key":"9_CR57","doi-asserted-by":"crossref","unstructured":"Pickup, L.C., et al.: Seeing the arrow of time. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2035\u20132042 (2014)","DOI":"10.1109\/CVPR.2014.262"},{"key":"9_CR58","doi-asserted-by":"crossref","unstructured":"Qian, R., et al.: Spatiotemporal contrastive video representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6964\u20136974 (2021)","DOI":"10.1109\/CVPR46437.2021.00689"},{"key":"9_CR59","doi-asserted-by":"publisher","unstructured":"Sermanet, P., et al.: Time-contrastive networks: self-supervised learning from video. In: 2018 IEEE International Conference on Robotics and Automation (ICRA), pp. 1134\u20131141 (2018). https:\/\/doi.org\/10.1109\/ICRA.2018.8462891","DOI":"10.1109\/ICRA.2018.8462891"},{"key":"9_CR60","doi-asserted-by":"crossref","unstructured":"Shah, A., Lundell, B., Sawhney, H., Chellappa, R.: Steps: self-supervised key step extraction and localization from unlabeled procedural videos. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 10375\u201310387 (2023)","DOI":"10.1109\/ICCV51070.2023.00952"},{"key":"9_CR61","doi-asserted-by":"publisher","first-page":"35","DOI":"10.1007\/978-3-030-58571-6_3","volume-title":"Computer Vision \u2013 ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part VII","author":"C Si","year":"2020","unstructured":"Si, C., Nie, X., Wang, W., Wang, L., Tan, T., Feng, J.: Adversarial self-supervised learning for semi-supervised 3d action recognition. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) Computer Vision \u2013 ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part VII, pp. 35\u201351. Springer International Publishing, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58571-6_3"},{"key":"9_CR62","doi-asserted-by":"crossref","unstructured":"Song, Y.F., Zhang, Z., Shan, C., Wang, L.: Stronger, faster and more explainable: a graph convolutional baseline for skeleton-based action recognition. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 1625\u20131633 (2020)","DOI":"10.1145\/3394171.3413802"},{"issue":"2","key":"9_CR63","doi-asserted-by":"publisher","first-page":"1474","DOI":"10.1109\/TPAMI.2022.3157033","volume":"45","author":"YF Song","year":"2022","unstructured":"Song, Y.F., Zhang, Z., Shan, C., Wang, L.: Constructing stronger and faster baselines for skeleton-based action recognition. IEEE Trans. Pattern Anal. Mach. Intell. 45(2), 1474\u20131488 (2022)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"9_CR64","unstructured":"Srivastava, N., Mansimov, E., Salakhudinov, R.: Unsupervised learning of video representations using lstms. In: International conference on machine learning, pp. 843\u2013852 (2015)"},{"key":"9_CR65","doi-asserted-by":"crossref","unstructured":"Su, K., Liu, X., Shlizerman, E.: Predict & cluster: Unsupervised skeleton based action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9631\u20139640 (2020)","DOI":"10.1109\/CVPR42600.2020.00965"},{"key":"9_CR66","doi-asserted-by":"crossref","unstructured":"Su, Y., Lin, G., Wu, Q.: Self-supervised 3d skeleton action representation learning with motion consistency and continuity. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 13328\u201313338 (2021)","DOI":"10.1109\/ICCV48922.2021.01308"},{"key":"9_CR67","doi-asserted-by":"crossref","unstructured":"Sun, J., Shen, Z., Wang, Y., Bao, H., Zhou, X.: Loftr: detector-free local feature matching with transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8922\u20138931 (2021)","DOI":"10.1109\/CVPR46437.2021.00881"},{"key":"9_CR68","doi-asserted-by":"crossref","unstructured":"Tran, Q.H., et al.: Permutation-aware activity segmentation via unsupervised frame-to-segment alignment. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 6426\u20136436 (2024)","DOI":"10.1109\/WACV57701.2024.00630"},{"key":"9_CR69","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Advances in Neural Information Processing Systems, vol. 30 (2017)"},{"key":"9_CR70","unstructured":"Vondrick, C., Pirsiavash, H., Torralba, A.: Generating videos with scene dynamics. In: Advances in Neural Information Processing Systems, pp. 613\u2013621 (2016)"},{"key":"9_CR71","doi-asserted-by":"publisher","first-page":"504","DOI":"10.1007\/978-3-030-58520-4_30","volume-title":"Computer Vision \u2013 ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XVII","author":"J Wang","year":"2020","unstructured":"Wang, J., Jiao, J., Liu, Y.-H.: Self-supervised video representation learning by pace prediction. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) Computer Vision \u2013 ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XVII, pp. 504\u2013521. Springer International Publishing, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58520-4_30"},{"key":"9_CR72","doi-asserted-by":"crossref","unstructured":"Wei, D., Lim, J.J., Zisserman, A., Freeman, W.T.: Learning and using the arrow of time. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 8052\u20138060 (2018)","DOI":"10.1109\/CVPR.2018.00840"},{"key":"9_CR73","doi-asserted-by":"crossref","unstructured":"Xu, D., Xiao, J., Zhao, Z., Shao, J., Xie, D., Zhuang, Y.: Self-supervised spatiotemporal learning via video clip order prediction. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 10334\u201310343 (2019)","DOI":"10.1109\/CVPR.2019.01058"},{"key":"9_CR74","doi-asserted-by":"crossref","unstructured":"Yan, A., Wang, Y., Li, Z., Qiao, Y.: Pa3d: pose-action 3d machine for video recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7922\u20137931 (2019)","DOI":"10.1109\/CVPR.2019.00811"},{"key":"9_CR75","doi-asserted-by":"crossref","unstructured":"Yan, S., Xiong, Y., Lin, D.: Spatial temporal graph convolutional networks for skeleton-based action recognition. In: Thirty-Second AAAI Conference On Artificial Intelligence (2018)","DOI":"10.1609\/aaai.v32i1.12328"},{"key":"9_CR76","doi-asserted-by":"crossref","unstructured":"Yao, Y., Liu, C., Luo, D., Zhou, Y., Ye, Q.: Video playback rate perception for self-supervised spatio-temporal representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6548\u20136557 (2020)","DOI":"10.1109\/CVPR42600.2020.00658"},{"key":"9_CR77","doi-asserted-by":"crossref","unstructured":"Zhang, H., Liu, D., Zheng, Q., Su, B.: Modeling video as stochastic processes for fine-grained video representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2225\u20132234 (2023)","DOI":"10.1109\/CVPR52729.2023.00221"},{"key":"9_CR78","doi-asserted-by":"publisher","unstructured":"Zhang, W., Zhu, M., Derpanis, K.G.: From actemes to action: a strongly-supervised representation for detailed action understanding. In: 2013 IEEE International Conference on Computer Vision, pp. 2248\u20132255 (2013). https:\/\/doi.org\/10.1109\/ICCV.2013.280","DOI":"10.1109\/ICCV.2013.280"},{"key":"9_CR79","doi-asserted-by":"crossref","unstructured":"Zheng, N., Wen, J., Liu, R., Long, L., Dai, J., Gong, Z.: Unsupervised representation learning with long-term dynamics for skeleton based action recognition. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a032 (2018)","DOI":"10.1609\/aaai.v32i1.11853"},{"key":"9_CR80","doi-asserted-by":"crossref","unstructured":"Zhu, D., Zhang, Z., Cui, P., Zhu, W.: Robust graph convolutional networks against adversarial attacks. In: Proceedings of the 25th ACM SIGKDD International Conference on Knowledge Discovery and data mining, pp. 1399\u20131407 (2019)","DOI":"10.1145\/3292500.3330851"},{"key":"9_CR81","unstructured":"Zou, W.Y., Ng, A.Y., Yu, K.: Unsupervised learning of visual invariance with temporal coherence. In: NIPS 2011 Workshop on Deep Learning and Unsupervised Feature Learning, vol.\u00a03 (2011)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72973-7_9","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,1,17]],"date-time":"2025-01-17T12:26:14Z","timestamp":1737116774000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72973-7_9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,1]]},"ISBN":["9783031729720","9783031729737"],"references-count":81,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72973-7_9","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,1]]},"assertion":[{"value":"1 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}