{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T17:48:49Z","timestamp":1777657729537,"version":"3.51.4"},"publisher-location":"Cham","reference-count":45,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031731150","type":"print"},{"value":"9783031731167","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T00:00:00Z","timestamp":1730332800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T00:00:00Z","timestamp":1730332800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-73116-7_24","type":"book-chapter","created":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T15:15:38Z","timestamp":1730301338000},"page":"414-431","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Masked Motion Prediction with\u00a0Semantic Contrast for\u00a0Point Cloud Sequence Learning"],"prefix":"10.1007","author":[{"given":"Yuehui","family":"Han","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Can","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rui","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jianjun","family":"Qian","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jin","family":"Xie","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,10,31]]},"reference":[{"key":"24_CR1","doi-asserted-by":"crossref","unstructured":"Chen, A., et al.: PiMAE: point cloud and image interactive masked autoencoders for 3D object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5291\u20135301 (2023)","DOI":"10.1109\/CVPR52729.2023.00512"},{"key":"24_CR2","unstructured":"Chen, T., Kornblith, S., Norouzi, M., Hinton, G.: A simple framework for contrastive learning of visual representations. In: International Conference on Machine Learning, pp. 1597\u20131607. PMLR (2020)"},{"key":"24_CR3","unstructured":"Chen, Z., Mao, J., Wu, J., Wong, K.Y.K., Tenenbaum, J.B., Gan, C.: Grounding physical concepts of objects and events through dynamic visual reasoning. arXiv preprint arXiv:2103.16564 (2021)"},{"key":"24_CR4","unstructured":"Chollet, F.: Deep learning with Python. Simon and Schuster (2021)"},{"key":"24_CR5","unstructured":"De\u00a0Smedt, Q., Wannous, H., Vandeborre, J.P., Guerry, J., Le\u00a0Saux, B., Filliat, D.: Shrec\u201917 track: 3D hand gesture recognition using a depth and skeletal dataset. In: 3DOR-10th Eurographics Workshop on 3D Object Retrieval, pp.\u00a01\u20136 (2017)"},{"key":"24_CR6","doi-asserted-by":"crossref","unstructured":"Fan, H., Su, H., Guibas, L.J.: A point set generation network for 3D object reconstruction from a single image. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 605\u2013613 (2017)","DOI":"10.1109\/CVPR.2017.264"},{"key":"24_CR7","doi-asserted-by":"crossref","unstructured":"Fan, H., Yang, Y., Kankanhalli, M.: Point 4D transformer networks for spatio-temporal modeling in point cloud videos. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14204\u201314213 (2021)","DOI":"10.1109\/CVPR46437.2021.01398"},{"issue":"2","key":"24_CR8","doi-asserted-by":"publisher","first-page":"2181","DOI":"10.1109\/TPAMI.2022.3161735","volume":"45","author":"H Fan","year":"2022","unstructured":"Fan, H., Yang, Y., Kankanhalli, M.: Point spatio-temporal transformer networks for point cloud video modeling. IEEE Trans. Pattern Anal. Mach. Intell. 45(2), 2181\u20132192 (2022)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"24_CR9","unstructured":"Fan, H., Yu, X., Ding, Y., Yang, Y., Kankanhalli, M.: PSTNet: point spatio-temporal convolution on point cloud sequences. arXiv preprint arXiv:2205.13713 (2022)"},{"issue":"12","key":"24_CR10","doi-asserted-by":"publisher","first-page":"9918","DOI":"10.1109\/TPAMI.2021.3135117","volume":"44","author":"H Fan","year":"2021","unstructured":"Fan, H., Yu, X., Yang, Y., Kankanhalli, M.: Deep hierarchical representation of point cloud videos via spatio-temporal decomposition. IEEE Trans. Pattern Anal. Mach. Intell. 44(12), 9918\u20139930 (2021)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"24_CR11","first-page":"35946","volume":"35","author":"C Feichtenhofer","year":"2022","unstructured":"Feichtenhofer, C., Li, Y., He, K., et al.: Masked autoencoders as spatiotemporal learners. Adv. Neural. Inf. Process. Syst. 35, 35946\u201335958 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"issue":"5","key":"24_CR12","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3645095","volume":"18","author":"Y Han","year":"2024","unstructured":"Han, Y.: Generation-based multi-view contrast for self-supervised graph representation learning. ACM Trans. Knowl. Discov. Data 18(5), 1\u201317 (2024)","journal-title":"ACM Trans. Knowl. Discov. Data"},{"key":"24_CR13","doi-asserted-by":"crossref","unstructured":"Han, Y., Chen, J., Qian, J., Xie, J.: Graph spectral perturbation for 3D point cloud contrastive learning. In: Proceedings of the 31st ACM International Conference on Multimedia, pp. 5389\u20135398 (2023)","DOI":"10.1145\/3581783.3612469"},{"key":"24_CR14","doi-asserted-by":"publisher","first-page":"91","DOI":"10.1007\/978-3-031-20056-4_6","volume-title":"Computer Vision \u2013 ECCV 2022: 17th European Conference, Tel Aviv, Israel, October 23\u201327, 2022, Proceedings, Part XXX","author":"Y Han","year":"2022","unstructured":"Han, Y., Hui, L., Jiang, H., Qian, J., Xie, J.: Generative subgraph contrast for\u00a0self-supervised graph representation learning. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) Computer Vision \u2013 ECCV 2022: 17th European Conference, Tel Aviv, Israel, October 23\u201327, 2022, Proceedings, Part XXX, pp. 91\u2013107. Springer Nature Switzerland, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-20056-4_6"},{"key":"24_CR15","doi-asserted-by":"crossref","unstructured":"He, K., Chen, X., Xie, S., Li, Y., Doll\u00e1r, P., Girshick, R.: Masked autoencoders are scalable vision learners. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16000\u201316009 (2022)","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"24_CR16","doi-asserted-by":"crossref","unstructured":"Huang, G., et al.: Siamese DETR. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15722\u201315731 (2023)","DOI":"10.1109\/CVPR52729.2023.01509"},{"key":"24_CR17","unstructured":"Kingma, D.P., Ba, J.: Adam: a method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)"},{"key":"24_CR18","doi-asserted-by":"crossref","unstructured":"Li, W., Zhang, Z., Liu, Z.: Action recognition based on a bag of 3D points. In: 2010 IEEE Computer Society Conference on Computer Vision and Pattern Recognition-Workshops, pp. 9\u201314. IEEE (2010)","DOI":"10.1109\/CVPRW.2010.5543273"},{"key":"24_CR19","doi-asserted-by":"publisher","unstructured":"Liu, H., Cai, M., Lee, Y.J.: Masked discrimination for self-supervised learning on point clouds. In: European Conference on Computer Vision, pp. 657\u2013675. Springer (2022). https:\/\/doi.org\/10.1007\/978-3-031-20086-1_38","DOI":"10.1007\/978-3-031-20086-1_38"},{"key":"24_CR20","doi-asserted-by":"crossref","unstructured":"Liu, X., Yan, M., Bohg, J.: MeteorNet: deep learning on dynamic 3D point cloud sequences. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9246\u20139255 (2019)","DOI":"10.1109\/ICCV.2019.00934"},{"key":"24_CR21","unstructured":"Min, Y., Chai, X., Zhao, L., Chen, X.: FlickerNet: adaptive 3D gesture recognition from sparse point clouds. In: BMVC, vol.\u00a02, p.\u00a05 (2019)"},{"key":"24_CR22","doi-asserted-by":"crossref","unstructured":"Min, Y., Zhang, Y., Chai, X., Chen, X.: An efficient PointLSTM for point clouds based gesture recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5761\u20135770 (2020)","DOI":"10.1109\/CVPR42600.2020.00580"},{"key":"24_CR23","doi-asserted-by":"crossref","unstructured":"Molchanov, P., Yang, X., Gupta, S., Kim, K., Tyree, S., Kautz, J.: Online detection and classification of dynamic hand gestures with recurrent 3D convolutional neural network. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4207\u20134215 (2016)","DOI":"10.1109\/CVPR.2016.456"},{"key":"24_CR24","doi-asserted-by":"publisher","first-page":"604","DOI":"10.1007\/978-3-031-20086-1_35","volume-title":"Computer Vision \u2013 ECCV 2022: 17th European Conference, Tel Aviv, Israel, October 23\u201327, 2022, Proceedings, Part II","author":"Y Pang","year":"2022","unstructured":"Pang, Y., Wang, W., Tay, F.E.H., Liu, W., Tian, Y., Yuan, L.: Masked autoencoders for\u00a0point cloud self-supervised learning. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) Computer Vision \u2013 ECCV 2022: 17th European Conference, Tel Aviv, Israel, October 23\u201327, 2022, Proceedings, Part II, pp. 604\u2013621. Springer Nature Switzerland, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-20086-1_35"},{"key":"24_CR25","unstructured":"Ranasinghe, K., Ryoo, M.: Language-based action concept spaces improve video self-supervised learning. arXiv preprint arXiv:2307.10922 (2023)"},{"key":"24_CR26","doi-asserted-by":"crossref","unstructured":"Shahroudy, A., Liu, J., Ng, T.T., Wang, G.: NTU RGB+ D: a large scale dataset for 3D human activity analysis. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1010\u20131019 (2016)","DOI":"10.1109\/CVPR.2016.115"},{"key":"24_CR27","doi-asserted-by":"crossref","unstructured":"Shen, Z., et al.: Masked spatio-temporal structure prediction for self-supervised learning on point cloud videos. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 16580\u201316589 (2023)","DOI":"10.1109\/ICCV51070.2023.01520"},{"key":"24_CR28","doi-asserted-by":"crossref","unstructured":"Shen, Z., Sheng, X., Wang, L., Guo, Y., Liu, Q., Zhou, X.: PointCMP: contrastive mask prediction for self-supervised learning on point cloud videos. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1212\u20131222 (2023)","DOI":"10.1109\/CVPR52729.2023.00123"},{"key":"24_CR29","doi-asserted-by":"crossref","unstructured":"Sheng, X., Shen, Z., Xiao, G.: Contrastive predictive autoencoders for dynamic point cloud self-supervised learning. arXiv preprint arXiv:2305.12959 (2023)","DOI":"10.1609\/aaai.v37i8.26170"},{"key":"24_CR30","doi-asserted-by":"crossref","unstructured":"Sheng, X., Shen, Z., Xiao, G., Wang, L., Guo, Y., Fan, H.: Point contrastive prediction with semantic clustering for self-supervised learning on point cloud videos. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 16515\u201316524 (2023)","DOI":"10.1109\/ICCV51070.2023.01514"},{"key":"24_CR31","doi-asserted-by":"crossref","unstructured":"Sun, X., et al.: Masked motion encoding for self-supervised video representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2235\u20132245 (2023)","DOI":"10.1109\/CVPR52729.2023.00222"},{"key":"24_CR32","doi-asserted-by":"publisher","first-page":"404","DOI":"10.1007\/978-3-030-66096-3_28","volume-title":"Computer Vision \u2013 ECCV 2020 Workshops: Glasgow, UK, August 23\u201328, 2020, Proceedings, Part II","author":"P Tokmakov","year":"2020","unstructured":"Tokmakov, P., Hebert, M., Schmid, C.: Unsupervised learning of video representations via dense trajectory clustering. In: Bartoli, A., Fusiello, A. (eds.) Computer Vision \u2013 ECCV 2020 Workshops: Glasgow, UK, August 23\u201328, 2020, Proceedings, Part II, pp. 404\u2013421. Springer International Publishing, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-66096-3_28"},{"key":"24_CR33","first-page":"10078","volume":"35","author":"Z Tong","year":"2022","unstructured":"Tong, Z., Song, Y., Wang, J., Wang, L.: VideoMAE: masked autoencoders are data-efficient learners for self-supervised video pre-training. Adv. Neural. Inf. Process. Syst. 35, 10078\u201310093 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"24_CR34","doi-asserted-by":"crossref","unstructured":"Wang, G., Zhou, Y., Luo, C., Xie, W., Zeng, W., Xiong, Z.: Unsupervised visual representation learning by tracking patches in video. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2563\u20132572 (2021)","DOI":"10.1109\/CVPR46437.2021.00259"},{"key":"24_CR35","doi-asserted-by":"crossref","unstructured":"Wang, H., Yang, L., Rong, X., Feng, J., Tian, Y.: Self-supervised 4D spatio-temporal feature learning via order prediction of sequential point cloud clips. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 3762\u20133771 (2021)","DOI":"10.1109\/WACV48630.2021.00381"},{"key":"24_CR36","doi-asserted-by":"crossref","unstructured":"Wang, X., Gupta, A.: Unsupervised learning of visual representations using videos. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2794\u20132802 (2015)","DOI":"10.1109\/ICCV.2015.320"},{"key":"24_CR37","doi-asserted-by":"crossref","unstructured":"Wang, Y., et al.: 3DV: 3D dynamic voxel for action recognition in depth video. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 511\u2013520 (2020)","DOI":"10.1109\/CVPR42600.2020.00059"},{"key":"24_CR38","doi-asserted-by":"crossref","unstructured":"Wei, C., Fan, H., Xie, S., Wu, C.Y., Yuille, A., Feichtenhofer, C.: Masked feature prediction for self-supervised visual pre-training. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14668\u201314678 (2022)","DOI":"10.1109\/CVPR52688.2022.01426"},{"key":"24_CR39","doi-asserted-by":"publisher","first-page":"19","DOI":"10.1007\/978-3-031-19818-2_2","volume-title":"Computer Vision \u2013 ECCV 2022: 17th European Conference, Tel Aviv, Israel, October 23\u201327, 2022, Proceedings, Part XXIX","author":"H Wen","year":"2022","unstructured":"Wen, H., Liu, Y., Huang, J., Duan, B., Yi, L.: Point primitive transformer for\u00a0long-term 4D point cloud video understanding. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) Computer Vision \u2013 ECCV 2022: 17th European Conference, Tel Aviv, Israel, October 23\u201327, 2022, Proceedings, Part XXIX, pp. 19\u201335. Springer Nature Switzerland, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19818-2_2"},{"key":"24_CR40","unstructured":"Yu, Y., Wang, X., Zhang, M., Liu, N., Shi, C.: Provable training for graph contrastive learning. arXiv preprint arXiv:2309.13944 (2023)"},{"key":"24_CR41","doi-asserted-by":"crossref","unstructured":"Zeng, Y., Qian, Y., Zhu, Z., Hou, J., Yuan, H., He, Y.: CorrNet3D: unsupervised end-to-end learning of dense correspondence for 3D point clouds. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6052\u20136061 (2021)","DOI":"10.1109\/CVPR46437.2021.00599"},{"key":"24_CR42","first-page":"27061","volume":"35","author":"R Zhang","year":"2022","unstructured":"Zhang, R., et al.: Point-M2AE: multi-scale masked autoencoders for hierarchical point cloud pre-training. Adv. Neural. Inf. Process. Syst. 35, 27061\u201327074 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"24_CR43","doi-asserted-by":"crossref","unstructured":"Zhang, R., Wang, L., Qiao, Y., Gao, P., Li, H.: Learning 3D representations from 2D pre-trained models via image-to-point masked autoencoders. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 21769\u201321780 (2023)","DOI":"10.1109\/CVPR52729.2023.02085"},{"key":"24_CR44","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Dong, Y., Liu, Y., Yi, L.: Complete-to-partial 4D distillation for self-supervised point cloud sequence representation learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 17661\u201317670 (2023)","DOI":"10.1109\/CVPR52729.2023.01694"},{"key":"24_CR45","doi-asserted-by":"crossref","unstructured":"Zhong, J.X., Zhou, K., Hu, Q., Wang, B., Trigoni, N., Markham, A.: No Pain, Big Gain: classify dynamic point cloud sequences with static models by fitting feature-level space-time surfaces. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8510\u20138520 (2022)","DOI":"10.1109\/CVPR52688.2022.00832"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-73116-7_24","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T15:26:53Z","timestamp":1730302013000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-73116-7_24"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,31]]},"ISBN":["9783031731150","9783031731167"],"references-count":45,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-73116-7_24","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,31]]},"assertion":[{"value":"31 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}