{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T15:30:44Z","timestamp":1778081444017,"version":"3.51.4"},"publisher-location":"Cham","reference-count":59,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031915772","type":"print"},{"value":"9783031915789","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-91578-9_3","type":"book-chapter","created":{"date-parts":[[2025,6,6]],"date-time":"2025-06-06T09:23:20Z","timestamp":1749201800000},"page":"49-67","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Generative Hierarchical Temporal Transformer for\u00a0Hand Pose and\u00a0Action Modeling"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5981-1276","authenticated-orcid":false,"given":"Yilin","family":"Wen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3628-9777","authenticated-orcid":false,"given":"Hao","family":"Pan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2329-8797","authenticated-orcid":false,"given":"Takehiko","family":"Ohkawa","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3284-4019","authenticated-orcid":false,"given":"Lei","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9003-2054","authenticated-orcid":false,"given":"Jia","family":"Pan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0097-4537","authenticated-orcid":false,"given":"Yoichi","family":"Sato","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2729-5860","authenticated-orcid":false,"given":"Taku","family":"Komura","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2284-3952","authenticated-orcid":false,"given":"Wenping","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,5,12]]},"reference":[{"key":"3_CR1","doi-asserted-by":"crossref","unstructured":"Aliakbarian, S., Saleh, F.S., Salzmann, M., Petersson, L., Gould, S.: A stochastic conditioning scheme for diverse human motion prediction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5223\u20135232 (2020)","DOI":"10.1109\/CVPR42600.2020.00527"},{"key":"3_CR2","doi-asserted-by":"crossref","unstructured":"Bao, W., et al.: Uncertainty-aware state space transformer for egocentric 3D hand trajectory forecasting. arXiv preprint arXiv:2307.08243 (2023)","DOI":"10.1109\/ICCV51070.2023.01260"},{"key":"3_CR3","doi-asserted-by":"crossref","unstructured":"Cai, Y., et al.: Exploiting spatial-temporal relationships for 3D pose estimation via graph convolutional networks. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2272\u20132281 (2019)","DOI":"10.1109\/ICCV.2019.00236"},{"key":"3_CR4","doi-asserted-by":"crossref","unstructured":"Cai, Y., et\u00a0al.: A unified 3D human motion synthesis model via conditional variational auto-encoder. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 11645\u201311655 (2021)","DOI":"10.1109\/ICCV48922.2021.01144"},{"key":"3_CR5","doi-asserted-by":"crossref","unstructured":"Carreira, J., Zisserman, A.: Quo vadis, action recognition? A new model and the kinetics dataset. In: proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6299\u20136308 (2017)","DOI":"10.1109\/CVPR.2017.502"},{"key":"3_CR6","doi-asserted-by":"crossref","unstructured":"Cherti, M., et al.: Reproducible scaling laws for contrastive language-image learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2818\u20132829 (2023)","DOI":"10.1109\/CVPR52729.2023.00276"},{"key":"3_CR7","doi-asserted-by":"crossref","unstructured":"Chi, S., Chi, H.G., Huang, Q., Ramani, K.: InfoGCN++: learning representation by predicting the future for online human skeleton-based action recognition. arXiv preprint arXiv:2310.10547 (2023)","DOI":"10.1109\/CVPR52688.2022.01955"},{"key":"3_CR8","doi-asserted-by":"crossref","unstructured":"Chik, D., Trumpf, J., Schraudolph, N.N.: Using an adaptive VAR model for motion prediction in 3D hand tracking. In: 2008 8th IEEE International Conference on Automatic Face & Gesture Recognition, pp.\u00a01\u20138. IEEE (2008)","DOI":"10.1109\/AFGR.2008.4813414"},{"key":"3_CR9","doi-asserted-by":"crossref","unstructured":"Cho, H., Kim, C., Kim, J., Lee, S., Ismayilzada, E., Baek, S.: Transformer-based unified recognition of two hands manipulating objects. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4769\u20134778 (2023)","DOI":"10.1109\/CVPR52729.2023.00462"},{"key":"3_CR10","doi-asserted-by":"crossref","unstructured":"Fan, Z., Liu, J., Wang, Y.: Adaptive computationally efficient network for monocular 3D hand pose estimation. In: European Conference on Computer Vision, pp. 127\u2013144. Springer (2020)","DOI":"10.1007\/978-3-030-58548-8_8"},{"key":"3_CR11","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C.: X3D: expanding architectures for efficient video recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 203\u2013213 (2020)","DOI":"10.1109\/CVPR42600.2020.00028"},{"key":"3_CR12","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Fan, H., Malik, J., He, K.: Slowfast networks for video recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6202\u20136211 (2019)","DOI":"10.1109\/ICCV.2019.00630"},{"key":"3_CR13","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Pinz, A., Zisserman, A.: Convolutional two-stream network fusion for video action recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1933\u20131941 (2016)","DOI":"10.1109\/CVPR.2016.213"},{"key":"3_CR14","doi-asserted-by":"crossref","unstructured":"Gammulle, H., Denman, S., Sridharan, S., Fookes, C.: Predicting the future: a jointly learnt model for action anticipation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 5562\u20135571 (2019)","DOI":"10.1109\/ICCV.2019.00566"},{"key":"3_CR15","doi-asserted-by":"crossref","unstructured":"Guo, C., et al.: Action2Motion: conditioned generation of 3D human motions. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 2021\u20132029 (2020)","DOI":"10.1145\/3394171.3413635"},{"key":"3_CR16","doi-asserted-by":"crossref","unstructured":"Han, S., et al.: MegaTrack: monochrome egocentric articulated hand-tracking for virtual reality. ACM Trans. Graph. (ToG) 39(4), 87-1 (2020)","DOI":"10.1145\/3386569.3392452"},{"key":"3_CR17","doi-asserted-by":"crossref","unstructured":"Han, S., et\u00a0al.: UmeTrack: unified multi-view end-to-end hand tracking for VR. In: SIGGRAPH Asia 2022 Conference Papers, pp.\u00a01\u20139 (2022)","DOI":"10.1145\/3550469.3555378"},{"key":"3_CR18","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"3_CR19","unstructured":"Heusel, M., Ramsauer, H., Unterthiner, T., Nessler, B., Hochreiter, S.: GANs trained by a two time-scale update rule converge to a local nash equilibrium. In: Advances in Neural Information Processing Systems, vol.\u00a030. Curran Associates, Inc. (2017)"},{"key":"3_CR20","doi-asserted-by":"crossref","unstructured":"Iqbal, U., Molchanov, P., Gall, T.B.J., Kautz, J.: Hand pose estimation via latent 2.5 D heatmap regression. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 118\u2013134 (2018)","DOI":"10.1007\/978-3-030-01252-6_8"},{"key":"3_CR21","unstructured":"Jiang, B., Chen, X., Liu, W., Yu, J., Yu, G., Chen, T.: MotionGPT: human motion as a foreign language. arXiv preprint arXiv:2306.14795 (2023)"},{"key":"3_CR22","unstructured":"Kingma, D.P., Welling, M.: Auto-encoding variational bayes. arXiv preprint arXiv:1312.6114 (2013)"},{"key":"3_CR23","doi-asserted-by":"crossref","unstructured":"Kwon, T., Tekin, B., St\u00fchmer, J., Bogo, F., Pollefeys, M.: H2O: two hands manipulating objects for first person interaction recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 10138\u201310148 (2021)","DOI":"10.1109\/ICCV48922.2021.00998"},{"key":"3_CR24","doi-asserted-by":"crossref","unstructured":"Li, M., et al.: Interacting attention graph for single image two-hand reconstruction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2761\u20132770 (2022)","DOI":"10.1109\/CVPR52688.2022.00278"},{"key":"3_CR25","doi-asserted-by":"crossref","unstructured":"Li, Y., Ye, Z., Rehg, J.M.: Delving into egocentric actions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 287\u2013295 (2015)","DOI":"10.1109\/CVPR.2015.7298625"},{"key":"3_CR26","doi-asserted-by":"crossref","unstructured":"Liu, M., Tang, S., Li, Y., Rehg, J.M.: Forecasting human-object interaction: joint prediction of motor attention and actions in first person video. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, 23\u201328 August 2020, Proceedings, Part I, pp. 704\u2013721. Springer (2020)","DOI":"10.1007\/978-3-030-58452-8_41"},{"key":"3_CR27","doi-asserted-by":"crossref","unstructured":"Liu, S., Tripathi, S., Majumdar, S., Wang, X.: Joint hand motion and interaction hotspots prediction from egocentric videos. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3282\u20133292 (2022)","DOI":"10.1109\/CVPR52688.2022.00328"},{"key":"3_CR28","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101 (2017)"},{"key":"3_CR29","doi-asserted-by":"crossref","unstructured":"Lucas, T., Baradel, F., Weinzaepfel, P., Rogez, G.: PoseGPT: quantization-based 3D human motion generation and forecasting. In: European Conference on Computer Vision, pp. 417\u2013435. Springer (2022)","DOI":"10.1007\/978-3-031-20068-7_24"},{"key":"3_CR30","doi-asserted-by":"crossref","unstructured":"Luo, R.C., Mai, L.: Human intention inference and on-line human hand motion prediction for human-robot collaboration. In: 2019 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS), pp. 5958\u20135964. IEEE (2019)","DOI":"10.1109\/IROS40897.2019.8968192"},{"key":"3_CR31","doi-asserted-by":"crossref","unstructured":"Ma, H., Li, J., Hosseini, R., Tomizuka, M., Choi, C.: Multi-objective diverse human motion prediction with knowledge distillation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8161\u20138171 (2022)","DOI":"10.1109\/CVPR52688.2022.00799"},{"key":"3_CR32","doi-asserted-by":"crossref","unstructured":"Ma, M., Fan, H., Kitani, K.M.: Going deeper into first-person activity recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1894\u20131903 (2016)","DOI":"10.1109\/CVPR.2016.209"},{"key":"3_CR33","doi-asserted-by":"crossref","unstructured":"Mao, W., Liu, M., Salzmann, M.: Weakly-supervised action transition learning for stochastic human motion prediction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8151\u20138160 (2022)","DOI":"10.1109\/CVPR52688.2022.00798"},{"key":"3_CR34","doi-asserted-by":"crossref","unstructured":"Moon, G.: Bringing inputs to shared domains for 3D interacting hands recovery in the wild. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 17028\u201317037 (2023)","DOI":"10.1109\/CVPR52729.2023.01633"},{"key":"3_CR35","doi-asserted-by":"crossref","unstructured":"Mueller, F., et al.: Ganerated hands for real-time 3D hand tracking from monocular RGB. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 49\u201359 (2018)","DOI":"10.1109\/CVPR.2018.00013"},{"key":"3_CR36","doi-asserted-by":"crossref","unstructured":"Ohkawa, T., He, K., Sener, F., Hodan, T., Tran, L., Keskin, C.: AssemblyHands: towards egocentric activity understanding via 3d hand pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 12999\u201313008 (2023)","DOI":"10.1109\/CVPR52729.2023.01249"},{"key":"3_CR37","doi-asserted-by":"crossref","unstructured":"Petrovich, M., Black, M.J., Varol, G.: Action-conditioned 3D human motion synthesis with transformer VAE. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 10985\u201310995 (2021)","DOI":"10.1109\/ICCV48922.2021.01080"},{"key":"3_CR38","doi-asserted-by":"crossref","unstructured":"Petrovich, M., Black, M.J., Varol, G.: TEMOS: generating diverse human motions from textual descriptions. In: European Conference on Computer Vision (ECCV) (2022)","DOI":"10.1007\/978-3-031-20047-2_28"},{"key":"3_CR39","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: ICML (2021)"},{"key":"3_CR40","doi-asserted-by":"crossref","unstructured":"Rempe, D., Birdal, T., Hertzmann, A., Yang, J., Sridhar, S., Guibas, L.J.: Humor: 3D human motion model for robust pose estimation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 11488\u201311499 (2021)","DOI":"10.1109\/ICCV48922.2021.01129"},{"key":"3_CR41","unstructured":"Schuhmann, C., et al.: LAION-5B: an open large-scale dataset for training next generation image-text models. In: Thirty-sixth Conference on Neural Information Processing Systems Datasets and Benchmarks Track (2022). https:\/\/openreview.net\/forum?id=M3Y74vmsMcY"},{"key":"3_CR42","doi-asserted-by":"crossref","unstructured":"Sener, F., et al.: Assembly101: a large-scale multi-view video dataset for understanding procedural activities. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 21096\u201321106 (2022)","DOI":"10.1109\/CVPR52688.2022.02042"},{"key":"3_CR43","doi-asserted-by":"crossref","unstructured":"Shi, L., Zhang, Y., Cheng, J., Lu, H.: Two-stream adaptive graph convolutional networks for skeleton-based action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12026\u201312035 (2019)","DOI":"10.1109\/CVPR.2019.01230"},{"key":"3_CR44","doi-asserted-by":"crossref","unstructured":"Shi, M., Starke, S., Ye, Y., Komura, T., Won, J.: PhaseMP: robust 3D pose estimation via phase-conditioned human motion prior. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 14725\u201314737 (2023)","DOI":"10.1109\/ICCV51070.2023.01353"},{"key":"3_CR45","doi-asserted-by":"crossref","unstructured":"Singh, S., Arora, C., Jawahar, C.: First person action recognition using deep learned descriptors. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2620\u20132628 (2016)","DOI":"10.1109\/CVPR.2016.287"},{"key":"3_CR46","doi-asserted-by":"crossref","unstructured":"Spurr, A., Iqbal, U., Molchanov, P., Hilliges, O., Kautz, J.: Weakly supervised 3D hand pose estimation via biomechanical constraints. In: European Conference on Computer Vision, pp. 211\u2013228. Springer (2020)","DOI":"10.1007\/978-3-030-58520-4_13"},{"key":"3_CR47","doi-asserted-by":"crossref","unstructured":"Tekin, B., Bogo, F., Pollefeys, M.: H+O: unified egocentric recognition of 3D hand-object poses and interactions. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4511\u20134520 (2019)","DOI":"10.1109\/CVPR.2019.00464"},{"key":"3_CR48","doi-asserted-by":"crossref","unstructured":"Tevet, G., Gordon, B., Hertz, A., Bermano, A.H., Cohen-Or, D.: MotionCLIP: exposing human motion generation to clip space. In: European Conference on Computer Vision, pp. 358\u2013374. Springer (2022)","DOI":"10.1007\/978-3-031-20047-2_21"},{"key":"3_CR49","unstructured":"Tevet, G., Raab, S., Gordon, B., Shafir, Y., Cohen-Or, D., Bermano, A.H.: Human motion diffusion model. arXiv preprint arXiv:2209.14916 (2022)"},{"key":"3_CR50","doi-asserted-by":"crossref","unstructured":"Vondrick, C., Pirsiavash, H., Torralba, A.: Anticipating visual representations from unlabeled video. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 98\u2013106 (2016)","DOI":"10.1109\/CVPR.2016.18"},{"issue":"6","key":"3_CR51","first-page":"1","volume":"39","author":"J Wang","year":"2020","unstructured":"Wang, J., et al.: RGB2Hands: real-time tracking of 3D hand interactions from monocular RGB video. ACM Trans. Graph. (ToG) 39(6), 1\u201316 (2020)","journal-title":"ACM Trans. Graph. (ToG)"},{"key":"3_CR52","doi-asserted-by":"crossref","unstructured":"Wen, Y., Pan, H., Yang, L., Pan, J., Komura, T., Wang, W.: Hierarchical temporal transformer for 3D hand pose estimation and action recognition from egocentric RGB videos. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (2023)","DOI":"10.1109\/CVPR52729.2023.02035"},{"key":"3_CR53","doi-asserted-by":"crossref","unstructured":"Xu, M., Gao, M., Chen, Y.T., Davis, L.S., Crandall, D.J.: Temporal recurrent networks for online action detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 5532\u20135541 (2019)","DOI":"10.1109\/ICCV.2019.00563"},{"key":"3_CR54","doi-asserted-by":"crossref","unstructured":"Yang, S., Liu, J., Lu, S., Er, M.H., Kot, A.C.: Collaborative learning of gesture recognition and 3d hand pose estimation with multi-order feature analysis. In: European Conference on Computer Vision, pp. 769\u2013786. Springer (2020)","DOI":"10.1007\/978-3-030-58580-8_45"},{"key":"3_CR55","doi-asserted-by":"crossref","unstructured":"Yu, Z., Huang, S., Fang, C., Breckon, T.P., Wang, J.: ACR: attention collaboration-based regressor for arbitrary two-hand reconstruction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12955\u201312964 (2023)","DOI":"10.1109\/CVPR52729.2023.01245"},{"key":"3_CR56","doi-asserted-by":"crossref","unstructured":"Yuan, Y., Kitani, K.: DLow: diversifying latent flows for diverse human motion prediction. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, 23\u201328 August 2020, Proceedings, Part IX, pp. 346\u2013364. Springer (2020)","DOI":"10.1007\/978-3-030-58545-7_20"},{"key":"3_CR57","unstructured":"Zhang, Y., et al.: MotionGPT: finetuned LLMs are general-purpose motion generators. arXiv preprint arXiv:2306.10900 (2023)"},{"key":"3_CR58","doi-asserted-by":"crossref","unstructured":"Zhao, Y., Kr\u00e4henb\u00fchl, P.: Real-time online video detection with temporal smoothing transformers. In: European Conference on Computer Vision, pp. 485\u2013502. Springer (2022)","DOI":"10.1007\/978-3-031-19830-4_28"},{"key":"3_CR59","doi-asserted-by":"crossref","unstructured":"Zimmermann, C., Brox, T.: Learning to estimate 3D hand pose from single RGB images. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 4903\u20134911 (2017)","DOI":"10.1109\/ICCV.2017.525"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024 Workshops"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-91578-9_3","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,6]],"date-time":"2025-06-06T09:23:43Z","timestamp":1749201823000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-91578-9_3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9783031915772","9783031915789"],"references-count":59,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-91578-9_3","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"12 May 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}