{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,24]],"date-time":"2026-06-24T15:14:46Z","timestamp":1782314086086,"version":"3.54.5"},"publisher-location":"Cham","reference-count":41,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031733963","type":"print"},{"value":"9783031733970","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,11,3]],"date-time":"2024-11-03T00:00:00Z","timestamp":1730592000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,3]],"date-time":"2024-11-03T00:00:00Z","timestamp":1730592000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-73397-0_26","type":"book-chapter","created":{"date-parts":[[2024,11,2]],"date-time":"2024-11-02T19:08:22Z","timestamp":1730574502000},"page":"447-463","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Decomposed Vector-Quantized Variational Autoencoder for\u00a0Human Grasp Generation"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-2722-5563","authenticated-orcid":false,"given":"Zhe","family":"Zhao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6955-6635","authenticated-orcid":false,"given":"Mengshi","family":"Qi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7199-5047","authenticated-orcid":false,"given":"Huadong","family":"Ma","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,11,3]]},"reference":[{"key":"26_CR1","doi-asserted-by":"crossref","unstructured":"Boukhayma, A., Bem, R.d., Torr, P.H.: 3D hand shape and pose from images in the wild. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10843\u201310852 (2019)","DOI":"10.1109\/CVPR.2019.01110"},{"key":"26_CR2","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"361","DOI":"10.1007\/978-3-030-58601-0_22","volume-title":"Computer Vision \u2013 ECCV 2020","author":"S Brahmbhatt","year":"2020","unstructured":"Brahmbhatt, S., Tang, C., Twigg, C.D., Kemp, C.C., Hays, J.: ContactPose: a dataset of grasps with object contact and hand pose. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020, Part XIII. LNCS, vol. 12358, pp. 361\u2013378. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58601-0_22"},{"key":"26_CR3","doi-asserted-by":"crossref","unstructured":"Corona, E., Pumarola, A., Alenya, G., Moreno-Noguer, F., Rogez, G.: Ganhand: predicting human grasp affordances in multi-object scenes. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5031\u20135041 (2020)","DOI":"10.1109\/CVPR42600.2020.00508"},{"key":"26_CR4","doi-asserted-by":"crossref","unstructured":"Dibra, E., Melchior, S., Balkis, A., Wolf, T., Oztireli, C., Gross, M.: Monocular RGB hand pose inference from unsupervised refinable nets. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition Workshops, pp. 1075\u20131085 (2018)","DOI":"10.1109\/CVPRW.2018.00155"},{"key":"26_CR5","doi-asserted-by":"crossref","unstructured":"Fan, H., Su, H., Guibas, L.J.: A point set generation network for 3d object reconstruction from a single image. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 605\u2013613 (2017)","DOI":"10.1109\/CVPR.2017.264"},{"key":"26_CR6","doi-asserted-by":"crossref","unstructured":"Garcia-Hernando, G., Yuan, S., Baek, S., Kim, T.K.: First-person hand action benchmark with rgb-d videos and 3d hand pose annotations. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 409\u2013419 (2018)","DOI":"10.1109\/CVPR.2018.00050"},{"key":"26_CR7","doi-asserted-by":"crossref","unstructured":"Ge, L., Cai, Y., Weng, J., Yuan, J.: Hand pointnet: 3D hand pose estimation using point sets. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 8417\u20138426 (2018)","DOI":"10.1109\/CVPR.2018.00878"},{"key":"26_CR8","unstructured":"Goodfellow, I., et al.: Generative adversarial nets. In: Advances in Neural Information Processing Systems, vol. 27 (2014)"},{"key":"26_CR9","doi-asserted-by":"crossref","unstructured":"Grady, P., Tang, C., Twigg, C.D., Vo, M., Brahmbhatt, S., Kemp, C.C.: ContactOPT: optimizing contact to improve grasps. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1471\u20131481 (2021)","DOI":"10.1109\/CVPR46437.2021.00152"},{"key":"26_CR10","doi-asserted-by":"crossref","unstructured":"Hampali, S., Rad, M., Oberweger, M., Lepetit, V.: Honnotate: a method for 3D annotation of hand and object poses. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3196\u20133206 (2020)","DOI":"10.1109\/CVPR42600.2020.00326"},{"key":"26_CR11","doi-asserted-by":"crossref","unstructured":"Hasson, Y., et al.: Learning joint reconstruction of hands and manipulated objects. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11807\u201311816 (2019)","DOI":"10.1109\/CVPR.2019.01208"},{"key":"26_CR12","doi-asserted-by":"crossref","unstructured":"Jiang, H., Liu, S., Wang, J., Wang, X.: Hand-object contact consistency reasoning for human grasps generation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 11107\u201311116 (2021)","DOI":"10.1109\/ICCV48922.2021.01092"},{"key":"26_CR13","doi-asserted-by":"crossref","unstructured":"Karunratanakul, K., Spurr, A., Fan, Z., Hilliges, O., Tang, S.: A skeleton-driven neural occupancy representation for articulated hands. In: 2021 International Conference on 3D Vision (3DV), pp. 11\u201321. IEEE (2021)","DOI":"10.1109\/3DV53792.2021.00012"},{"key":"26_CR14","doi-asserted-by":"crossref","unstructured":"Karunratanakul, K., Yang, J., Zhang, Y., Black, M.J., Muandet, K., Tang, S.: Grasping field: learning implicit representations for human grasps. In: 2020 International Conference on 3D Vision (3DV), pp. 333\u2013344. IEEE (2020)","DOI":"10.1109\/3DV50981.2020.00043"},{"key":"26_CR15","unstructured":"Kingma, D.P., Welling, M.: Auto-encoding variational Bayes. arXiv preprint arXiv:1312.6114 (2013)"},{"key":"26_CR16","doi-asserted-by":"crossref","unstructured":"Liu, S., Zhou, Y., Yang, J., Gupta, S., Wang, S.: ContactGEN: generative contact modeling for grasp generation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 20609\u201320620 (2023)","DOI":"10.1109\/ICCV51070.2023.01884"},{"key":"26_CR17","doi-asserted-by":"crossref","unstructured":"Lv, C., Qi, M., Li, X., Yang, Z., Ma, H.: Sgformer: semantic graph transformer for point cloud-based 3D scene graph generation. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a038, pp. 4035\u20134043 (2024)","DOI":"10.1609\/aaai.v38i5.28197"},{"key":"26_CR18","unstructured":"Lv, C., Zhang, S., Tian, Y., Qi, M., Ma, H.: Disentangled counterfactual learning for physical audiovisual commonsense reasoning. In: Advances in Neural Information Processing Systems, vol. 36 (2024)"},{"issue":"4","key":"26_CR19","doi-asserted-by":"publisher","first-page":"110","DOI":"10.1109\/MRA.2004.1371616","volume":"11","author":"AT Miller","year":"2004","unstructured":"Miller, A.T., Allen, P.K.: Graspit! A versatile simulator for robotic grasping. IEEE Robot. Autom. Mag. 11(4), 110\u2013122 (2004)","journal-title":"IEEE Robot. Autom. Mag."},{"key":"26_CR20","doi-asserted-by":"crossref","unstructured":"Mittal, P., Cheng, Y.C., Singh, M., Tulsiani, S.: AutoSDF: shape priors for 3D completion, reconstruction and generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 306\u2013315 (2022)","DOI":"10.1109\/CVPR52688.2022.00040"},{"key":"26_CR21","unstructured":"Oleynikova, H., Millane, A., Taylor, Z., Galceran, E., Nieto, J., Siegwart, R.: Signed distance fields: a natural representation for both mapping and planning. In: RSS 2016 Workshop: Geometry and Beyond-representations, Physics, and Scene Understanding for Robotics. University of Michigan (2016)"},{"key":"26_CR22","unstructured":"Van\u00a0den Oord, A., Kalchbrenner, N., Espeholt, L., Vinyals, O., Graves, A., et\u00a0al.: Conditional image generation with pixelCNN decoders. In: Advances in Neural Information Processing Systems, vol. 29 (2016)"},{"key":"26_CR23","doi-asserted-by":"crossref","unstructured":"Pi, H., Peng, S., Yang, M., Zhou, X., Bao, H.: Hierarchical generation of human-object interactions with diffusion probabilistic models. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 15061\u201315073 (2023)","DOI":"10.1109\/ICCV51070.2023.01383"},{"key":"26_CR24","unstructured":"Qi, C.R., Su, H., Mo, K., Guibas, L.J.: Pointnet: deep learning on point sets for 3D classification and segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 652\u2013660 (2017)"},{"key":"26_CR25","doi-asserted-by":"crossref","unstructured":"Qi, M., Li, W., Yang, Z., Wang, Y., Luo, J.: Attentive relational networks for mapping images to scene graphs. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3957\u20133966 (2019)","DOI":"10.1109\/CVPR.2019.00408"},{"key":"26_CR26","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"104","DOI":"10.1007\/978-3-030-01249-6_7","volume-title":"Computer Vision \u2013 ECCV 2018","author":"M Qi","year":"2018","unstructured":"Qi, M., Qin, J., Li, A., Wang, Y., Luo, J., Van Gool, L.: stagNet: an attentive semantic RNN for group activity recognition. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11214, pp. 104\u2013120. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01249-6_7"},{"key":"26_CR27","doi-asserted-by":"publisher","first-page":"2989","DOI":"10.1109\/TIP.2020.3048680","volume":"30","author":"M Qi","year":"2021","unstructured":"Qi, M., Qin, J., Yang, Y., Wang, Y., Luo, J.: Semantics-aware spatial-temporal binaries for cross-modal video retrieval. IEEE Trans. Image Process. 30, 2989\u20133004 (2021)","journal-title":"IEEE Trans. Image Process."},{"key":"26_CR28","doi-asserted-by":"publisher","first-page":"5420","DOI":"10.1109\/TIP.2020.2983567","volume":"29","author":"M Qi","year":"2020","unstructured":"Qi, M., Wang, Y., Li, A., Luo, J.: STC-GAN: spatio-temporally coupled generative adversarial networks for predictive scene parsing. IEEE Trans. Image Process. 29, 5420\u20135430 (2020)","journal-title":"IEEE Trans. Image Process."},{"key":"26_CR29","unstructured":"Razavi, A., Van\u00a0den Oord, A., Vinyals, O.: Generating diverse high-fidelity images with vq-vae-2. In: Advances in Neural Information Processing Systems, vol. 32 (2019)"},{"key":"26_CR30","unstructured":"Romero, J., Tzionas, D., Black, M.J.: Embodied hands: modeling and capturing hands and bodies together. arXiv preprint arXiv:2201.02610 (2022)"},{"key":"26_CR31","unstructured":"Sohn, K., Lee, H., Yan, X.: Learning structured output representation using deep conditional generative models. In: Advances in Neural Information Processing Systems, vol. 28 (2015)"},{"key":"26_CR32","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"581","DOI":"10.1007\/978-3-030-58548-8_34","volume-title":"Computer Vision \u2013 ECCV 2020","author":"O Taheri","year":"2020","unstructured":"Taheri, O., Ghorbani, N., Black, M.J., Tzionas, D.: GRAB: a dataset of whole-body human grasping of objects. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020, Part IV. LNCS, vol. 12349, pp. 581\u2013600. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58548-8_34"},{"key":"26_CR33","doi-asserted-by":"publisher","first-page":"172","DOI":"10.1007\/s11263-016-0895-4","volume":"118","author":"D Tzionas","year":"2016","unstructured":"Tzionas, D., Ballan, L., Srikantha, A., Aponte, P., Pollefeys, M., Gall, J.: Capturing hands in action using discriminative salient points and physics simulation. Int. J. Comput. Vision 118, 172\u2013193 (2016)","journal-title":"Int. J. Comput. Vision"},{"key":"26_CR34","doi-asserted-by":"crossref","unstructured":"Tzionas, D., Gall, J.: 3D object reconstruction from hand-object interactions. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 729\u2013737 (2015)","DOI":"10.1109\/ICCV.2015.90"},{"key":"26_CR35","unstructured":"Van Den\u00a0Oord, A., Vinyals, O., et\u00a0al.: Neural discrete representation learning. In: Advances in Neural Information Processing Systems, vol. 30 (2017)"},{"key":"26_CR36","unstructured":"Wang, C., Wang, L.: Adaptive weight learning for multiple outcome optimization with continuous treatment. arXiv preprint arXiv:2402.11092 (2024)"},{"key":"26_CR37","doi-asserted-by":"crossref","unstructured":"Wang, H., et al.: RGB-depth fusion GAN for indoor depth completion. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6209\u20136218 (2022)","DOI":"10.1109\/CVPR52688.2022.00611"},{"key":"26_CR38","doi-asserted-by":"crossref","unstructured":"Wang, X., Wu, Y., Zhu, L., Yang, Y.: Symbiotic attention with privileged information for egocentric action recognition. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a034, pp. 12249\u201312256 (2020)","DOI":"10.1609\/aaai.v34i07.6907"},{"key":"26_CR39","doi-asserted-by":"crossref","unstructured":"Wang, X., Zhu, L., Wang, H., Yang, Y.: Interactive prototype learning for egocentric action recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8168\u20138177 (2021)","DOI":"10.1109\/ICCV48922.2021.00806"},{"key":"26_CR40","doi-asserted-by":"crossref","unstructured":"Zheng, Y., Shi, Y., Cui, Y., Zhao, Z., Luo, Z., Zhou, W.: Coop: decoupling and coupling of whole-body grasping pose generation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2163\u20132173 (2023)","DOI":"10.1109\/ICCV51070.2023.00206"},{"key":"26_CR41","doi-asserted-by":"crossref","unstructured":"Zimmermann, C., Brox, T.: Learning to estimate 3D hand pose from single RGB images. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 4903\u20134911 (2017)","DOI":"10.1109\/ICCV.2017.525"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-73397-0_26","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,2]],"date-time":"2024-11-02T19:21:19Z","timestamp":1730575279000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-73397-0_26"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,3]]},"ISBN":["9783031733963","9783031733970"],"references-count":41,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-73397-0_26","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,3]]},"assertion":[{"value":"3 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}