{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,26]],"date-time":"2025-03-26T09:46:22Z","timestamp":1742982382119,"version":"3.40.3"},"publisher-location":"Singapore","reference-count":41,"publisher":"Springer Nature Singapore","isbn-type":[{"type":"print","value":"9789819785070"},{"type":"electronic","value":"9789819785087"}],"license":[{"start":{"date-parts":[[2024,11,3]],"date-time":"2024-11-03T00:00:00Z","timestamp":1730592000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,3]],"date-time":"2024-11-03T00:00:00Z","timestamp":1730592000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-97-8508-7_22","type":"book-chapter","created":{"date-parts":[[2024,11,2]],"date-time":"2024-11-02T06:06:34Z","timestamp":1730527594000},"page":"313-327","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["MMIDM: Generating 3D Gesture from\u00a0Multimodal Inputs with\u00a0Diffusion Models"],"prefix":"10.1007","author":[{"given":"Ji","family":"Ye","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4673-5806","authenticated-orcid":false,"given":"Changhong","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Haocong","family":"Wan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5979-7590","authenticated-orcid":false,"given":"Aiwen","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5846-7402","authenticated-orcid":false,"given":"Zhenchun","family":"Lei","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,11,3]]},"reference":[{"key":"22_CR1","doi-asserted-by":"crossref","unstructured":"Ao, T., Zhang, Z., Liu, L.: Gesturediffuclip: gesture diffusion model with clip latents (2023). arXiv preprint arXiv:2303.14613","DOI":"10.1145\/3592097"},{"key":"22_CR2","unstructured":"Bai, S., Kolter, J.Z., Koltun, V.: An empirical evaluation of generic convolutional and recurrent networks for sequence modeling (2018). arXiv preprint arXiv:1803.01271"},{"key":"22_CR3","doi-asserted-by":"publisher","first-page":"135","DOI":"10.1162\/tacl_a_00051","volume":"5","author":"P Bojanowski","year":"2017","unstructured":"Bojanowski, P., Grave, E., Joulin, A., Mikolov, T.: Enriching word vectors with subword information. Trans. Assoc. Comput. Linguist. 5, 135\u2013146 (2017)","journal-title":"Trans. Assoc. Comput. Linguist."},{"key":"22_CR4","doi-asserted-by":"crossref","unstructured":"Chen, X., Jiang, B., Liu, W., Huang, Z., Fu, B., Chen, T., Yu, G.: Executing your commands via motion diffusion in latent space. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18000\u201318010 (2023)","DOI":"10.1109\/CVPR52729.2023.01726"},{"key":"22_CR5","first-page":"8780","volume":"34","author":"P Dhariwal","year":"2021","unstructured":"Dhariwal, P., Nichol, A.: Diffusion models beat GANs on image synthesis. Adv. Neural. Inf. Process. Syst. 34, 8780\u20138794 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"22_CR6","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T.: Transformers for image recognition at scale (2020). arXiv preprint arXiv:2010.11929"},{"key":"22_CR7","unstructured":"Driess, D., Xia, F., Sajjadi, M.S., Lynch, C., Chowdhery, A., Ichter, B., Wahid, A., Tompson, J., Vuong, Q., Yu, T., et\u00a0al.: Palm-e: an embodied multimodal language model (2023). arXiv preprint arXiv:2303.03378"},{"key":"22_CR8","doi-asserted-by":"crossref","unstructured":"Du, Y., Kips, R., Pumarola, A., Starke, S., Thabet, A., Sanakoyeu, A.: Avatars grow legs: generating smooth human motion from sparse tracking inputs with diffusion model. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 481\u2013490 (2023)","DOI":"10.1109\/CVPR52729.2023.00054"},{"key":"22_CR9","doi-asserted-by":"crossref","unstructured":"Ginosar, S., Bar, A., Kohavi, G., Chan, C., Owens, A., Malik, J.: Learning individual styles of conversational gesture. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3497\u20133506 (2019)","DOI":"10.1109\/CVPR.2019.00361"},{"key":"22_CR10","doi-asserted-by":"crossref","unstructured":"Habibie, I., Xu, W., Mehta, D., Liu, L., Seidel, H.P., Pons-Moll, G., Elgharib, M., Theobalt, C.: Learning speech-driven 3d conversational gestures from video. In: Proceedings of the 21st ACM International Conference on Intelligent Virtual Agents, pp. 101\u2013108 (2021)","DOI":"10.1145\/3472306.3478335"},{"key":"22_CR11","first-page":"6840","volume":"33","author":"J Ho","year":"2020","unstructured":"Ho, J., Jain, A., Abbeel, P.: Denoising diffusion probabilistic models. Adv. Neural. Inf. Process. Syst. 33, 6840\u20136851 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"issue":"1","key":"22_CR12","doi-asserted-by":"publisher","first-page":"79","DOI":"10.1162\/neco.1991.3.1.79","volume":"3","author":"RA Jacobs","year":"1991","unstructured":"Jacobs, R.A., Jordan, M.I., Nowlan, S.J., Hinton, G.E.: Adaptive mixtures of local experts. Neural Comput. 3(1), 79\u201387 (1991)","journal-title":"Neural Comput."},{"key":"22_CR13","unstructured":"Ji, L., Wei, P., Ren, Y., Liu, J., Zhang, C., Yin, X.: C2g2: controllable co-speech gesture generation with latent diffusion model (2023). arXiv preprint arXiv:2308.15016"},{"key":"22_CR14","doi-asserted-by":"crossref","unstructured":"Kucherenko, T., Jonell, P., Van\u00a0Waveren, S., Henter, G.E., Alexandersson, S., Leite, I., Kjellstr\u00f6m, H.: Gesticulator: a framework for semantically-aware speech-driven gesture generation. In: Proceedings of the 2020 International Conference on Multimodal Interaction, pp. 242\u2013250 (2020)","DOI":"10.1145\/3382507.3418815"},{"key":"22_CR15","doi-asserted-by":"crossref","unstructured":"Kucherenko, T., Jonell, P., Yoon, Y., Wolfert, P., Henter, G.E.: A large, crowdsourced evaluation of gesture generation systems on common data: the GENEA challenge 2020. In: 26th International Conference on Intelligent User Interfaces, pp. 11\u201321 (2021)","DOI":"10.1145\/3397481.3450692"},{"key":"22_CR16","doi-asserted-by":"crossref","unstructured":"Li, J., Kang, D., Pei, W., Zhe, X., Zhang, Y., He, Z., Bao, L.: Audio2gestures: generating diverse gestures from speech audio with conditional variational autoencoders. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 11293\u201311302 (2021)","DOI":"10.1109\/ICCV48922.2021.01110"},{"key":"22_CR17","unstructured":"Li, J., Li, D., Xiong, C., Hoi, S.: Blip: bootstrapping language-image pre-training for unified vision-language understanding and generation. In: International Conference on Machine Learning, pp. 12888\u201312900. PMLR (2022)"},{"key":"22_CR18","doi-asserted-by":"crossref","unstructured":"Li, R., Yang, S., Ross, D.A., Kanazawa, A.: Ai choreographer: music conditioned 3d dance generation with AIST++. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 13401\u201313412 (2021)","DOI":"10.1109\/ICCV48922.2021.01315"},{"key":"22_CR19","doi-asserted-by":"crossref","unstructured":"Liu, H., Zhu, Z., Iwamoto, N., Peng, Y., Li, Z., Zhou, Y., Bozkurt, E., Zheng, B.: Beat: a large-scale semantic and emotional multi-modal dataset for conversational gestures synthesis. In: European Conference on Computer Vision, pp. 612\u2013630. Springer (2022)","DOI":"10.1007\/978-3-031-20071-7_36"},{"key":"22_CR20","doi-asserted-by":"crossref","unstructured":"Liu, X., Wu, Q., Zhou, H., Xu, Y., Qian, R., Lin, X., Zhou, X., Wu, W., Dai, B., Zhou, B.: Learning hierarchical cross-modal association for co-speech gesture generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10462\u201310472 (2022)","DOI":"10.1109\/CVPR52688.2022.01021"},{"key":"22_CR21","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization (2017). arXiv preprint arXiv:1711.05101"},{"key":"22_CR22","first-page":"9564","volume":"35","author":"B Mustafa","year":"2022","unstructured":"Mustafa, B., Riquelme, C., Puigcerver, J., Jenatton, R., Houlsby, N.: Multimodal contrastive learning with LIMOE: the language-image mixture of experts. Adv. Neural. Inf. Process. Syst. 35, 9564\u20139576 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"22_CR23","unstructured":"Nichol, A., Dhariwal, P., Ramesh, A., Shyam, P., Mishkin, P., McGrew, B., Sutskever, I., Chen, M.: Glide: towards photorealistic image generation and editing with text-guided diffusion models (2021). arXiv preprint arXiv:2112.10741"},{"key":"22_CR24","unstructured":"Nichol, A.Q., Dhariwal, P.: Improved denoising diffusion probabilistic models. In: International Conference on Machine Learning, pp. 8162\u20138171. Proceedings of Machine Learning Research (2021)"},{"issue":"2","key":"22_CR25","doi-asserted-by":"publisher","first-page":"569","DOI":"10.1111\/cgf.14776","volume":"42","author":"S Nyatsanga","year":"2023","unstructured":"Nyatsanga, S., Kucherenko, T., Ahuja, C., Henter, G.E., Neff, M.: A comprehensive review of data-driven co-speech gesture generation. Comput. Graph. Forum 42(2), 569\u2013596 (2023)","journal-title":"Comput. Graph. Forum"},{"key":"22_CR26","doi-asserted-by":"crossref","unstructured":"Pan, X., Qin, P., Li, Y., Xue, H., Chen, W.: Synthesizing coherent story with auto-regressive latent diffusion models. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 2920\u20132930 (2024)","DOI":"10.1109\/WACV57701.2024.00290"},{"key":"22_CR27","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., Ommer, B.: High-resolution image synthesis with latent diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10684\u201310695 (2022)","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"22_CR28","unstructured":"Shazeer, N., Mirhoseini, A., Maziarz, K., Davis, A., Le, Q., Hinton, G., Dean, J.: Outrageously large neural networks: the sparsely-gated mixture-of-experts layer (2017). arXiv preprint arXiv:1701.06538"},{"key":"22_CR29","unstructured":"Van Den\u00a0Oord, A., Vinyals, O., et\u00a0al.: Neural discrete representation learning. Adv. Neural Inform. Process. Syst. 30 (2017)"},{"key":"22_CR30","doi-asserted-by":"crossref","unstructured":"Xu, P., Zhu, X., Clifton, D.A.: Multimodal learning with transformers: a survey. In: IEEE Transactions on Pattern Analysis and Machine Intelligence (2023)","DOI":"10.1109\/TPAMI.2023.3275156"},{"key":"22_CR31","doi-asserted-by":"crossref","unstructured":"Xu, X., Wang, Z., Zhang, G., Wang, K., Shi, H.: Versatile diffusion: text, images and variations all in one diffusion model. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7754\u20137765 (2023)","DOI":"10.1109\/ICCV51070.2023.00713"},{"key":"22_CR32","doi-asserted-by":"crossref","unstructured":"Yang, S., Wu, Z., Li, M., Zhang, Z., Hao, L., Bao, W., Zhuang, H.: Qpgesture: quantization-based and phase-guided motion matching for natural speech-driven gesture generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2321\u20132330 (2023)","DOI":"10.1109\/CVPR52729.2023.00230"},{"key":"22_CR33","doi-asserted-by":"crossref","unstructured":"Yang, S., Wu, Z., Li, M., Zhao, M., Lin, J., Chen, L., Bao, W.: The ReprGesture entry to the GENEA challenge 2022. In: Proceedings of the 2022 International Conference on Multimodal Interaction, pp. 758\u2013763 (2022)","DOI":"10.1145\/3536221.3558066"},{"key":"22_CR34","doi-asserted-by":"crossref","unstructured":"Yang, S., Xue, H., Zhang, Z., Li, M., Wu, Z., Wu, X., Xu, S., Dai, Z.: The DiffuseStyleGesture+ entry to the GENEA challenge 2023. In: Proceedings of the 25th International Conference on Multimodal Interaction, pp. 779\u2013785 (2023)","DOI":"10.1145\/3577190.3616114"},{"key":"22_CR35","doi-asserted-by":"crossref","unstructured":"Ye, S., Wen, Y.H., Sun, Y., He, Y., Zhang, Z., Wang, Y., He, W., Liu, Y.J.: Audio-driven stylized gesture generation with flow-based model. In: European Conference on Computer Vision, pp. 712\u2013728. Springer (2022)","DOI":"10.1007\/978-3-031-20065-6_41"},{"key":"22_CR36","doi-asserted-by":"crossref","unstructured":"Yin, L., Wang, Y., He, T., Liu, J., Zhao, W., Li, B., Jin, X., Li, J.: EMoG: synthesizing emotive co-speech 3d gesture with diffusion model (2023)","DOI":"10.2139\/ssrn.4818829"},{"issue":"6","key":"22_CR37","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3414685.3417838","volume":"39","author":"Y Yoon","year":"2020","unstructured":"Yoon, Y., Cha, B., Lee, J.H., Jang, M., Lee, J., Kim, J., Lee, G.: Speech gesture generation from the trimodal context of text, audio, and speaker identity. ACM Trans. Graph. (TOG) 39(6), 1\u201316 (2020)","journal-title":"ACM Trans. Graph. (TOG)"},{"key":"22_CR38","doi-asserted-by":"crossref","unstructured":"Zhang, M., Liu, C., Chen, Y., Lei, Z., Wang, M.: Music-to-dance generation with multiple conformer. In: Proceedings of the 2022 International Conference on Multimedia Retrieval, pp. 34\u201338 (2022)","DOI":"10.1145\/3512527.3531430"},{"key":"22_CR39","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Cai, R., Chen, T., Zhang, G., Zhang, H., Chen, P.Y., Chang, S., Wang, Z., Liu, S.: Robust mixture-of-expert training for convolutional neural networks. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 90\u2013101 (2023)","DOI":"10.1109\/ICCV51070.2023.00015"},{"key":"22_CR40","unstructured":"Zhenxing, M., Xu, D.: Switch-nerf: learning scene decomposition with mixture of experts for large-scale EEURAL radiance fields. In: The Eleventh International Conference on Learning Representations (2022)"},{"key":"22_CR41","doi-asserted-by":"crossref","unstructured":"Zhu, L., Liu, X., Liu, X., Qian, R., Liu, Z., Yu, L.: Taming diffusion models for audio-driven co-speech gesture generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10544\u201310553 (2023)","DOI":"10.1109\/CVPR52729.2023.01016"}],"container-title":["Lecture Notes in Computer Science","Pattern Recognition and Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-97-8508-7_22","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,2]],"date-time":"2024-11-02T06:16:18Z","timestamp":1730528178000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-97-8508-7_22"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,3]]},"ISBN":["9789819785070","9789819785087"],"references-count":41,"URL":"https:\/\/doi.org\/10.1007\/978-981-97-8508-7_22","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2024,11,3]]},"assertion":[{"value":"3 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"PRCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Chinese Conference on Pattern Recognition and Computer Vision  (PRCV)","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Urumqi","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18 October 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"7","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ccprcv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/2024.prcv.cn\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}