{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,27]],"date-time":"2025-03-27T23:15:05Z","timestamp":1743117305274,"version":"3.40.3"},"publisher-location":"Cham","reference-count":25,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783030057091"},{"type":"electronic","value":"9783030057107"}],"license":[{"start":{"date-parts":[[2018,12,8]],"date-time":"2018-12-08T00:00:00Z","timestamp":1544227200000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2019]]},"DOI":"10.1007\/978-3-030-05710-7_6","type":"book-chapter","created":{"date-parts":[[2018,12,7]],"date-time":"2018-12-07T12:48:55Z","timestamp":1544186935000},"page":"68-79","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Deep Neural Network Based 3D Articulatory Movement Prediction Using Both Text and Audio Inputs"],"prefix":"10.1007","author":[{"given":"Lingyun","family":"Yu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jun","family":"Yu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qiang","family":"Ling","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2018,12,8]]},"reference":[{"issue":"5","key":"6_CR1","doi-asserted-by":"publisher","first-page":"991","DOI":"10.1109\/TCYB.2014.2341737","volume":"45","author":"J Yu","year":"2015","unstructured":"Yu, J., Wang, Z.-F.: A video, text, and speech-driven realistic 3-D virtual head for human-machine interface. IEEE Trans. Cybern. 45(5), 991\u20131002 (2015)","journal-title":"IEEE Trans. Cybern."},{"issue":"7","key":"6_CR2","doi-asserted-by":"publisher","first-page":"1254","DOI":"10.1109\/TMM.2009.2030637","volume":"11","author":"G Zhao","year":"2009","unstructured":"Zhao, G., Barnard, M., Pietikainen, M.: Lipreading with local spatiotemporal descriptors. IEEE Trans. Multimedia 11(7), 1254\u20131265 (2009)","journal-title":"IEEE Trans. Multimedia"},{"issue":"6","key":"6_CR3","doi-asserted-by":"publisher","first-page":"591","DOI":"10.1109\/TMM.2010.2052239","volume":"12","author":"G Fanelli","year":"2010","unstructured":"Fanelli, G., Gall, J., Romsdorfer, H., Weise, T., Van Gool, L.: A 3-D audio-visual corpus of affective communication. IEEE Trans. Multimedia 12(6), 591\u2013598 (2010)","journal-title":"IEEE Trans. Multimedia"},{"key":"6_CR4","unstructured":"Mitra, V.: Articulatory information for robust speech recognition, Ph.D. dissertation (2010)"},{"issue":"3","key":"6_CR5","doi-asserted-by":"publisher","first-page":"215","DOI":"10.1016\/j.specom.2007.09.001","volume":"50","author":"T Toda","year":"2008","unstructured":"Toda, T., Black, A.W., Tokuda, K.: Statistical mapping between articulatory movements and acoustic spectrum using a gaussian mixture model. Speech Commun. 50(3), 215\u2013227 (2008)","journal-title":"Speech Commun."},{"key":"6_CR6","doi-asserted-by":"publisher","first-page":"245","DOI":"10.1109\/LSP.2008.917004","volume":"15","author":"L Zhang","year":"2008","unstructured":"Zhang, L., Renals, S.: Acoustic-articulatory modeling with the trajectory HMM. IEEE Signal Process. Lett. 15, 245\u2013248 (2008)","journal-title":"IEEE Signal Process. Lett."},{"key":"6_CR7","doi-asserted-by":"crossref","unstructured":"Deng, L., Hinton, G., Kingsbury, B.: New types of deep neural network learning for speech recognition and related applications: an overview. In: IEEE International Conference on Acoustics, Speech and Signal Processing, pp. 8599\u20138603 (2013)","DOI":"10.1109\/ICASSP.2013.6639344"},{"key":"6_CR8","doi-asserted-by":"crossref","unstructured":"Qian, Y., Fan, Y., Hu, W., Soong, F.K.: On the training aspects of deep neural network (DNN) for parametric TTS synthesis. In: IEEE International Conference on Acoustics, Speech and Signal Processing, pp. 3829\u20133833 (2014)","DOI":"10.1109\/ICASSP.2014.6854318"},{"key":"6_CR9","doi-asserted-by":"crossref","unstructured":"Uria, B., Murray, I., Renals, S., Richmond, K.: Deep architectures for articulatory inversion. In: Thirteenth Annual Conference of the International Speech Communication Association (2012)","DOI":"10.21437\/Interspeech.2012-263"},{"key":"6_CR10","unstructured":"Uria, B., Renals, S., Richmond, K.: A deep neural network for acoustic-articulatory speech inversion. In: NIPS 2011 Workshop on Deep Learning and Unsupervised Feature Learning (2011)"},{"key":"6_CR11","doi-asserted-by":"crossref","unstructured":"Zhu, P., Xie, L., Chen, Y.: Articulatory movement prediction using deep bidirectional long short-term memory based recurrent neural networks and word\/phone embeddings. In: INTERSPEECH, pp. 2192\u20132196 (2015)","DOI":"10.21437\/Interspeech.2015-493"},{"key":"6_CR12","doi-asserted-by":"crossref","unstructured":"Wei, Z., Wu, Z., Xie, L.: Predicting articulatory movement from text using deep architecture with stacked bottleneck features. In: 2016 Asia-Pacific Signal and Information Processing Association Annual Summit and Conference (APSIPA), pp. 1\u20136. IEEE (2016)","DOI":"10.1109\/APSIPA.2016.7820703"},{"issue":"10","key":"6_CR13","doi-asserted-by":"publisher","first-page":"834","DOI":"10.1016\/j.specom.2010.06.006","volume":"52","author":"ZH Ling","year":"2010","unstructured":"Ling, Z.H., Richmond, K., Yamagishi, J.: An analysis of HMM-based prediction of articulatory movements. Speech Commun. 52(10), 834\u2013846 (2010)","journal-title":"Speech Commun."},{"issue":"10","key":"6_CR14","doi-asserted-by":"publisher","first-page":"1533","DOI":"10.1109\/TASLP.2014.2339736","volume":"22","author":"O Abdel-Hamid","year":"2014","unstructured":"Abdel-Hamid, O., Mohamed, A.-R., Jiang, H., Deng, L., Penn, G., Yu, D.: Convolutional neural networks for speech recognition. IEEE\/ACM Trans. Audio, Speech, Lang. Process. 22(10), 1533\u20131545 (2014)","journal-title":"IEEE\/ACM Trans. Audio, Speech, Lang. Process."},{"key":"6_CR15","doi-asserted-by":"crossref","unstructured":"Hu, J., Shen, L., Sun, G.: Squeeze-and-excitation networks, arXiv preprint arXiv:1709.01507 , vol. 7 (2017)","DOI":"10.1109\/CVPR.2018.00745"},{"key":"6_CR16","doi-asserted-by":"crossref","unstructured":"Yu, D., Seltzer, M.L.: Improved bottleneck features using pretrained deep neural networks. In: Twelfth Annual Conference of the International Speech Communication Association (2011)","DOI":"10.21437\/Interspeech.2011-91"},{"key":"6_CR17","doi-asserted-by":"crossref","unstructured":"Cheng, X., Li, X., Tai, Y., Yang, J.: SESR: Single image super resolution with recursive squeeze and excitation networks, arXiv preprint arXiv:1801.10319 (2018)","DOI":"10.1109\/ICPR.2018.8546130"},{"issue":"1","key":"6_CR18","doi-asserted-by":"publisher","first-page":"26","DOI":"10.1016\/0093-934X(87)90058-7","volume":"31","author":"PW Sch\u00f6nle","year":"1987","unstructured":"Sch\u00f6nle, P.W., Gr\u00e4be, K., Wenig, P., H\u00f6hne, J., Schrader, J., Conrad, B.: Electromagnetic articulography: use of alternating magnetic fields for trackingmovements of multiple points inside and outside the vocal tract. Brain Lang. 31(1), 26\u201335 (1987)","journal-title":"Brain Lang."},{"key":"6_CR19","doi-asserted-by":"crossref","unstructured":"Wu, Z., Watts, O., King, S.: Merlin: an open source neural network speech synthesis system. Proc. SSW, Sunnyvale, USA (2016)","DOI":"10.21437\/SSW.2016-33"},{"key":"6_CR20","doi-asserted-by":"crossref","unstructured":"Jia, Y., Shelhamer, E., Donahue, J., et al.: Caffe: Convolutional architecture for fast feature embedding. In: Proceedings of the 22nd ACM International Conference on Multimedia, pp. 675\u2013678. ACM (2014)","DOI":"10.1145\/2647868.2654889"},{"key":"6_CR21","doi-asserted-by":"crossref","unstructured":"Ling, Z.-H., Richmond, K., Yamagishi, J.: HMM-based text-to-articulatory-movement prediction and analysis of critical articulators. In: Proc. Interspeech, pp. 2194\u20132197, Sep. 2010","DOI":"10.21437\/Interspeech.2010-604"},{"key":"6_CR22","doi-asserted-by":"crossref","unstructured":"Richmond, K.: Preliminary inversion mapping results with a new EMA corpus (2009)","DOI":"10.21437\/Interspeech.2009-724"},{"key":"6_CR23","doi-asserted-by":"crossref","unstructured":"Liu, P., Yu, Q., Wu, Z., Kang, S., Meng, H., Cai, L.: A deep recurrent approach for acoustic-to-articulatory inversion. In: 2015 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 4450\u20134454. IEEE (2015)","DOI":"10.1109\/ICASSP.2015.7178812"},{"key":"6_CR24","doi-asserted-by":"crossref","unstructured":"Yu, J., Li, A., Hu, F., et al.: Data-driven 3D visual pronunciation of Chinese IPA for language learning. In: 2013 International Conference on Oriental COCOSDA Held Jointly with 2013 Conference on Asian Spoken Language Research and Evaluation (O-COCOSDA\/CASLRE), pp. 1\u20136. IEEE (2013)","DOI":"10.1109\/ICSDA.2013.6709888"},{"issue":"3","key":"6_CR25","doi-asserted-by":"publisher","first-page":"176","DOI":"10.1016\/j.intcom.2009.12.002","volume":"22","author":"S Marcos","year":"2010","unstructured":"Marcos, S., G\u00f3mez-Garc\u00eda-Bermejo, J., Zalama, E.: A realistic, virtual head for human-computer interaction. Interact. Comput. 22(3), 176\u2013192 (2010)","journal-title":"Interact. Comput."}],"container-title":["Lecture Notes in Computer Science","MultiMedia Modeling"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-05710-7_6","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,9,11]],"date-time":"2023-09-11T22:46:08Z","timestamp":1694472368000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-030-05710-7_6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,12,8]]},"ISBN":["9783030057091","9783030057107"],"references-count":25,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-05710-7_6","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2018,12,8]]},"assertion":[{"value":"8 December 2018","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"MMM","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Multimedia Modeling","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Thessaloniki","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Greece","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2019","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"8 January 2019","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"11 January 2019","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"25","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"mmm2019","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/mmm2019.iti.gr\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double blind for full papers and workshop papers, single blind for other paper types","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"EasyChair","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"204","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"96","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"47% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"2.67","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"3","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}},{"value":"6 demonstration papers, 5 industry papers, 6 workshop papers, and 6 Video Browser Showdown papers were also accepted.","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information"}}]}}