{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T16:34:43Z","timestamp":1783701283707,"version":"3.55.0"},"reference-count":101,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2019,10,8]],"date-time":"2019-10-08T00:00:00Z","timestamp":1570492800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2019,10,8]],"date-time":"2019-10-08T00:00:00Z","timestamp":1570492800000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2020,4]]},"DOI":"10.1007\/s11263-019-01245-6","type":"journal-article","created":{"date-parts":[[2019,10,8]],"date-time":"2019-10-08T08:03:11Z","timestamp":1570521791000},"page":"855-872","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":148,"title":["Modeling Human Motion with Quaternion-Based Neural Networks"],"prefix":"10.1007","volume":"128","author":[{"given":"Dario","family":"Pavllo","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Christoph","family":"Feichtenhofer","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5974-4459","authenticated-orcid":false,"given":"Michael","family":"Auli","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"David","family":"Grangier","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2019,10,8]]},"reference":[{"key":"1245_CR1","doi-asserted-by":"crossref","unstructured":"Akhter, I., & Black M. J. (2015). Pose-conditioned joint angle limits for 3d human pose reconstruction. In 2015 IEEE conference on computer vision and pattern recognition (CVPR).","DOI":"10.1109\/CVPR.2015.7298751"},{"key":"1245_CR2","doi-asserted-by":"crossref","unstructured":"Arikan O., Forsyth D. A., & O\u2019Brien J. F. (2003). Motion synthesis from annotations. In ACM transactions on graphics (SIGGRAPH).","DOI":"10.1145\/882262.882284"},{"key":"1245_CR3","unstructured":"Ba, J. L., Kiros, J. R., & Hinton, G. E. (2016). Layer normalization. arXiv preprint arXiv:1607.06450 ."},{"key":"1245_CR4","doi-asserted-by":"crossref","DOI":"10.1093\/oso\/9780195073591.001.0001","volume-title":"Simulating humans: Computer graphics animation and control","author":"NI Badler","year":"1993","unstructured":"Badler, N. I., Phillips, C. B., & Webber, B. L. (1993). Simulating humans: Computer graphics animation and control. Oxford: Oxford University Press."},{"key":"1245_CR5","unstructured":"Bahdanau, D., Cho, K., & Bengio, Y. (2015). Neural machine translation by jointly learning to align and translate. In International conference on learning representations (ICLR)."},{"key":"1245_CR6","first-page":"1137","volume":"3","author":"Y Bengio","year":"2003","unstructured":"Bengio, Y., Ducharme, R., Vincent, P., & Jauvin, C. (2003). A neural probabilistic language model. Journal of Machine Learning Research, 3, 1137\u20131155.","journal-title":"Journal of Machine Learning Research"},{"key":"1245_CR7","unstructured":"Bengio, S., Vinyals, O., Jaitly, N., & Shazeer, N. (2015). Scheduled sampling for sequence prediction with recurrent neural networks. In Advances in neural information processing systems (NIPS)."},{"key":"1245_CR8","doi-asserted-by":"crossref","unstructured":"B\u00fctepage, J., Black, M. J., Kragic, D., & Kjellstr\u00f6m, H. (2017). Deep representation learning for human motion prediction and classification. In Conference on computer vision and pattern recognition (CVPR).","DOI":"10.1109\/CVPR.2017.173"},{"key":"1245_CR9","doi-asserted-by":"crossref","unstructured":"B\u00fctepage, J., Kjellstr\u00f6m, H., & Kragic, D. (2018). Anticipating many futures: Online human motion prediction and generation for human\u2013robot interaction. In 2018 IEEE international conference on robotics and automation (ICRA), pp. 1\u20139.","DOI":"10.1109\/ICRA.2018.8460651"},{"key":"1245_CR10","doi-asserted-by":"crossref","unstructured":"Byravan, A., & Fox, D. (2017). SE3-nets: Learning rigid body motion using deep neural networks. In IEEE international conference on robotics and automation (ICRA).","DOI":"10.1109\/ICRA.2017.7989023"},{"key":"1245_CR11","doi-asserted-by":"crossref","unstructured":"Chao, Y. W., Yang, J., Price, B. L., Cohen, S., & Deng, J. (2017). Forecasting human dynamics from static images. In Conference on computer vision and pattern recognition (CVPR).","DOI":"10.1109\/CVPR.2017.388"},{"key":"1245_CR12","doi-asserted-by":"crossref","unstructured":"Cho, K., Van\u00a0Merri\u00ebnboer, B., Gulcehre, C., Bahdanau, D., Bougares, F., Schwenk, H., et al. (2014). Learning phrase representations using RNN encoder-decoder for statistical machine translation. In Conference on empirical methods in natural language processing (EMNLP).","DOI":"10.3115\/v1\/D14-1179"},{"key":"1245_CR13","unstructured":"Chung, J., Gulcehre, C., Cho, K., & Bengio, Y. (2014). Empirical evaluation of gated recurrent neural networks on sequence modeling. In NIPS deep learning and representation learning workshop."},{"key":"1245_CR14","unstructured":"CMU (2003) CMU graphics lab motion capture database. http:\/\/mocap.cs.cmu.edu . The database was created with funding from NSF EIA-0196217."},{"key":"1245_CR15","unstructured":"Collobert, R., Puhrsch, C., & Synnaeve, G. (2016). Wav2letter: an end-to-end convnet-based speech recognition system. arXiv:1609.03193 ."},{"key":"1245_CR16","volume-title":"Image processing and analysis","author":"TF Cootes","year":"2000","unstructured":"Cootes, T. F. (2000). An introduction to active shape models, chap 7. In R. Baldock & J. Graham (Eds.), Image processing and analysis. Oxford: Oxford University Press."},{"key":"1245_CR17","doi-asserted-by":"publisher","first-page":"144","DOI":"10.1016\/j.mechmachtheory.2015.03.004","volume":"92","author":"JS Dai","year":"2015","unstructured":"Dai, J. S. (2015). Euler\u2013Rodrigues formula variations, quaternion conjugation and intrinsic connections. Mechanism and Machine Theory, 92, 144\u2013152.","journal-title":"Mechanism and Machine Theory"},{"key":"1245_CR18","unstructured":"Dauphin, Y. N., Fan, A., Auli, M., & Grangier, D. (2017). Language modeling with gated convolutional networks. In: Proceedings of ICLR."},{"key":"1245_CR19","unstructured":"Du, Y., Wang, W., & Wang, L. (2015). Hierarchical recurrent neural network for skeleton based action recognition. In Conference on computer vision and pattern recognition (CVPR), pp. 1110\u20131118."},{"key":"1245_CR20","volume-title":"3D math primer for graphics and game development","author":"F Dunn","year":"2010","unstructured":"Dunn, F., Parberry, I., et al. (2010). 3D math primer for graphics and game development. Burlington: Jones & Bartlett Publishers."},{"issue":"2\u20133","key":"1245_CR21","first-page":"77","volume":"1","author":"DA Forsyth","year":"2006","unstructured":"Forsyth, D. A., Arikan, O., Ikemoto, L., O\u2019Brien, J., Ramanan, D., et al. (2006). Computational studies of human motion: Part 1, tracking and motion synthesis. Foundations and Trends in Computer Graphics and Vision, 1(2\u20133), 77\u2013254.","journal-title":"Foundations and Trends in Computer Graphics and Vision"},{"key":"1245_CR22","doi-asserted-by":"crossref","unstructured":"Fragkiadaki, K., Levine, S., Felsen, P., & Malik, J. (2015). Recurrent network models for human dynamics. In Conference on vision and pattern recognition (CVPR)","DOI":"10.1109\/ICCV.2015.494"},{"key":"1245_CR23","doi-asserted-by":"crossref","unstructured":"Gaudet, C. J., & Maida, A. S. (2018). Deep quaternion networks. In International joint conference on neural networks (IJCNN), IEEE, pp. 1\u20138.","DOI":"10.1109\/IJCNN.2018.8489651"},{"key":"1245_CR24","unstructured":"Gehring, J., Auli, M., Grangier, D., Yarats, D., & Dauphin, Y. N. (2017). Convolutional sequence to sequence learning. In International conference on machine learning (ICML)."},{"key":"1245_CR25","doi-asserted-by":"crossref","unstructured":"Ghosh, P., Song, J., Aksan, E., & Hilliges, O. (2017). Learning human motion models for long-term predictions. In International conference on 3D vision.","DOI":"10.1109\/3DV.2017.00059"},{"issue":"12","key":"1245_CR26","doi-asserted-by":"publisher","first-page":"1966","DOI":"10.3390\/s16121966","volume":"16","author":"W Gong","year":"2016","unstructured":"Gong, W., Zhang, X., Gonz\u00e0lez, J., Sobral, A., Bouwmans, T., Tu, C., et al. (2016). Human pose estimation from monocular images: A comprehensive survey. Sensors, 16(12), 1966.","journal-title":"Sensors"},{"key":"1245_CR27","unstructured":"Gopalakrishnan, A., Mali, A., Kifer, D., Giles, C. L., & Ororbia, A. G. (2018). A neural temporal model for human motion prediction. arXiv:1809.03036 ."},{"key":"1245_CR28","doi-asserted-by":"publisher","first-page":"29","DOI":"10.1080\/10867651.1998.10487493","volume":"3","author":"FS Grassia","year":"1998","unstructured":"Grassia, F. S. (1998). Practical parameterization of rotations using the exponential map. Journal of Graphics Tools, 3, 29\u201348.","journal-title":"Journal of Graphics Tools"},{"key":"1245_CR29","doi-asserted-by":"crossref","unstructured":"Gu, C., Sun, C., Ross, D. A., Vondrick, C., Pantofaru, C., & Li, Y., et al. (2018). AVA: A video dataset of spatio-temporally localized atomic visual actions. In Computer vision and pattern recognition (CVPR).","DOI":"10.1109\/CVPR.2018.00633"},{"key":"1245_CR30","doi-asserted-by":"crossref","unstructured":"Gui, L. Y., Wang, Y. X., Liang, X., & Moura, J. M. (2018). Adversarial geometry-aware human motion prediction. In European conference on computer vision (ECCV).","DOI":"10.1007\/978-3-030-01225-0_48"},{"key":"1245_CR31","doi-asserted-by":"publisher","first-page":"85","DOI":"10.1016\/j.cviu.2017.01.011","volume":"158","author":"F Han","year":"2017","unstructured":"Han, F., Reily, B., Hoff, W., & Zhang, H. (2017). Space-time representation of people based on 3D skeletal data: A review. Computer Vision and Image Understanding, 158, 85\u2013105.","journal-title":"Computer Vision and Image Understanding"},{"key":"1245_CR32","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2016) Deep residual learning for image recognition. In Conference on computer vision and pattern recognition (CVPR), pp. 770\u2013778.","DOI":"10.1109\/CVPR.2016.90"},{"key":"1245_CR33","doi-asserted-by":"publisher","first-page":"189","DOI":"10.1016\/j.cviu.2005.01.005","volume":"99","author":"L Herda","year":"2005","unstructured":"Herda, L., Urtasun, R., & Fua, P. (2005). Hierarchical implicit surface joint limits for human body tracking. Computer Vision and Image Understanding, 99, 189\u2013209.","journal-title":"Computer Vision and Image Understanding"},{"key":"1245_CR34","doi-asserted-by":"publisher","first-page":"82","DOI":"10.1109\/MSP.2012.2205597","volume":"29","author":"G Hinton","year":"2012","unstructured":"Hinton, G., Deng, L., Yu, D., Dahl, G., Rahman Mohamed, A., et al. (2012). Deep neural networks for acoustic modeling in speech recognition. IEEE Signal Processing Magazine., 29, 82\u201397.","journal-title":"IEEE Signal Processing Magazine."},{"key":"1245_CR35","volume-title":"A field guide to dynamical recurrent neural networks","author":"S Hochreiter","year":"2001","unstructured":"Hochreiter, S., Bengio, Y., Frasconi, P., & Schmidhuber, J. (2001). Gradient flow in recurrent nets: The difficulty of learning long-term dependencies. In S. C. Kremer & J. F. Kolen (Eds.), A field guide to dynamical recurrent neural networks. IEEE Press: Piscataway."},{"key":"1245_CR36","first-page":"42","volume":"36","author":"D Holden","year":"2017","unstructured":"Holden, D., Komura, T., & Saito, J. (2017). Phase-functioned neural networks for character control. ACM Transaction on Graphics (SIGGRAPH), 36, 42.","journal-title":"ACM Transaction on Graphics (SIGGRAPH)"},{"key":"1245_CR37","first-page":"138","volume":"35","author":"D Holden","year":"2016","unstructured":"Holden, D., Saito, J., & Komura, T. (2016). A deep learning framework for character motion synthesis and editing. ACM Transaction on Graphics (SIGGRAPH), 35, 138.","journal-title":"ACM Transaction on Graphics (SIGGRAPH)"},{"key":"1245_CR38","doi-asserted-by":"crossref","unstructured":"Holden, D., Saito, J., Komura, T., & Joyce, T. (2015). Learning motion manifolds with convolutional autoencoders. In SIGGRAPH Asia 2015 technical briefs.","DOI":"10.1145\/2820903.2820918"},{"issue":"2","key":"1245_CR39","doi-asserted-by":"publisher","first-page":"155","DOI":"10.1007\/s10851-009-0161-2","volume":"35","author":"DQ Huynh","year":"2009","unstructured":"Huynh, D. Q. (2009). Metrics for 3d rotations: Comparison and analysis. Journal of Mathematical Imaging and Vision, 35(2), 155\u2013164. https:\/\/doi.org\/10.1007\/s10851-009-0161-2 .","journal-title":"Journal of Mathematical Imaging and Vision"},{"key":"1245_CR40","unstructured":"Ioffe, S., & Szegedy, C. (2015). Batch normalization: Accelerating deep network training by reducing internal covariate shift. In International conference on machine learning (ICML), pp. 448\u2013456."},{"key":"1245_CR41","doi-asserted-by":"crossref","unstructured":"Ionescu, C., Li, F., & Sminchisescu, C. (2011). Latent structured models for human pose estimation. In International conference on computer vision (ICCV).","DOI":"10.1109\/ICCV.2011.6126500"},{"key":"1245_CR42","doi-asserted-by":"publisher","first-page":"1325","DOI":"10.1109\/TPAMI.2013.248","volume":"36","author":"C Ionescu","year":"2014","unstructured":"Ionescu, C., Papava, D., Olaru, V., & Sminchisescu, C. (2014). Human3.6m: Large scale datasets and predictive methods for 3D human sensing in natural environments. IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI), 36, 1325\u20131339.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI)"},{"key":"1245_CR43","doi-asserted-by":"crossref","unstructured":"Jain, A., Zamir, A. R., Savarese, S., Saxena, A. (2016). Structural-RNN: Deep learning on spatio-temporal graphs. In Conference on computer vision and pattern recognition (CVPR).","DOI":"10.1109\/CVPR.2016.573"},{"key":"1245_CR44","unstructured":"Kiasari, M. A., Moirangthem, D. S., & Lee, M. (2018). Human action generation with generative adversarial networks. arXiv:1805.10416 ."},{"key":"1245_CR45","unstructured":"Kingma, D. P., & Ba, J. (2014). Adam: A method for stochastic optimization. In International conference on learning represention (ICLR)."},{"key":"1245_CR46","doi-asserted-by":"crossref","unstructured":"Kitani, K. M., Ziebart, B. D., Bagnell, J. A., & Hebert, M. (2012a). Activity forecasting. In European conference on computer vision (ECCV).","DOI":"10.1007\/978-3-642-33765-9_15"},{"key":"1245_CR47","doi-asserted-by":"crossref","unstructured":"Kitani, K. M., Ziebart, B. D., Bagnell, J. A., & Hebert, M. (2012b) Activity forecasting. In European conference on computer vision (ECCV).","DOI":"10.1007\/978-3-642-33765-9_15"},{"key":"1245_CR48","doi-asserted-by":"publisher","first-page":"14","DOI":"10.1109\/TPAMI.2015.2430335","volume":"38","author":"HS Koppula","year":"2016","unstructured":"Koppula, H. S., & Saxena, A. (2016). Anticipating human activities using object affordances for reactive robotic response. Transaction on Pattern Analysis and Machine Intelligence (TPAMI), 38, 14\u201339.","journal-title":"Transaction on Pattern Analysis and Machine Intelligence (TPAMI)"},{"key":"1245_CR49","unstructured":"Krizhevsky, A., Sutskever, I., & Hinton, G. E. (2012). Imagenet classification with deep convolutional neural networks. In F. Pereira, C. J. C. Burges, L. Bottou, & K. Q. Weinberger (Eds.), Advances in Neural Information Processing Systems 25 (pp. 1097\u20131105). Curran Associates, Inc. http:\/\/papers.nips.cc\/paper\/4824-imagenet-classification-with-deep-convolutional-neural-networks.pdf ."},{"issue":"4","key":"1245_CR50","doi-asserted-by":"publisher","first-page":"205","DOI":"10.22266\/ijies2017.0831.22","volume":"10","author":"S Kumar","year":"2017","unstructured":"Kumar, S., & Tripathi, B. K. (2017). Machine learning with resilient propagation in quaternionic domain. International Journal of Intelligent Engineering & Systems, 10(4), 205\u2013216.","journal-title":"International Journal of Intelligent Engineering & Systems"},{"key":"1245_CR51","doi-asserted-by":"crossref","unstructured":"Lan, T., Chen, T. C., & Savarese, S. (2014). A hierarchical representation for future action prediction. In European conference on computer vision (ECCV).","DOI":"10.1007\/978-3-319-10578-9_45"},{"key":"1245_CR52","doi-asserted-by":"publisher","first-page":"150","DOI":"10.1017\/CBO9780511546877","volume-title":"Planning algorithms","author":"SM LaValle","year":"2006","unstructured":"LaValle, S. M. (2006). Planning algorithms (pp. 150\u2013152). Cambridge: Cambridge University Press."},{"key":"1245_CR53","doi-asserted-by":"crossref","unstructured":"Lehrmann, A. M., Gehler, P. V., & Nowozin, S. (2014). Efficient nonlinear Markov models for human motion. In Conference on computer vision and pattern recognition (CVPR).","DOI":"10.1109\/CVPR.2014.171"},{"key":"1245_CR54","doi-asserted-by":"crossref","unstructured":"Li, C., Zhang, Z., Sun\u00a0Lee, W., & Hee\u00a0Lee, G. (2018a). Convolutional sequence to sequence model for human dynamics. In The IEEE conference on computer vision and pattern recognition (CVPR).","DOI":"10.1109\/CVPR.2018.00548"},{"key":"1245_CR55","unstructured":"Li, Z., Zhou, Y., Xiao, S., He, C., & Li, H. (2018b). Auto-conditioned LSTM network for extended complex human motion synthesis. In International conference on learning representations (ICLR)."},{"key":"1245_CR56","unstructured":"Lin, X., & Amer, M. R. (2018) Human motion modeling using dvgans. arXiv:1804.10652 ."},{"key":"1245_CR57","doi-asserted-by":"publisher","first-page":"1071","DOI":"10.1145\/1073204.1073314","volume":"24","author":"CK Liu","year":"2005","unstructured":"Liu, C. K., Hertzmann, A., & Popovi\u0107, Z. (2005). Learning physics-based motion style with nonlinear inverse optimization. ACM Transaction on Graphics (SIGGRAPH), 24, 1071\u20131081.","journal-title":"ACM Transaction on Graphics (SIGGRAPH)"},{"key":"1245_CR58","doi-asserted-by":"crossref","unstructured":"Liu, J., Shahroudy, A., Xu, D., & Wang, G. (2016). Spatio-temporal LSTM with trust gates for 3D human action recognition. In European conference on computer vision (ECCV).","DOI":"10.1007\/978-3-319-46487-9_50"},{"key":"1245_CR59","unstructured":"Luc, P., Couprie, C., Lecun, Y., & Verbeek, J. (2018). Predicting future instance segmentations by forecasting convolutional features. arXiv preprint arXiv:1803.11496 ."},{"key":"1245_CR60","doi-asserted-by":"crossref","unstructured":"Luc, P., Neverova, N., Couprie, C., Verbeek, J., & LeCun, Y. (2017) Predicting deeper into the future of semantic segmentation. In International conference in computer vision (ICCV).","DOI":"10.1109\/ICCV.2017.77"},{"key":"1245_CR61","doi-asserted-by":"crossref","unstructured":"Martinez, J., Black, M. J., & Romero, J. (2017). On human motion prediction using recurrent neural networks. In Conference on vision and pattern recognition (CVPR).","DOI":"10.1109\/CVPR.2017.497"},{"key":"1245_CR62","unstructured":"Mathieu, M., Couprie, C., & LeCun, Y. (2016). Deep multi-scale video prediction beyond mean square error. In International conference on learning representations (ICLR)."},{"key":"1245_CR63","unstructured":"McCarthy, J. (1990). An introduction to theoretical kinematics. MIT Press. https:\/\/books.google.ca\/books?id=glOqQgAACAAJ ."},{"key":"1245_CR64","volume-title":"Understanding motion capture for computer animation and video games","author":"A Menache","year":"1999","unstructured":"Menache, A. (1999). Understanding motion capture for computer animation and video games. San Francisco, CA: Morgan Kaufmann Publishers Inc."},{"key":"1245_CR65","unstructured":"M\u00fcller, M., R\u00f6der, T., Clausen, M., Eberhardt, B., Kr\u00fcger, B., & Weber, A. (2007). Documentation Mocap database HDM05. Technical report no. CG-2007-2, ISSN 1610-8892, Universit\u00e4t Bonn, the data used in this project was obtained from HDM05."},{"issue":"1","key":"1245_CR66","doi-asserted-by":"publisher","first-page":"39","DOI":"10.1002\/(SICI)1099-1778(199901\/03)10:1<39::AID-VIS195>3.0.CO;2-2","volume":"10","author":"F Multon","year":"1999","unstructured":"Multon, F., France, L., Cani-Gascuel, M. P., & Debunne, G. (1999). Computer animation of human walking: A survey. The Journal of Visualization and Computer Animation, 10(1), 39\u201354.","journal-title":"The Journal of Visualization and Computer Animation"},{"key":"1245_CR67","unstructured":"Oberweger, M., Wohlhart, P., & Lepetit, V. (2015). Hands deep in deep learning for hand pose estimation. In Computer vision winter workshop (CVWW)."},{"key":"1245_CR68","doi-asserted-by":"crossref","unstructured":"Ofli, F., Chaudhry, R., Kurillo, G., Vidal, R., & Bajcsy, R. (2013). Berkeley MHAD: A comprehensive multimodal human action database. In Proceedings of the IEEE workshop on applications on computer vision (WACV).","DOI":"10.1109\/WACV.2013.6474999"},{"key":"1245_CR69","doi-asserted-by":"crossref","unstructured":"Parameswaran, V., & Chellappa, R. (2004). View independent human body pose estimation from a single perspective image. In Conference on computer vision and pattern recognition (CVPR).","DOI":"10.1109\/CVPR.2004.1315139"},{"key":"1245_CR70","unstructured":"Parcollet, T., Ravanelli, M., Morchid, M., Linar\u00e8s, G., Trabelsi, C., De\u00a0Mori, R., et al. (2018a). Quaternion recurrent neural networks. arXiv preprint arXiv:1806.04418 ."},{"key":"1245_CR71","doi-asserted-by":"crossref","unstructured":"Parcollet, T., Zhang, Y., Morchid, M., Trabelsi, C., & Linar\u00e8s, G., Mori, R. D. et al. (2018b) Quaternion convolutional neural networks for end-to-end automatic speech recognition. In Interspeech.","DOI":"10.21437\/Interspeech.2018-1898"},{"key":"1245_CR72","unstructured":"Pascanu, R., Mikolov, T., & Bengio, Y. (2013). On the difficulty of training recurrent neural networks. In Proceedings of the 30th international conference on international conference on machine learning, JMLR.org, ICML\u201913, (Vol 28, pp. III\u20131310\u2013III\u20131318). http:\/\/dl.acm.org\/citation.cfm?id=3042817.3043083 ."},{"key":"1245_CR73","doi-asserted-by":"crossref","unstructured":"Pavllo, D., Feichtenhofer, C., Grangier, D., & Auli, M. (2019). 3d human pose estimation in video with temporal convolutions and semi-supervised training. In Conference on computer vision and pattern recognition (CVPR).","DOI":"10.1109\/CVPR.2019.00794"},{"key":"1245_CR74","unstructured":"Pavllo, D., Grangier, D., & Auli, M. (2018). Quaternet: A quaternion-based recurrent model for human motion. In British machine vision conference (BMVC)."},{"key":"1245_CR75","unstructured":"Pavlovic, V., Rehg, J. M., & MacCormick, J. (2001). Learning switching linear models of human motion. In T. K. Leen, T. G. Dietterich, & V. Tresp (Eds.), Advances in neural information processing systems 13 (pp. 981\u2013987). MIT Press. http:\/\/papers.nips.cc\/paper\/1892-learning-switching-linear-models-of-human-motion.pdf"},{"key":"1245_CR76","unstructured":"Pervin, E., & Webb, J. (1983). Quaternions for computer vision and robotics. In Conference on computer vision and pattern recognition (CVPR)."},{"key":"1245_CR77","doi-asserted-by":"crossref","unstructured":"Radwan, I., Dhall, A., & G\u00f6cke, R. (2013). Monocular image 3D human pose estimation under self-occlusion. In International conference on computer vision (ICCV), pp. 1888\u20131895.","DOI":"10.1109\/ICCV.2013.237"},{"key":"1245_CR78","unstructured":"Ranzato, M., Chopra, S., Auli, M., & Zaremba, W. (2015). Sequence-level training with recurrent neural networks. In International conference on learning represention (ICLR)."},{"key":"1245_CR79","doi-asserted-by":"crossref","unstructured":"Shlizerman, E., Dery, L. M., Schoen, H., & Kemelmacher-Shlizerman, I. (2018). Audio to body dynamics. In Conference on computer vision and pattern recognition (CVPR), pp. 7574\u20137583","DOI":"10.1109\/CVPR.2018.00790"},{"key":"1245_CR80","doi-asserted-by":"publisher","first-page":"245","DOI":"10.1145\/325165.325242","volume":"19","author":"K Shoemake","year":"1985","unstructured":"Shoemake, K. (1985). Animating rotation with quaternion curves. Transactions on Computer Graphics (SIGGRAPH), 19, 245\u2013254.","journal-title":"Transactions on Computer Graphics (SIGGRAPH)"},{"key":"1245_CR81","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4757-2272-7","volume-title":"Introduction to numerical analysis","author":"J Stoer","year":"1993","unstructured":"Stoer, J., & Bulirsch, R. (1993). Introduction to numerical analysis. New York: Springer."},{"key":"1245_CR82","doi-asserted-by":"crossref","unstructured":"Tanco, L. M., & Hilton, A. (2000). Realistic synthesis of novel human movements from a database of motion. In Workshop on human motion (HUMO).","DOI":"10.1109\/HUMO.2000.897383"},{"key":"1245_CR83","unstructured":"Taylor, G. W., Hinton, G. E., & Roweis, S. T. (2006). Modeling human motion using binary latent variables. In Advances in neural information processing systems (NIPS)."},{"key":"1245_CR84","unstructured":"Toyer, S., Cherian, A., Han, T., & Gould, S. (2017). Human pose forecasting via deep Markov models. In International conference on digital image computing: Techniques and applications (DICTA)."},{"issue":"3","key":"1245_CR85","doi-asserted-by":"publisher","first-page":"7","DOI":"10.1145\/1276377.1276386","volume":"26","author":"A Treuille","year":"2007","unstructured":"Treuille, A., Lee, Y., & Popovi\u0107, Z. (2007). Near-optimal character animation with continuous control. ACM Transactions on Graphics (tog), 26(3), 7.","journal-title":"ACM Transactions on Graphics (tog)"},{"key":"1245_CR86","doi-asserted-by":"publisher","DOI":"10.1201\/b11324","volume-title":"Game physics pearls","author":"G van den Bergen","year":"2010","unstructured":"van den Bergen, G., & Gregorius, D. (2010). Game physics pearls. Natick: AK Peters."},{"key":"1245_CR87","unstructured":"van\u00a0den Oord, A., Dieleman, S., Zen, H., Simonyan, K., Vinyals, O., Graves, A., et al. (2016a) Wavenet: A generative model for raw audio. arXiv preprint arXiv:1609.03499 ."},{"key":"1245_CR88","unstructured":"van\u00a0den Oord, A., Kalchbrenner, N., & Kavukcuoglu, K. (2016b). Pixel recurrent neural networks. In International conference on machine learning (ICML)."},{"key":"1245_CR89","doi-asserted-by":"crossref","unstructured":"Villegas, R., Yang, J., Ceylan, D., & Lee, H. (2018), Neural kinematic networks for unsupervised motion retargetting. In Conference on computer vision and pattern recognition (CVPR), pp. 8639\u20138648.","DOI":"10.1109\/CVPR.2018.00901"},{"key":"1245_CR90","unstructured":"Villegas, R., Yang, J., Zou, Y., Sohn, S., Lin, X., & Lee, H. (2017). Learning to generate long-term future via hierarchical prediction. In International conference on machine learning (ICML)."},{"key":"1245_CR91","doi-asserted-by":"crossref","unstructured":"Walker, J., Doersch, C., Gupta, A., & Hebert, M. (2016). An uncertain future: Forecasting from static images using variational autoencoders. In European conference on computer vision (ECCV).","DOI":"10.1007\/978-3-319-46478-7_51"},{"key":"1245_CR92","doi-asserted-by":"crossref","unstructured":"Walker, J., Marino, K., Gupta, A., & Hebert, M. (2017). The pose knows: Video forecasting by generating pose futures. In International conference on computer vision (ICCV).","DOI":"10.1109\/ICCV.2017.361"},{"key":"1245_CR93","doi-asserted-by":"publisher","first-page":"283","DOI":"10.1109\/TPAMI.2007.1167","volume":"30","author":"JM Wang","year":"2008","unstructured":"Wang, J. M., Fleet, D. J., & Hertzmann, A. (2008). Gaussian process dynamical models for human motion. Transaction on Pattern Analysis and Machine Intelligence (TPAMI), 30, 283\u2013298.","journal-title":"Transaction on Pattern Analysis and Machine Intelligence (TPAMI)"},{"key":"1245_CR94","unstructured":"Wang, Z., Chai, J., & Xia, S. (2018). Combining recurrent neural networks and adversarial training for human motion synthesis and control. arXiv:1806.08666 ."},{"key":"1245_CR95","doi-asserted-by":"crossref","unstructured":"Wiseman, S., & Rush, A. M. (2016). Sequence-to-sequence learning as beam-search optimization. In Conference on empirical methods in natural language processing (EMNLP).","DOI":"10.18653\/v1\/D16-1137"},{"key":"1245_CR96","doi-asserted-by":"publisher","first-page":"119","DOI":"10.1145\/2766999","volume":"34","author":"S Xia","year":"2015","unstructured":"Xia, S., Wang, C., Chai, J., & Hodgins, J. (2015). Realtime style transfer for unlabeled heterogeneous human motion. ACM Transactions on Graphics (SIGGRAPH), 34, 119.","journal-title":"ACM Transactions on Graphics (SIGGRAPH)"},{"key":"1245_CR97","doi-asserted-by":"publisher","first-page":"582","DOI":"10.1109\/TPAMI.2012.137","volume":"35","author":"F Zhou","year":"2013","unstructured":"Zhou, F., De la Torre, F., & Hodgins, J. K. (2013). Hierarchical aligned cluster analysis for temporal clustering of human motion. Transactions on Pattern Analysis and Machine Intelligence (TPAMI), 35, 582\u2013596.","journal-title":"Transactions on Pattern Analysis and Machine Intelligence (TPAMI)"},{"key":"1245_CR98","doi-asserted-by":"crossref","unstructured":"Zhou, X., Sun, X., Zhang, W., Liang, S., & Wei, Y. (2016a). Deep kinematic pose regression. In European conference on computer vision (ECCV) workshops.","DOI":"10.1007\/978-3-319-49409-8_17"},{"key":"1245_CR99","unstructured":"Zhou, X., Wan, Q., Zhang, W., Xue, X., & Wei, Y. (2016b). Model-based deep hand pose estimation. In IJCAI."},{"key":"1245_CR100","unstructured":"Zhou, Y., Li, Z., Xiao, S., He, C., & Li, H. (2018). Auto-conditioned LSTM network for extended complex human motion synthesis. In International conference on learning representations (ICLR)."},{"key":"1245_CR101","doi-asserted-by":"crossref","unstructured":"Zhu, X., Xu, Y., Xu, H., & Chen, C. (2018). Quaternion convolutional neural networks. In European conference on computer vision (ECCV).","DOI":"10.1007\/978-3-030-01237-3_39"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-019-01245-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11263-019-01245-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-019-01245-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,1,24]],"date-time":"2021-01-24T09:35:41Z","timestamp":1611480941000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11263-019-01245-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,10,8]]},"references-count":101,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2020,4]]}},"alternative-id":["1245"],"URL":"https:\/\/doi.org\/10.1007\/s11263-019-01245-6","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2019,10,8]]},"assertion":[{"value":"7 January 2019","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 September 2019","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 October 2019","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}