{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,2]],"date-time":"2026-08-02T08:03:08Z","timestamp":1785657788327,"version":"3.56.0"},"reference-count":32,"publisher":"Springer Science and Business Media LLC","issue":"2-4","license":[{"start":{"date-parts":[[2016,10,4]],"date-time":"2016-10-04T00:00:00Z","timestamp":1475539200000},"content-version":"unspecified","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100003132","name":"Agentschap voor Innovatie door Wetenschap en Technologie","doi-asserted-by":"publisher","award":["B\/14438\/23"],"award-info":[{"award-number":["B\/14438\/23"]}],"id":[{"id":"10.13039\/501100003132","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2018,4]]},"DOI":"10.1007\/s11263-016-0957-7","type":"journal-article","created":{"date-parts":[[2016,10,3]],"date-time":"2016-10-03T23:42:24Z","timestamp":1475538144000},"page":"430-439","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":190,"title":["Beyond Temporal Pooling: Recurrence and Temporal Convolutions for Gesture Recognition in Video"],"prefix":"10.1007","volume":"126","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3054-4960","authenticated-orcid":false,"given":"Lionel","family":"Pigou","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"A\u00e4ron","family":"van den Oord","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sander","family":"Dieleman","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mieke","family":"Van Herreweghe","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Joni","family":"Dambre","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2016,10,4]]},"reference":[{"key":"957_CR1","doi-asserted-by":"crossref","first-page":"29","DOI":"10.1007\/978-3-642-25446-8_4","volume-title":"Human behavior understanding","author":"M Baccouche","year":"2011","unstructured":"Baccouche, M., Mamalet, F., Wolf, C., Garcia, C., & Baskurt, A. (2011). Sequential deep learning for human action recognition. In A. Salah & B. Lepri (Eds.), Human behavior understanding (pp. 29\u201339). Berlin Heidelberg: Springer."},{"key":"957_CR2","unstructured":"Chang, J. Y. (2014). Nonparametric gesture labeling from multi-modal data. Computer vision-ECCV 2014 workshops (pp. 503\u2013517). Springer."},{"key":"957_CR3","unstructured":"Dieleman, S., van\u00a0den Oord, A., Korshunova, I., Burms, J., Degrave, J., Pigou, L., & Buteneers, P. (2015). Classifying plankton with deep neural networks. http:\/\/benanne.github.io\/2015\/03\/17\/plankton.html . Accessed 17 Mar 2015."},{"key":"957_CR4","doi-asserted-by":"crossref","unstructured":"Donahue, J., Anne\u00a0Hendricks, L., Guadarrama, S., Rohrbach, M., Venugopalan, S., Saenko, K., & Darrell, T. (2015). Long-term recurrent convolutional networks for visual recognition and description. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 2625\u20132634.","DOI":"10.1109\/CVPR.2015.7298878"},{"key":"957_CR5","unstructured":"Escalera, S., Bar, X., Gonzlez, J., Bautista, M.A., Madadi, M., Reyes, M., Ponce, V., Escalante, H.J., Shotton, J., & Guyon, I. (2014). Chalearn looking at people challenge 2014: Dataset and results. In: ECCV workshop."},{"key":"957_CR6","doi-asserted-by":"crossref","first-page":"363","DOI":"10.1007\/3-540-45103-X_50","volume-title":"Scandinavian conference on image analysis","author":"G Farneb\u00e4ck","year":"2003","unstructured":"Farneb\u00e4ck, G. (2003). Two-frame motion estimation based on polynomial expansion. In J. Bigun & T. Gustavsson (Eds.), Scandinavian conference on image analysis (pp. 363\u2013370). Berlin Heidelberg: Springer."},{"key":"957_CR7","first-page":"115","volume":"3","author":"FA Gers","year":"2003","unstructured":"Gers, F. A., Schraudolph, N. N., & Schmidhuber, J. (2003). Learning precise timing with lstm recurrent networks. The Journal of Machine Learning Research, 3, 115\u2013143.","journal-title":"The Journal of Machine Learning Research"},{"key":"957_CR8","unstructured":"Graham, B. (2014). Spatially-sparse convolutional neural networks. arXiv:1409.6070 (preprint)."},{"key":"957_CR9","doi-asserted-by":"crossref","unstructured":"Graves, A., Mohamed, A.R., & Hinton, G. (2013). Speech recognition with deep recurrent neural networks. In: Acoustics, speech and signal processing (ICASSP), 2013 IEEE international conference on, IEEE, pp. 6645\u20136649.","DOI":"10.1109\/ICASSP.2013.6638947"},{"key":"957_CR10","unstructured":"Hannun, A., Case, C., Casper, J., Catanzaro, B., Diamos, G., Elsen, E., Prenger, R., Satheesh, S., Sengupta, S., & Coates, A., et\u00a0al. (2014). Deepspeech: Scaling up end-to-end speech recognition. arXiv:1412.5567 (preprint)."},{"issue":"8","key":"957_CR11","doi-asserted-by":"crossref","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter, S., & Schmidhuber, J. (1997). Long short-term memory. Neural Computation, 9(8), 1735\u20131780.","journal-title":"Neural Computation"},{"key":"957_CR12","first-page":"302","volume":"2014","author":"A Jain","year":"2014","unstructured":"Jain, A., Tompson, J., LeCun, Y., & Bregler, C. (2014). MoDeep: A deep learning framework using motion features for human pose estimation. Computer Vision ACCV, 2014, 302\u2013315.","journal-title":"Computer Vision ACCV"},{"issue":"1","key":"957_CR13","doi-asserted-by":"crossref","first-page":"221","DOI":"10.1109\/TPAMI.2012.59","volume":"35","author":"S Ji","year":"2013","unstructured":"Ji, S., Xu, W., Yang, M., & Yu, K. (2013). 3d convolutional neural networks for human action recognition. IEEE Transactions on Pattern Analysis and Machine Intelligence, 35(1), 221\u2013231.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"957_CR14","doi-asserted-by":"crossref","unstructured":"Karpathy, A., Toderici, G., Shetty, S., Leung, T., Sukthankar, R., & Fei-Fei, L. (2014). Large-scale video classification with convolutional neural networks. In: Computer vision and pattern recognition (CVPR), 2014 IEEE conference on, IEEE, pp. 1725\u20131732.","DOI":"10.1109\/CVPR.2014.223"},{"key":"957_CR15","unstructured":"Kingma, D., & Ba, J. (2015). Adam: A method for stochastic optimization. ICLR 2015."},{"key":"957_CR16","unstructured":"Krizhevsky, A., Sutskever, I., & Hinton, GE, (2012). Imagenet classification with deep convolutional neural networks. In F. Pereira, C. J. C. Burges, L. Bottou & K. Q. Weinberger (Eds.), Advances in neural information processing systems (pp. 1097\u20131105). http:\/\/papers.nips.cc\/paper\/4824-imagenet-classification-withdeep-convolutional-neural-networks.pdf ."},{"key":"957_CR17","doi-asserted-by":"crossref","unstructured":"Kuehne, H., Jhuang, H., Garrote, E., Poggio, T., & Serre, T. (2011). Hmdb: A large video database for human motion recognition. In: Computer vision (ICCV), 2011 IEEE international conference on, IEEE, pp. 2556\u20132563.","DOI":"10.1109\/ICCV.2011.6126543"},{"issue":"11","key":"957_CR18","doi-asserted-by":"crossref","first-page":"2278","DOI":"10.1109\/5.726791","volume":"86","author":"Y LeCun","year":"1998","unstructured":"LeCun, Y., Bottou, L., Bengio, Y., & Haffner, P. (1998). Gradient-based learning applied to document recognition. Proceedings of the IEEE, 86(11), 2278\u20132324.","journal-title":"Proceedings of the IEEE"},{"key":"957_CR19","unstructured":"Maas, A.L., Hannun, A.Y., & Ng, A.Y. (2013). Rectifier nonlinearities improve neural network acoustic models. In: Proc. ICML, vol.\u00a030."},{"key":"957_CR20","unstructured":"Monnier, C., German, S., & Ost, A. (2014). A multi-scale boosted detector for efficient and robust gesture recognition. In: Computer vision-ECCV 2014 workshops (pp. 491\u2013502). Springer"},{"key":"957_CR21","unstructured":"Neverova, N., Wolf, C., Taylor, G.W., & Nebout, F. (2014). ModDrop: Adaptive multi-modal gesture recognition. arXiv:1501.00102 (preprint)."},{"key":"957_CR22","unstructured":"Ng, J.Y.H., Hausknecht, M., Vijayanarasimhan, S., Vinyals, O., Monga, R., & Toderici, G. (2015). Beyond short snippets: Deep networks for video classification. In: Computer vision and pattern recognition (CVPR), 2015 IEEE conference on, IEEE, pp. 4694\u20134702."},{"key":"957_CR23","unstructured":"Saxe, A.M., McClelland, J.L., & Ganguli, S. (2013). Exact solutions to the nonlinear dynamics of learning in deep linear neural networks. arXiv:1312.6120 (preprint)."},{"key":"957_CR24","unstructured":"Sermanet, P., Eigen, D., Zhang, X., Mathieu, M., Fergus, R., & LeCun, Y. (2013). Overfeat: Integrated recognition, localization and detection using convolutional networks. arXiv:1312.6229 (preprint)."},{"key":"957_CR25","unstructured":"Simonyan, K., & Zisserman, A. (2014). Two-stream convolutional networks for action recognition in videos. In Z. Ghahramani, M. Welling, C. Cortes, N. D. Lawrence & K. Q. Weinberger (Eds.), Advances in neural information processing systems (pp. 568\u2013576). http:\/\/papers.nips.cc\/paper\/5353-two-stream-convolutionalnetworks-for-action-recognition-in-videos.pdf ."},{"key":"957_CR26","unstructured":"Soomro, K., Zamir, A.R., & Shah, M. (2012). UCF101: A dataset of 101 human actions classes from videos in the wild. arXiv:1212.0402 (preprint)."},{"key":"957_CR27","unstructured":"Sutskever. I., Vinyals, O., & Le, Q.V. (2014). Sequence to sequence learning with neural networks. In Z. Ghahramani, M. Welling, C. Cortes, N. D. Lawrence & K. Q. Weinberger (Eds.), Advances in neural information processing systems (pp. 3104\u20133112). http:\/\/papers.nips.cc\/paper\/5346-sequence-to-sequence-learningwith-neural-networks.pdf ."},{"key":"957_CR28","doi-asserted-by":"crossref","first-page":"140","DOI":"10.1007\/978-3-642-15567-3_11","volume-title":"Computer vision-ECCV 2010","author":"GW Taylor","year":"2010","unstructured":"Taylor, G. W., Fergus, R., LeCun, Y., & Bregler, C. (2010). Convolutional learning of spatio-temporal features. In K. Daniilidis, P. Maragos, & N. Paragios (Eds.), Computer vision-ECCV 2010 (pp. 140\u2013153). Berlin Heidelberg: Springer."},{"key":"957_CR29","doi-asserted-by":"crossref","unstructured":"Toshev, A., & Szegedy, C. (2014). DeepPose: Human pose estimation via deep neural networks. In: Computer vision and pattern recognition (CVPR), 2014 IEEE conference on, IEEE, pp. 1653\u20131660.","DOI":"10.1109\/CVPR.2014.214"},{"key":"957_CR30","unstructured":"Venugopalan, S., Rohrbach, M., Donahue, J., Mooney, R., Darrell, T., & Saenko, K. (2015). Sequence to sequence\u2013video to text. arXiv:1505.00487 (preprint)."},{"key":"957_CR31","doi-asserted-by":"crossref","unstructured":"Vinyals, O., Toshev, A., Bengio, S., & Erhan, D. (2015). Show and tell: A neural image caption generator. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp. 3156\u20133164.","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"957_CR32","unstructured":"Xu, B., Wang, N., Chen, T., & Li, M. (2015). Empirical evaluation of rectified activations in convolutional network. In: ICML deep learning workshop."}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11263-016-0957-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-016-0957-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-016-0957-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,9,14]],"date-time":"2019-09-14T02:58:39Z","timestamp":1568429919000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11263-016-0957-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2016,10,4]]},"references-count":32,"journal-issue":{"issue":"2-4","published-print":{"date-parts":[[2018,4]]}},"alternative-id":["957"],"URL":"https:\/\/doi.org\/10.1007\/s11263-016-0957-7","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2016,10,4]]}}}