{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,27]],"date-time":"2025-10-27T16:15:11Z","timestamp":1761581711744,"version":"3.41.0"},"reference-count":50,"publisher":"Springer Science and Business Media LLC","issue":"24","license":[{"start":{"date-parts":[[2018,6,18]],"date-time":"2018-06-18T00:00:00Z","timestamp":1529280000000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["61672299"],"award-info":[{"award-number":["61672299"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"published-print":{"date-parts":[[2018,12]]},"DOI":"10.1007\/s11042-018-6260-6","type":"journal-article","created":{"date-parts":[[2018,6,18]],"date-time":"2018-06-18T07:30:47Z","timestamp":1529307047000},"page":"32275-32285","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":9,"title":["Time-varying LSTM networks for action recognition"],"prefix":"10.1007","volume":"77","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3228-651X","authenticated-orcid":false,"given":"Zichao","family":"Ma","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhixin","family":"Sun","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2018,6,18]]},"reference":[{"key":"6260_CR1","unstructured":"Amodei D, Anubhai R, Battenberg E, et al. (2015) Deep speech 2: End-to-end speech recognition in english and mandarin[J]. arXiv preprint arXiv:1512.02595"},{"key":"6260_CR2","unstructured":"Bahdanau D, Cho K, Bengio Y (2014) Neural machine translation by jointly learning to align and translate. arXiv preprint arXiv:1409.0473"},{"key":"6260_CR3","doi-asserted-by":"publisher","first-page":"157","DOI":"10.1109\/72.279181","volume":"5","author":"Y Bengio","year":"1994","unstructured":"Bengio Y, Simard P, Frasconi P (1994) Learning long-term dependencies with gradient descent is difficult. IEEE Trans Neural Netw 5:157\u2013166","journal-title":"IEEE Trans Neural Netw"},{"key":"6260_CR4","doi-asserted-by":"crossref","unstructured":"Cho K et al. (2014) Learning phrase representations using RNN encoder-decoder for statistical machine translation. arXiv preprint arXiv:1406.1078","DOI":"10.3115\/v1\/D14-1179"},{"key":"6260_CR5","doi-asserted-by":"crossref","unstructured":"Dalal N, Triggs B (2005) Histograms of oriented gradients for human detection. In Computer Vision and Pattern Recognition, 2005. CVPR 2005. IEEE Computer Society Conference on, vol. 1, 886\u2013893 (IEEE)","DOI":"10.1109\/CVPR.2005.177"},{"key":"6260_CR6","doi-asserted-by":"crossref","unstructured":"Deng, J. et al. (2009) Imagenet: A large-scale hierarchical image database. In Computer Vision and Pattern Recognition, 2009. CVPR 2009. IEEE Conference on, 248\u2013255 (IEEE)","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"6260_CR7","doi-asserted-by":"crossref","unstructured":"Donahue J et al. (2015) Long-term recurrent convolutional networks for visual recognition and description. In Proceedings of the IEEE conference on computer vision and pattern recognition, 2625\u20132634","DOI":"10.1109\/CVPR.2015.7298878"},{"key":"6260_CR8","unstructured":"El Hihi S., Bengio Y (1995) Hierarchical Recurrent Neural Networks for Long-Term Dependencies. In NIPS, vol. 400, 409 (Citeseer)"},{"key":"6260_CR9","doi-asserted-by":"crossref","unstructured":"Gers FA, Schmidhuber J (2000) Recurrent nets that time and count. In Neural Networks, 2000. IJCNN 2000, Proceedings of the IEEE-INNS-ENNS International Joint Conference on, vol. 3, 189\u2013194 (IEEE)","DOI":"10.1109\/IJCNN.2000.861302"},{"key":"6260_CR10","doi-asserted-by":"publisher","first-page":"2451","DOI":"10.1162\/089976600300015015","volume":"12","author":"FA Gers","year":"2000","unstructured":"Gers FA, Schmidhuber J, Cummins F (2000) Learning to forget: continual prediction with LSTM. Neural Comput 12:2451\u20132471","journal-title":"Neural Comput"},{"key":"6260_CR11","unstructured":"Goodfellow I, Bengio Y, Courville A (2016) Deep Learning. URL http:\/\/www.deeplearningbook.org , book in preparation for MIT Press"},{"key":"6260_CR12","doi-asserted-by":"crossref","unstructured":"Graves A, Mohamed A-R, Hinton G (2013) Speech recognition with deep recurrent neural networks. In 2013 IEEE international conference on acoustics, speech and signal processing, 6645\u20136649 (IEEE)","DOI":"10.1109\/ICASSP.2013.6638947"},{"key":"6260_CR13","doi-asserted-by":"publisher","first-page":"602","DOI":"10.1016\/j.neunet.2005.06.042","volume":"18","author":"A Graves","year":"2005","unstructured":"Graves A, Schmidhuber J (2005) Framewise phoneme classification with bidirectional LSTM and other neural network architectures. Neural Netw 18:602\u2013610","journal-title":"Neural Netw"},{"key":"6260_CR14","unstructured":"Greff K, Srivastava RK, Koutnk J, Steunebrink BR, Schmidhuber J (2015) LSTM: A search space odyssey. arXiv preprint arXiv:1503.04069"},{"key":"6260_CR15","unstructured":"Hannun A, Case C, Casper J et al. (2014) Deep speech: Scaling up end-to-end speech recognition[J]. arXiv preprint arXiv:1412.5567"},{"key":"6260_CR16","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"6260_CR17","doi-asserted-by":"publisher","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter S, Schmidhuber J (1997) Long short-term memory. Neural Comput 9:1735\u20131780 This paper introduced LSTM recurrent networks, which have become a crucial ingredient in recent advances with recurrent networks because they are good at learning long-range dependencies","journal-title":"Neural Comput"},{"key":"6260_CR18","unstructured":"Hochreiter S, Schmidhuber J (1995) Long Short-term Memory"},{"key":"6260_CR19","doi-asserted-by":"crossref","unstructured":"Jain M, van Gemert JC, Snoek CG (2015) What do 15,000 object categories tell us about classifying and localizing actions? In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 46\u201355","DOI":"10.1109\/CVPR.2015.7298599"},{"key":"6260_CR20","doi-asserted-by":"crossref","unstructured":"Jhuang H, Serre T, Wolf L, Poggio T (2007) A biologically inspired system for action recognition. In ICCV","DOI":"10.1109\/ICCV.2007.4408988"},{"key":"6260_CR21","doi-asserted-by":"publisher","first-page":"221","DOI":"10.1109\/TPAMI.2012.59","volume":"35","author":"S Ji","year":"2013","unstructured":"Ji S, Xu W, Yang M, Yu K (2013) 3D convolutional neural networks for human action recognition. IEEE Trans Pattern Anal Mach Intell 35:221\u2013231","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"6260_CR22","doi-asserted-by":"crossref","unstructured":"Karpathy A et al. (2014) Large-scale video classification with convolutional neural networks. In Proceedings of the IEEE conference on Computer Vision and Pattern Recognition, 1725\u20131732","DOI":"10.1109\/CVPR.2014.223"},{"key":"6260_CR23","unstructured":"Lan Z, Lin M, Li X, Hauptmann AG, Raj B (2015) Beyond gaussian pyramid: Multi-skip feature stacking for action recognition. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 204\u2013212"},{"key":"6260_CR24","doi-asserted-by":"crossref","unstructured":"Laptev I, Marszalek M, Schmid C, Rozenfeld B (2008) Learning realistic human actions from movies. In Proceedings CVPR08 (citeseer)","DOI":"10.1109\/CVPR.2008.4587756"},{"key":"6260_CR25","doi-asserted-by":"publisher","first-page":"436","DOI":"10.1038\/nature14539","volume":"521","author":"Y LeCun","year":"2015","unstructured":"LeCun Y, Bengio Y, Hinton G (2015) Deep learning. Nature 521:436\u2013444","journal-title":"Nature"},{"key":"6260_CR26","unstructured":"Lipton ZC, Berkowitz J, Elkan C (2015) A critical review of recurrent neural networks for sequence learning. arXiv preprint arXiv:1506.00019"},{"key":"6260_CR27","unstructured":"Ng JY-H. et al. (2015) Beyond short snippets: Deep networks for video classification. In 2015 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)"},{"key":"6260_CR28","doi-asserted-by":"crossref","unstructured":"Otte S, Liwicki M, Zell A (2014) Dynamic cortex memory: enhancing recurrent neural networks for gradient-based sequence learning. In International Conference on Artificial Neural Networks, 1\u20138 (Springer)","DOI":"10.1007\/978-3-319-11179-7_1"},{"issue":"3","key":"6260_CR29","first-page":"1310","volume":"28","author":"R Pascanu","year":"2013","unstructured":"Pascanu R, Mikolov T, Bengio Y (2013) On the difficulty of training recurrent neural networks. ICML 28(3):1310\u20131318","journal-title":"ICML"},{"key":"6260_CR30","doi-asserted-by":"publisher","first-page":"109","DOI":"10.1016\/j.cviu.2016.03.013","volume":"150","author":"X Peng","year":"2016","unstructured":"Peng X, Wang L, Wang X, Qiao Y (2016) Bag of visual words and fusion methods for action recognition: comprehensive study and good practice. Comput Vis Image Underst 150:109\u2013125","journal-title":"Comput Vis Image Underst"},{"key":"6260_CR31","doi-asserted-by":"crossref","unstructured":"Peng X, Zou C, Qiao Y, Peng Q (2014) Action recognition with stacked fisher vectors. In European Conference on Computer Vision, 581\u2013595 (Springer)","DOI":"10.1007\/978-3-319-10602-1_38"},{"key":"6260_CR32","doi-asserted-by":"crossref","unstructured":"Sak H, Senior AW, Beaufays F (2014) Long short-term memory recurrent neural network architectures for large scale acoustic modeling. In Interspeech, 338\u2013342","DOI":"10.21437\/Interspeech.2014-80"},{"key":"6260_CR33","unstructured":"Sharma S, Kiros R, Salakhutdinov R (2015) Action recognition using visual attention. arXiv preprint arXiv:1511.04119"},{"key":"6260_CR34","unstructured":"Simonyan K, Zisserman A (2014) Two-stream convolutional networks for action recognition in videos. In Advances in neural information processing systems, 568\u2013576"},{"key":"6260_CR35","unstructured":"Soomro, K., Zamir, A. R. & Shah, M. (2012) UCF101: A Dataset of 101 Human Actions Classes From Videos in The Wild. CRCV-TR-12-01"},{"key":"6260_CR36","first-page":"1929","volume":"15","author":"N Srivastava","year":"2014","unstructured":"Srivastava N, Hinton GE, Krizhevsky A, Sutskever I, Salakhutdinov R (2014) Dropout: a simple way to prevent neural networks from overfitting. J Mach Learn Res 15:1929\u20131958","journal-title":"J Mach Learn Res"},{"key":"6260_CR37","unstructured":"Srivastava N, Mansimov E, Salakhutdinov R (2015) Unsupervised Learning of Video Representations using LSTMs. In ICML, 843\u2013852"},{"key":"6260_CR38","unstructured":"Sutskever I (2013) Training recurrent neural networks. Ph.D. thesis, University of Toronto"},{"key":"6260_CR39","unstructured":"Sutskever I, Vinyals O, Le QV (2014) Sequence to sequence learning with neural networks. In Advances in neural information processing systems, 3104\u20133112"},{"key":"6260_CR40","doi-asserted-by":"crossref","unstructured":"Tran D, Bourdev L, Fergus R, Torresani L, Paluri M (2015) Learning spatiotemporal features with 3d convolutional networks. In Proceedings of the IEEE International Conference on Computer Vision, 4489\u20134497","DOI":"10.1109\/ICCV.2015.510"},{"key":"6260_CR41","unstructured":"Venugopalan S, Xu H, Donahue J, et al. (2014) Translating videos to natural language using deep recurrent neural networks[J]. arXiv preprint arXiv:1412.4729"},{"key":"6260_CR42","doi-asserted-by":"crossref","unstructured":"Vinyals O, Toshev A, Bengio S, et al. (2015) Show and tell: A neural image caption generator. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. 3156\u20134164","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"6260_CR43","doi-asserted-by":"crossref","unstructured":"Wang H, Schmid C (2013) Action Recognition with Improved Trajectories. In Computer Vision (ICCV), 2013 IEEE International Conference on, 3551\u20133558 (IEEE)","DOI":"10.1109\/ICCV.2013.441"},{"key":"6260_CR44","doi-asserted-by":"crossref","unstructured":"Wang H, Ullah MM, Klaser A, Laptev I, Schmid C (2009) Evaluation of local spatio-temporal features for action recognition. In BMVC 2009 British Machine Vision Conference, 124\u20131 (BMVA Press)","DOI":"10.5244\/C.23.124"},{"key":"6260_CR45","unstructured":"Wu Y, Zhang S, Zhang Y, Bengio Y, Salakhutdinov R (2016) On Multiplicative Integration with Recurrent Neural Networks. arXiv preprint arXiv:1606.06630"},{"key":"6260_CR46","unstructured":"Xu K et al. (2015) Show, Attend and Tell: Neural Image Caption Generation with Visual Attention. In ICML, vol. 14, 77\u201381"},{"key":"6260_CR47","doi-asserted-by":"crossref","unstructured":"Yao L, Torabi A, Cho K et al. (2015) Describing videos by exploiting temporal structure. In Proceedings of the IEEE international conference on computer vision: 4507\u20134515","DOI":"10.1109\/ICCV.2015.512"},{"key":"6260_CR48","unstructured":"Zaremba W, Sutskever I, Vinyals O (2014) Recurrent neural network regularization. arXiv preprint arXiv:1409.2329"},{"key":"6260_CR49","unstructured":"Zeiler MD (2012) ADADELTA: an adaptive learning rate method. arXiv preprint arXiv:1212.5701"},{"key":"6260_CR50","doi-asserted-by":"crossref","unstructured":"Zhang B, Wang L, Wang Z, et al. (2016) Real-time action recognition with enhanced motion vector CNNs. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2718\u20132726","DOI":"10.1109\/CVPR.2016.297"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11042-018-6260-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-018-6260-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-018-6260-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,5]],"date-time":"2025-07-05T04:45:29Z","timestamp":1751690729000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11042-018-6260-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,6,18]]},"references-count":50,"journal-issue":{"issue":"24","published-print":{"date-parts":[[2018,12]]}},"alternative-id":["6260"],"URL":"https:\/\/doi.org\/10.1007\/s11042-018-6260-6","relation":{},"ISSN":["1380-7501","1573-7721"],"issn-type":[{"type":"print","value":"1380-7501"},{"type":"electronic","value":"1573-7721"}],"subject":[],"published":{"date-parts":[[2018,6,18]]},"assertion":[{"value":"25 April 2017","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 June 2018","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 June 2018","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 June 2018","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}