{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T04:35:17Z","timestamp":1750221317093,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":69,"publisher":"ACM","license":[{"start":{"date-parts":[[2017,10,23]],"date-time":"2017-10-23T00:00:00Z","timestamp":1508716800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"JST CREST","award":["JPMJCR1687"],"award-info":[{"award-number":["JPMJCR1687"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2017,10,23]]},"DOI":"10.1145\/3126686.3126755","type":"proceedings-article","created":{"date-parts":[[2017,10,23]],"date-time":"2017-10-23T19:20:32Z","timestamp":1508786432000},"page":"393-401","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":6,"title":["CTC Network with Statistical Language Modeling for Action Sequence Recognition in Videos"],"prefix":"10.1145","author":[{"given":"Mengxi","family":"Lin","sequence":"first","affiliation":[{"name":"Tokyo Institute of Technology, Tokyo, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Nakamasa","family":"Inoue","sequence":"additional","affiliation":[{"name":"Tokyo Institute of Technology, Tokyo, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Koichi","family":"Shinoda","sequence":"additional","affiliation":[{"name":"Tokyo Institute of Technology, Tokyo, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2017,10,23]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2013.441"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298872"},{"key":"e_1_3_2_1_3_1","first-page":"568","volume-title":"Proc. NIPS.","author":"Simonyan K.","year":"2014","unstructured":"K. Simonyan and A. Zisserman . Two-stream convolutional networks for action recognition in videos . In Proc. NIPS. pp. 568 -- 576 . 2014 . K. Simonyan and A. Zisserman. Two-stream convolutional networks for action recognition in videos. In Proc. NIPS. pp. 568--576. 2014."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.510"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46493-0_9"},{"key":"e_1_3_2_1_6_1","volume-title":"CVIU.","author":"Kuehne H.","year":"2017","unstructured":"H. Kuehne Weakly supervised learning of actions from transcripts . In CVIU. 2017 . H. Kuehne et al. Weakly supervised learning of actions from transcripts. In CVIU. 2017."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.456"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/1143844.1143891"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.216"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2005.06.042"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.117"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.105"},{"key":"e_1_3_2_1_14_1","volume-title":"Proc. ICLR.","author":"Bahdanau D.","year":"2015","unstructured":"D. Bahdanau Neural machine translation by jointly learning to align and translate . In Proc. ICLR. 2015 . D. Bahdanau et al. Neural machine translation by jointly learning to align and translate. In Proc. ICLR. 2015."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.512"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-005-1838-7"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.213"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/1291233.1291311"},{"key":"e_1_3_2_1_19_1","volume-title":"Proc. BMVC.","author":"Klaser A.","year":"2008","unstructured":"A. Klaser A spatio-temporal descriptor based on 3dgradients . In Proc. BMVC. 2008 . A. Klaser et al. A spatio-temporal descriptor based on 3dgradients. In Proc. BMVC. 2008."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-88688-4_48"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2011.6126543"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-013-0636-x"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2013.207"},{"key":"e_1_3_2_1_25_1","first-page":"1097","volume-title":"Proc. NIPS.","author":"Krizhevsky A.","year":"2012","unstructured":"A. Krizhevsky Imagenet classification with deep convolutional neural networks . In Proc. NIPS. pp. 1097 -- 1105 . 2012 . A. Krizhevsky et al. Imagenet classification with deep convolutional neural networks. In Proc. NIPS. pp. 1097--1105. 2012."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2012.59"},{"key":"e_1_3_2_1_29_1","first-page":"1","volume-title":"Proc. CVPR.","author":"Shi Q.","year":"2008","unstructured":"Q. Shi Discriminative human action segmentation and recognition using semi-Markov model . In Proc. CVPR. pp. 1 -- 8 . 2008 . Q. Shi et al. Discriminative human action segmentation and recognition using semi-Markov model. In Proc. CVPR. pp. 1--8. 2008."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2011.5995555"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2013.65"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-49409-8_7"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46487-9_3"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.155"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-25446-8_4"},{"key":"e_1_3_2_1_36_1","first-page":"2625","volume-title":"Long-term recurrent convolutional networks for visual recognition and description.In Proc. CVPR","author":"Donahue J.","year":"2015","unstructured":"J. Donahue Long-term recurrent convolutional networks for visual recognition and description.In Proc. CVPR . pp. 2625 -- 2634 . 2015 . J. Donahue et al. Long-term recurrent convolutional networks for visual recognition and description.In Proc. CVPR. pp. 2625--2634. 2015."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.460"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/2911996.2912001"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.85"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.cviu.2015.07.006"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.341"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.678"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_41"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.495"},{"key":"e_1_3_2_1_45_1","volume-title":"Proc. ICML.","author":"Amodei D.","year":"2016","unstructured":"D. Amodei Deep speech 2: End-to-end speech recognition in english and mandarin . In Proc. ICML. 2016 . D. Amodei et al. Deep speech 2: End-to-end speech recognition in english and mandarin. In Proc. ICML. 2016."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2015.7333730"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/K15-1031"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-11581-8_39"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/319950.320022"},{"key":"e_1_3_2_1_50_1","first-page":"182","volume-title":"Trans. of the Faculty of Actuaries.","author":"Lidstone G. J.","year":"1920","unstructured":"G. J. Lidstone . Note on the general case of the Bayes-Laplace formula for inductive or a posteriori probabilities . In Trans. of the Faculty of Actuaries. pp. 182 -- 192 . 1920 . G. J. Lidstone. Note on the general case of the Bayes-Laplace formula for inductive or a posteriori probabilities. In Trans. of the Faculty of Actuaries. pp. 182--192. 1920."},{"key":"e_1_3_2_1_51_1","first-page":"381","volume-title":"Proc. of Workshop on Pattern Recognition in Practice.","author":"Jelinek F.","year":"1980","unstructured":"F. Jelinek and R. Mercer . Interpolated estimation of Markov source parameters from sparse data . In Proc. of Workshop on Pattern Recognition in Practice. pp. 381 -- 397 . 1980 . F. Jelinek and R. Mercer. Interpolated estimation of Markov source parameters from sparse data. In Proc. of Workshop on Pattern Recognition in Practice. pp. 381--397. 1980."},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASSP.1987.1165125"},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.2307\/2333344"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/34.476512"},{"key":"e_1_3_2_1_55_1","volume-title":"Text compression","author":"Bell T. C.","year":"1990","unstructured":"T. C. Bell Text compression . Prentice-Hall, Inc. 1990 . T. C. Bell et al. Text compression. Prentice-Hall, Inc. 1990."},{"key":"e_1_3_2_1_56_1","first-page":"1","volume-title":"Proc. LREC.","author":"Guthrie D.","year":"2006","unstructured":"D. Guthrie A closer look at skip-gram modelling . In Proc. LREC. pp. 1 -- 4 . 2006 . D. Guthrie et al. A closer look at skip-gram modelling. In Proc. LREC. pp. 1--4. 2006."},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.5555\/176313.176316"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"crossref","first-page":"2167","DOI":"10.21437\/Eurospeech.1999-479","volume-title":"Proc. EUROSPEECH.","author":"Gildea D.","year":"1999","unstructured":"D. Gildea and T. Hofmann . Topic-based language models using EM . In Proc. EUROSPEECH. pp. 2167 -- 2170 . 1999 . D. Gildea and T. Hofmann. Topic-based language models using EM. In Proc. EUROSPEECH. pp. 2167--2170. 1999."},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"crossref","first-page":"1045","DOI":"10.21437\/Interspeech.2010-343","volume-title":"Proc. Interspeech.","author":"Mikolov T.","year":"2010","unstructured":"T. Mikolov Recurrent neural network based language model . In Proc. Interspeech. pp. 1045 -- 1048 . 2010 . T. Mikolov et al. Recurrent neural network based language model. In Proc. Interspeech. pp. 1045--1048. 2010."},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.3115\/981863.981904"},{"key":"e_1_3_2_1_61_1","first-page":"901","volume-title":"Proc. Interspeech.","author":"Stolcke A.","year":"2002","unstructured":"A. Stolcke SRILM-an extensible language modeling toolkit . In Proc. Interspeech. pp. 901 -- 904 . 2002 . A. Stolcke et al. SRILM-an extensible language modeling toolkit. In Proc. Interspeech. pp. 901--904. 2002."},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"crossref","first-page":"2237","DOI":"10.21437\/Interspeech.2013-526","volume-title":"Proc. Interspeech.","author":"Miao Y.","year":"2013","unstructured":"Y. Miao , and F. Metze . Improving low-resource CD-DNN-HMM using dropout and multilingual DNN training . In Proc. Interspeech. pp. 2237 -- 2241 . 2013 . Y. Miao, and F. Metze. Improving low-resource CD-DNN-HMM using dropout and multilingual DNN training. In Proc. Interspeech. pp. 2237--2241. 2013."},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCSLP.2012.6423452"},{"key":"e_1_3_2_1_64_1","doi-asserted-by":"publisher","DOI":"10.1109\/5.18626"},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2015.7404790"},{"key":"e_1_3_2_1_66_1","volume-title":"First-pass large vocabulary continuous speech recognition using bi-directional recurrent DNNs. arXiv preprint arXiv:1408.2873","author":"Hannun A. Y.","year":"2014","unstructured":"A. Y. Hannun First-pass large vocabulary continuous speech recognition using bi-directional recurrent DNNs. arXiv preprint arXiv:1408.2873 . 2014 . A. Y. Hannun et al. First-pass large vocabulary continuous speech recognition using bi-directional recurrent DNNs. arXiv preprint arXiv:1408.2873. 2014."},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/N15-1038"},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV.2016.7477701"},{"key":"e_1_3_2_1_69_1","volume-title":"Lecture 6.5RmsProp: Divide the gradient by a running average of its recent magnitude. COURSERA: Neural Networks for Machine Learning","author":"Tieleman T.","year":"2012","unstructured":"T. Tieleman and G. E. Hinton . Lecture 6.5RmsProp: Divide the gradient by a running average of its recent magnitude. COURSERA: Neural Networks for Machine Learning . 2012 . T. Tieleman and G. E. Hinton. Lecture 6.5RmsProp: Divide the gradient by a running average of its recent magnitude. COURSERA: Neural Networks for Machine Learning. 2012."},{"key":"e_1_3_2_1_70_1","doi-asserted-by":"publisher","DOI":"10.25080\/Majora-8b375195-003"}],"event":{"name":"MM '17: ACM Multimedia Conference","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Mountain View California USA","acronym":"MM '17"},"container-title":["Proceedings of the on Thematic Workshops of ACM Multimedia 2017"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3126686.3126755","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3126686.3126755","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T02:13:47Z","timestamp":1750212827000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3126686.3126755"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2017,10,23]]},"references-count":69,"alternative-id":["10.1145\/3126686.3126755","10.1145\/3126686"],"URL":"https:\/\/doi.org\/10.1145\/3126686.3126755","relation":{},"subject":[],"published":{"date-parts":[[2017,10,23]]},"assertion":[{"value":"2017-10-23","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}