{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T00:14:15Z","timestamp":1730247255486,"version":"3.28.0"},"reference-count":25,"publisher":"IEEE","license":[{"start":{"date-parts":[[2022,10,16]],"date-time":"2022-10-16T00:00:00Z","timestamp":1665878400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2022,10,16]],"date-time":"2022-10-16T00:00:00Z","timestamp":1665878400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022,10,16]]},"DOI":"10.1109\/icip46576.2022.9897974","type":"proceedings-article","created":{"date-parts":[[2022,11,3]],"date-time":"2022-11-03T21:27:24Z","timestamp":1667510844000},"page":"1631-1635","source":"Crossref","is-referenced-by-count":1,"title":["Higher-Order Recurrent Network with Space-Time Attention for Video Early Action Recognition"],"prefix":"10.1109","author":[{"given":"Tsung-Ming","family":"Tai","sequence":"first","affiliation":[{"name":"NVIDIA AI Technology Center"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Giuseppe","family":"Fiameni","sequence":"additional","affiliation":[{"name":"NVIDIA AI Technology Center"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Cheng-Kuang","family":"Lee","sequence":"additional","affiliation":[{"name":"NVIDIA AI Technology Center"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Oswald","family":"Lanz","sequence":"additional","affiliation":[{"name":"Free University of Bozen-Bolzano"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.39"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"ref3","first-page":"813","article-title":"Is space-time attention all you need for video understanding?","volume-title":"ICML","author":"Bertasius"},{"key":"ref4","article-title":"N-gram-based text categorization","volume-title":"Proceedings of SDAIR-94, 3rd annual symposium on document analysis and information retrieval","volume":"161175","author":"Cavnar"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW50498.2020.00359"},{"key":"ref6","first-page":"720","article-title":"Scaling egocentric vision: The epic-kitchens dataset","volume-title":"ECCV","author":"Damen"},{"article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","volume-title":"ICLR","author":"Dosovitskiy","key":"ref7"},{"volume-title":"Statistical identification of language","year":"1994","author":"Dunning","key":"ref8"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00635"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2020.2992889"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/WACV.2018.00173"},{"key":"ref12","first-page":"5843","article-title":"something something","volume-title":"video database for learning and evaluating visual common sense. In ICCV","author":"Goyal"},{"key":"ref13","first-page":"448","article-title":"Batch normalization: Accelerating deep network training by reducing internal covariate shift","volume-title":"ICML","author":"Ioffe"},{"key":"ref14","article-title":"Human action recognition and prediction: A survey","volume":"abs\/1806.11230","author":"Kong","year":"2018","journal-title":"CoRR"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.390"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"ref17","article-title":"Christoph Feichtenhofer, Trevor Darrell, and Saining Xie. A convnet for the 2020s","volume":"abs\/2201.03545","author":"Liu","year":"2022","journal-title":"CoRR"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.214"},{"key":"ref19","article-title":"Higher order recurrent neural networks","volume":"abs\/1605.00064","author":"Soltani","year":"2016","journal-title":"CoRR"},{"article-title":"Convolutional tensor-train LSTM for spatio-temporal learning","volume-title":"NeurIPS","author":"Su","key":"ref20"},{"key":"ref21","first-page":"5998","article-title":"Attention is all you need","volume-title":"NeurIPS","author":"Vaswani"},{"article-title":"Eidetic 3d LSTM: A model for video prediction and beyond","volume-title":"ICLR","author":"Wang","key":"ref22"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01261-8_1"},{"key":"ref24","article-title":"Lookahead optimizer: k steps forward, 1 step back","volume-title":"NeurIPS","volume":"32","author":"Zhang"},{"article-title":"Adabelief optimizer: Adapting stepsizes by the belief in observed gradients","volume-title":"NeurIPS","author":"Zhuang","key":"ref25"}],"event":{"name":"2022 IEEE International Conference on Image Processing (ICIP)","start":{"date-parts":[[2022,10,16]]},"location":"Bordeaux, France","end":{"date-parts":[[2022,10,19]]}},"container-title":["2022 IEEE International Conference on Image Processing (ICIP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9897158\/9897159\/09897974.pdf?arnumber=9897974","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,1,22]],"date-time":"2024-01-22T20:57:03Z","timestamp":1705957023000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9897974\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,10,16]]},"references-count":25,"URL":"https:\/\/doi.org\/10.1109\/icip46576.2022.9897974","relation":{},"subject":[],"published":{"date-parts":[[2022,10,16]]}}}