{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,15]],"date-time":"2026-04-15T18:33:22Z","timestamp":1776278002020,"version":"3.50.1"},"reference-count":73,"publisher":"Springer Science and Business Media LLC","issue":"11","license":[{"start":{"date-parts":[[2021,8,18]],"date-time":"2021-08-18T00:00:00Z","timestamp":1629244800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,8,18]],"date-time":"2021-08-18T00:00:00Z","timestamp":1629244800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2021,11]]},"DOI":"10.1007\/s11263-021-01508-1","type":"journal-article","created":{"date-parts":[[2021,8,18]],"date-time":"2021-08-18T20:58:56Z","timestamp":1629320336000},"page":"2965-2977","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":44,"title":["A Coarse-to-Fine Framework for Resource Efficient Video Recognition"],"prefix":"10.1007","volume":"129","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8689-5807","authenticated-orcid":false,"given":"Zuxuan","family":"Wu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hengduo","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yingbin","family":"Zheng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Caiming","family":"Xiong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yu-Gang","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Larry S","family":"Davis","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2021,8,18]]},"reference":[{"key":"1508_CR1","unstructured":"Bejnordi, B. E., Blankevoort, T., & Welling, M. (2020). Batch-shaping for learning conditional channel gated networks. In ICLR."},{"key":"1508_CR2","unstructured":"Bolukbasi, T., Wang, J., Dekel, O., & Saligrama, V. (2017). Adaptive neural networks for fast test-time prediction. In ICML."},{"key":"1508_CR3","unstructured":"Chen, W., Wilson, J., Tyree, S., Weinberger, K., & Chen, Y. (2015). Compressing neural networks with the hashing trick. In ICML."},{"key":"1508_CR4","unstructured":"Chen, Y., Li, J., Xiao, H., Jin, X., Yan, S., & Feng, J. (2017). Dual path networks. In NIPS."},{"key":"1508_CR5","doi-asserted-by":"crossref","unstructured":"Chen, Y., Kalantidis, Y., Li, J., Yan, S., & Feng, J. (2018). Multi-fiber networks for video recognition. In ECCV.","DOI":"10.1007\/978-3-030-01246-5_22"},{"key":"1508_CR6","doi-asserted-by":"crossref","unstructured":"Cho, K., Van\u00a0Merri\u00ebnboer, B., Bahdanau, D., & Bengio, Y. (2014). On the properties of neural machine translation: Encoder\u2013decoder approaches. arXiv:1409.1259.","DOI":"10.3115\/v1\/W14-4012"},{"key":"1508_CR7","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L. J., Li, K., & Fei-Fei, L. (2009). Imagenet: A large-scale hierarchical image database. In CVPR.","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"1508_CR8","doi-asserted-by":"crossref","unstructured":"Donahue, J., Hendricks, L. A., Guadarrama, S., Rohrbach, M., Venugopalan, S., Saenko, K., & Darrell, T. (2015). Long-term recurrent convolutional networks for visual recognition and description. In CVPR.","DOI":"10.21236\/ADA623249"},{"key":"1508_CR9","unstructured":"Dong, X., & Yang, Y. (2019). Network pruning via transformable architecture search. In NeurIPS."},{"key":"1508_CR10","doi-asserted-by":"publisher","unstructured":"Fan, H., Xu, Z., Zhu, L., Yan, C., Ge, J., & Yang, Y. (2018). Watching a small portion could be as good as watching all: Towards efficient video classification. In IJCAI. https:\/\/doi.org\/10.24963\/ijcai.2018\/98.","DOI":"10.24963\/ijcai.2018\/98"},{"key":"1508_CR11","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C. (2020). X3d: Expanding architectures for efficient video recognition. In CVPR.","DOI":"10.1109\/CVPR42600.2020.00028"},{"key":"1508_CR12","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Pinz, A., & Zisserman, A. (2016). Convolutional two-stream network fusion for video action recognition. In CVPR.","DOI":"10.1109\/CVPR.2016.213"},{"key":"1508_CR13","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Fan, H., Malik, J., & He, K. (2019). Slowfast networks for video recognition. In ICCV.","DOI":"10.1109\/ICCV.2019.00630"},{"key":"1508_CR14","doi-asserted-by":"crossref","unstructured":"Gao, M., Yu, R., Li, A., Morariu, V. I., & Davis, L. S. (2018). Dynamic zoom-in network for fast object detection in large images. In CVPR.","DOI":"10.1109\/CVPR.2018.00724"},{"key":"1508_CR15","unstructured":"Hazan, T., & Jaakkola, T. S. (2012). On the partition function and random maximum a-posteriori perturbations. In ICML."},{"key":"1508_CR16","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2015). Delving deep into rectifiers: Surpassing human-level performance on imagenet classification. In ICCV.","DOI":"10.1109\/ICCV.2015.123"},{"key":"1508_CR17","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2016). Deep residual learning for image recognition. In CVPR.","DOI":"10.1109\/CVPR.2016.90"},{"key":"1508_CR18","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., & Girshick, R. (2017). Mask r-cnn. In ICCV.","DOI":"10.1109\/ICCV.2017.322"},{"key":"1508_CR19","doi-asserted-by":"crossref","unstructured":"He, Y., Lin, J., Liu, Z., Wang, H., Li, L. J., & Han, S. (2018). Amc: Automl for model compression and acceleration on mobile devices. In ECCV.","DOI":"10.1007\/978-3-030-01234-2_48"},{"key":"1508_CR20","doi-asserted-by":"crossref","unstructured":"Heilbron, F. C., Escorcia, V., Ghanem, B., & Niebles, J. C. (2015). Activitynet: A large-scale video benchmark for human activity understanding. In CVPR.","DOI":"10.1109\/CVPR.2015.7298698"},{"key":"1508_CR21","doi-asserted-by":"crossref","unstructured":"Hochreiter, S., & Schmidhuber, J. (1997). Long short-term memory. Neural Computation.","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"1508_CR22","unstructured":"Howard, A. G., Zhu, M., Chen, B., Kalenichenko, D., Wang, W., Weyand, T., Andreetto, M., & Adam, H. (2017). Mobilenets: Efficient convolutional neural networks for mobile vision applications. In CVPR."},{"key":"1508_CR23","doi-asserted-by":"crossref","unstructured":"Hu, J., Shen, L., & Sun, G. (2018). Squeeze-and-excitation networks. In CVPR.","DOI":"10.1109\/CVPR.2018.00745"},{"key":"1508_CR24","unstructured":"Huang, G., Chen, D., Li, T., Wu, F., van\u00a0der Maaten, L., & Weinberger, K. Q. (2018a). Multi-scale dense convolutional networks for efficient prediction. In ICLR."},{"key":"1508_CR25","unstructured":"Huang, G., Chen, D., Li, T., Wu, F., van\u00a0der Maaten, L., & Weinberger, K. Q. (2018b). Multi-scale dense networks for resource efficient image classification. In ICLR."},{"key":"1508_CR26","unstructured":"Iandola, F. N., Han, S., Moskewicz, M. W., Ashraf, K., Dally, W. J., & Keutzer, K. (2016). Squeezenet: Alexnet-level accuracy with 50x fewer parameters and $$<$$\u00a00.5\u00a0mb model size. arXiv:1602.07360."},{"key":"1508_CR27","unstructured":"Jang, E., Gu, S., & Poole, B. (2017). Categorical reparameterization with gumbel-softmax. In ICLR."},{"key":"1508_CR28","doi-asserted-by":"crossref","unstructured":"Jiang, Y. G., Wu, Z., Wang, J., Xue, X., & Chang, S. F. (2018). Exploiting feature and class relationships in video categorization with regularized deep neural networks. In IEEE TPAMI.","DOI":"10.1109\/TPAMI.2017.2670560"},{"key":"1508_CR29","unstructured":"Kay, W., Carreira, J., Simonyan, K., Zhang, B., Hillier, C., Vijayanarasimhan, S., Viola, F., Green, T., Back, T., Natsev, P., et\u00a0al. (2017). The kinetics human action video dataset. arXiv:1705.06950."},{"key":"1508_CR30","doi-asserted-by":"crossref","unstructured":"K\u00f6p\u00fckl\u00fc, O., Gunduz, A., Kose, N., & Rigoll, G. (2019). Real-time hand gesture detection and classification using convolutional neural networks. In FG.","DOI":"10.1109\/FG.2019.8756576"},{"key":"1508_CR31","doi-asserted-by":"crossref","unstructured":"Korbar, B., Tran, D., & Torresani, L. (2019). Scsampler: Sampling salient clips from video for efficient action recognition. In ICCV.","DOI":"10.1109\/ICCV.2019.00633"},{"key":"1508_CR32","doi-asserted-by":"crossref","unstructured":"Kuehne, H., Arslan, A., & Serre, T. (2014). The language of actions: Recovering the syntax and semantics of goal-directed human activities. In CVPR.","DOI":"10.1109\/CVPR.2014.105"},{"key":"1508_CR33","doi-asserted-by":"crossref","unstructured":"Lei, T., Zhang, Y., Wang, S. I., Dai, H., & Artzi, Y. (2017). Simple recurrent units for highly parallelizable recurrence. arXiv:1709.02755.","DOI":"10.18653\/v1\/D18-1477"},{"key":"1508_CR34","unstructured":"Li, H., Kadav, A., Durdanovic, I., Samet, H., & Graf, H. P. (2017). Pruning filters for efficient convnets. In ICLR."},{"key":"1508_CR35","unstructured":"Li, Z., Gavves, E., Jain, M., & Snoek, C. G. (2016). Videolstm convolves, attends and flows for action recognition. arXiv preprint arXiv:1607.01794."},{"key":"1508_CR36","unstructured":"Lin, J., Rao, Y., Lu, J., & Zhou, J. (2017). Runtime neural pruning. In NIPS."},{"key":"1508_CR37","doi-asserted-by":"crossref","unstructured":"Lin, J., Gan, C., & Han, S. (2019). Tsm: Temporal shift module for efficient video understanding. In ICCV.","DOI":"10.1109\/ICCV.2019.00718"},{"key":"1508_CR38","doi-asserted-by":"crossref","unstructured":"Lin, T. Y., Maire, M., Belongie, S., Bourdev, L., Girshick, R., Hays, J., Perona, P., Ramanan, D., Zitnick, C. L., & Doll\u00e1r, P. (2014). Microsoft coco: Common objects in context. In ECCV.","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"1508_CR39","doi-asserted-by":"crossref","unstructured":"Liu, Z., Miao, Z., Zhan, X., Wang, J., Gong, B., & Yu, S. X. (2019). Large-scale long-tailed recognition in an open world. In CVPR.","DOI":"10.1109\/CVPR.2019.00264"},{"key":"1508_CR40","unstructured":"Maddison, C. J., Mnih, A., & Teh, Y. W. (2017). The concrete distribution: A continuous relaxation of discrete random variables. In ICLR."},{"key":"1508_CR41","doi-asserted-by":"crossref","unstructured":"Molchanov, P., Gupta, S., Kim, K., & Pulli, K. (2015). Multi-sensor system for driver\u2019s hand-gesture recognition. In FG.","DOI":"10.1109\/FG.2015.7163132"},{"key":"1508_CR42","doi-asserted-by":"crossref","unstructured":"Najibi, M., Singh, B., & Davis, L. S. (2019). Autofocus: Efficient multi-scale inference. In ICCV.","DOI":"10.1109\/ICCV.2019.00984"},{"key":"1508_CR43","unstructured":"Ng, J. Y. H., Hausknecht, M., Vijayanarasimhan, S., Vinyals, O., Monga, R., & Toderici, G. (2015). Beyond short snippets: Deep networks for video classification. In CVPR."},{"key":"1508_CR44","doi-asserted-by":"crossref","unstructured":"Qiu, Z., Yao, T., & Mei, T. (2016). Deep quantization: Encoding convolutional activations with deep generative model. arXiv:1611.09502.","DOI":"10.1109\/CVPR.2017.435"},{"key":"1508_CR45","doi-asserted-by":"crossref","unstructured":"Qiu, Z., Yao, T., & Mei, T. (2017). Learning spatio-temporal representation with pseudo-3d residual networks. In ICCV.","DOI":"10.1109\/ICCV.2017.590"},{"key":"1508_CR46","doi-asserted-by":"crossref","unstructured":"Rastegari, M., Ordonez, V., Redmon, J., & Farhadi, A. (2016). Xnor-net: Imagenet classification using binary convolutional neural networks. In ECCV.","DOI":"10.1007\/978-3-319-46493-0_32"},{"key":"1508_CR47","unstructured":"Ren, S., He, K., Girshick, R., & Sun, J. (2015) Faster r-cnn: Towards real-time object detection with region proposal networks. In NeurIPS."},{"key":"1508_CR48","doi-asserted-by":"crossref","unstructured":"Sandler, M., Howard, A., Zhu, M., Zhmoginov, A., & Chen, L. C. (2018). Mobilenetv2: Inverted residuals and linear bottlenecks. In CVPR.","DOI":"10.1109\/CVPR.2018.00474"},{"key":"1508_CR49","unstructured":"Simonyan, K., & Zisserman, A. (2014). Two-stream convolutional networks for action recognition in videos. In NIPS."},{"key":"1508_CR50","volume-title":"Reinforcement learning: An introduction","author":"RS Sutton","year":"1998","unstructured":"Sutton, R. S., & Barto, A. G. (1998). Reinforcement learning: An introduction. Cambridge: MIT Press."},{"key":"1508_CR51","unstructured":"Tran, D., Bourdev, L. D., Fergus, R., Torresani, L., & Paluri, M. (2015). C3d: Generic features for video analysis. In ICCV."},{"key":"1508_CR52","doi-asserted-by":"crossref","unstructured":"Tran, D., Wang, H., Torresani, L., Ray, J., LeCun, Y., & Paluri, M. (2018). A closer look at spatiotemporal convolutions for action recognition. In CVPR.","DOI":"10.1109\/CVPR.2018.00675"},{"key":"1508_CR53","doi-asserted-by":"crossref","unstructured":"Tran, D., Wang, H., Torresani, L., & Feiszli, M. (2019). Video classification with channel-separated convolutional networks. In ICCV.","DOI":"10.1109\/ICCV.2019.00565"},{"key":"1508_CR54","doi-asserted-by":"crossref","unstructured":"Uzkent, B., & Ermon, S. (2020). Learning when and where to zoom with deep reinforcement learning. In CVPR.","DOI":"10.1109\/CVPR42600.2020.01236"},{"key":"1508_CR55","doi-asserted-by":"crossref","unstructured":"Veit, A., & Belongie, S. (2018) Convolutional networks with adaptive inference graphs. In ECCV.","DOI":"10.1007\/978-3-030-01246-5_1"},{"key":"1508_CR56","doi-asserted-by":"crossref","unstructured":"Viola, P., & Jones, M. J. (2004) Robust real-time face detection. In IJCV.","DOI":"10.1023\/B:VISI.0000013087.49260.fb"},{"key":"1508_CR57","doi-asserted-by":"crossref","unstructured":"Wang, H., & Schmid, C. (2013). Action recognition with improved trajectories. In ICCV.","DOI":"10.1109\/ICCV.2013.441"},{"key":"1508_CR58","doi-asserted-by":"crossref","unstructured":"Wang, L., Xiong, Y., Wang, Z., Qiao, Y., Lin, D., Tang, X., & Van Gool, L. (2016). Temporal segment networks: Towards good practices for deep action recognition. In ECCV.","DOI":"10.1007\/978-3-319-46484-8_2"},{"key":"1508_CR59","doi-asserted-by":"crossref","unstructured":"Wang, X., Girshick, R., Gupta, A., & He, K. (2018a). Non-local neural networks. In CVPR.","DOI":"10.1109\/CVPR.2018.00813"},{"key":"1508_CR60","doi-asserted-by":"crossref","unstructured":"Wang, X., Yu, F., Dou, Z. Y., & Gonzalez, J. E. (2018b). Skipnet: Learning dynamic routing in convolutional networks. In ECCV.","DOI":"10.1007\/978-3-030-01261-8_25"},{"key":"1508_CR61","doi-asserted-by":"crossref","unstructured":"Wu, C. Y., Zaheer, M., Hu, H., Manmatha, R., Smola, A. J., & Kr\u00e4henb\u00fchl, P. (2018a). Compressed video action recognition. In CVPR.","DOI":"10.1109\/CVPR.2018.00631"},{"key":"1508_CR62","doi-asserted-by":"crossref","unstructured":"Wu, W., He, D., Tan, X., Chen, S., & Wen, S. (2019a). Multi-agent reinforcement learning based frame sampling for effective untrimmed video recognition. In ICCV.","DOI":"10.1109\/ICCV.2019.00632"},{"key":"1508_CR63","doi-asserted-by":"crossref","unstructured":"Wu, Z., Nagarajan, T., Kumar, A., Rennie, S., Davis, L. S., Grauman, K., & Feris, R. (2018b). Blockdrop: Dynamic inference paths in residual networks. In CVPR.","DOI":"10.1109\/CVPR.2018.00919"},{"key":"1508_CR64","unstructured":"Wu, Z., Xiong, C., Jiang, Y. G., & Davis, L. S. (2019b). Liteeval: A coarse-to-fine framework for resource efficient video recognition. In NeurIPS."},{"key":"1508_CR65","doi-asserted-by":"crossref","unstructured":"Wu, Z., Xiong, C., Ma, C. Y., Socher, R., & Davis, L. S. (2019c). Adaframe: Adaptive frame selection for fast video recognition. In CVPR.","DOI":"10.1109\/CVPR.2019.00137"},{"key":"1508_CR66","doi-asserted-by":"crossref","unstructured":"Xie, S., Girshick, R., Doll\u00e1r, P., Tu, Z., & He, K. (2017). Aggregated residual transformations for deep neural networks. In CVPR.","DOI":"10.1109\/CVPR.2017.634"},{"key":"1508_CR67","doi-asserted-by":"crossref","unstructured":"Yang, L., Han, Y., Chen, X., Song, S., Dai, J., & Huang, G. (2020). Resolution adaptive networks for efficient inference. In CVPR.","DOI":"10.1109\/CVPR42600.2020.00244"},{"key":"1508_CR68","doi-asserted-by":"crossref","unstructured":"Yao, T., Ngo, C. W., & Zhu, S. (2012). Predicting domain adaptivity: Redo or recycle? In ACM Multimedia.","DOI":"10.1145\/2393347.2396321"},{"key":"1508_CR69","doi-asserted-by":"crossref","unstructured":"Yeung, S., Russakovsky, O., Mori, G., & Fei-Fei, L. (2016). End-to-end learning of action detection from frame glimpses in videos. In CVPR.","DOI":"10.1109\/CVPR.2016.293"},{"key":"1508_CR70","doi-asserted-by":"crossref","unstructured":"Zhang, B., Wang, L., Wang, Z., Qiao, Y., & Wang, H. (2016). Real-time action recognition with enhanced motion vector CNNs. In CVPR.","DOI":"10.1109\/CVPR.2016.297"},{"key":"1508_CR71","doi-asserted-by":"crossref","unstructured":"Zhou, B., Andonian, A., Oliva, A., & Torralba, A. (2018). Temporal relational reasoning in videos. In ECCV.","DOI":"10.1007\/978-3-030-01246-5_49"},{"key":"1508_CR72","doi-asserted-by":"crossref","unstructured":"Zhu, C., Tan, X., Zhou, F., Liu, X., Yue, K., Ding, E., & Ma, Y. (2018). Fine-grained video categorization with redundancy reduction attention. In ECCV.","DOI":"10.1007\/978-3-030-01228-1_9"},{"key":"1508_CR73","doi-asserted-by":"crossref","unstructured":"Zolfaghari, M., Singh, K., & Brox, T. (2018). Eco: Efficient convolutional network for online video understanding. In ECCV.","DOI":"10.1007\/978-3-030-01216-8_43"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-021-01508-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-021-01508-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-021-01508-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,10,17]],"date-time":"2021-10-17T09:53:37Z","timestamp":1634464417000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-021-01508-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,8,18]]},"references-count":73,"journal-issue":{"issue":"11","published-print":{"date-parts":[[2021,11]]}},"alternative-id":["1508"],"URL":"https:\/\/doi.org\/10.1007\/s11263-021-01508-1","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021,8,18]]},"assertion":[{"value":"9 December 2020","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 June 2021","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 August 2021","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}