{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,13]],"date-time":"2026-06-13T04:55:10Z","timestamp":1781326510560,"version":"3.54.1"},"reference-count":33,"publisher":"Springer Science and Business Media LLC","issue":"10","license":[{"start":{"date-parts":[[2022,8,1]],"date-time":"2022-08-01T00:00:00Z","timestamp":1659312000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,8,1]],"date-time":"2022-08-01T00:00:00Z","timestamp":1659312000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2022,10]]},"DOI":"10.1007\/s11263-022-01649-x","type":"journal-article","created":{"date-parts":[[2022,8,1]],"date-time":"2022-08-01T03:24:43Z","timestamp":1659324283000},"page":"2349-2363","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":15,"title":["Weakly-Supervised Action Localization, and Action Recognition Using Global\u2013Local Attention of 3D CNN"],"prefix":"10.1007","volume":"130","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5330-5930","authenticated-orcid":false,"given":"Novanto","family":"Yudistira","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Muthu Subash","family":"Kavitha","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Takio","family":"Kurita","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2022,8,1]]},"reference":[{"key":"1649_CR1","unstructured":"Adebayo, J., et al. (2018). Sanity checks for saliency maps. Advances in Neural Information Processing Systems."},{"key":"1649_CR2","doi-asserted-by":"crossref","unstructured":"Bargal, S.A., et al. (2018)Excitation backprop for RNNs. In Proceedings of the IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2018.00156"},{"key":"1649_CR3","doi-asserted-by":"crossref","unstructured":"Bazzani, L., et al. (2016). Self-taught object localization with deep networks. In 2016 IEEE winter conference on applications of computer vision (WACV). IEEE.","DOI":"10.1109\/WACV.2016.7477688"},{"issue":"8","key":"1649_CR4","doi-asserted-by":"publisher","first-page":"1798","DOI":"10.1109\/TPAMI.2013.50","volume":"35","author":"Y Bengio","year":"2013","unstructured":"Bengio, Y., Courville, A., & Vincent, P. (2013). Representation learning: A review and new perspectives. IEEE Transactions on Pattern Analysis and Machine Intelligence, 35(8), 1798\u20131828.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"1649_CR5","doi-asserted-by":"crossref","unstructured":"Carreira, J., & Zisserman, A. (2017) Quo vadis, action recognition? a new model and the kinetics dataset. In 2017 IEEE conference on computer vision and pattern recognition (CVPR). IEEE.","DOI":"10.1109\/CVPR.2017.502"},{"key":"1649_CR6","doi-asserted-by":"crossref","unstructured":"Chattopadhay, A., et al. (2018) Grad-CAM++: Generalized gradient-based visual explanations for deep convolutional networks. In 2018 IEEE winter conference on applications of computer vision (WACV). IEEE.","DOI":"10.1109\/WACV.2018.00097"},{"key":"1649_CR7","doi-asserted-by":"publisher","first-page":"839","DOI":"10.1109\/WACV.2018.00097","volume":"2018","author":"A Chattopadhay","year":"2018","unstructured":"Chattopadhay, A., Sarkar, A., Howlader, P., & Balasubramanian, V. N. (2018). Grad-CAM++: Generalized gradient-based visual explanations for deep convolutional networks. IEEE Winter Conference on Applications of Computer Vision (WACV), 2018, 839\u2013847. https:\/\/doi.org\/10.1109\/WACV.2018.00097.","journal-title":"IEEE Winter Conference on Applications of Computer Vision (WACV)"},{"key":"1649_CR8","doi-asserted-by":"crossref","unstructured":"Chen, L., et al. (2017). Sca-cnn: Spatial and channel-wise attention in convolutional networks for image captioning. In Proceedings of the IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2017.667"},{"key":"1649_CR9","doi-asserted-by":"crossref","unstructured":"Choe, J., et al. (2020). Evaluating weakly supervised object localization methods right. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR42600.2020.00320"},{"key":"1649_CR10","doi-asserted-by":"crossref","unstructured":"Deng, J., et al. (2009) Imagenet: A large-scale hierarchical image database. In 2009 IEEE conference on computer vision and pattern recognition. IEEE.","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"1649_CR11","doi-asserted-by":"crossref","unstructured":"Fukui, H., et al. (2019). Attention branch network: Learning of attention mechanism for visual explanation. In Proceedings of the IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2019.01096"},{"key":"1649_CR12","unstructured":"Girdhar, R., & Deva R. (2017). Attentional pooling for action recognition. Advances in Neural Information Processing Systems."},{"key":"1649_CR13","doi-asserted-by":"crossref","unstructured":"Hara, K., Kataoka, H., & Satoh, Y. (2018) Can spatiotemporal 3D CNNs retrace the history of 2D CNNs and ImageNet. In Proceedings of the IEEE conference on computer vision and pattern recognition, Salt Lake City, UT, USA.","DOI":"10.1109\/CVPR.2018.00685"},{"key":"1649_CR14","doi-asserted-by":"crossref","unstructured":"He, K., et al. (2016). Deep residual learning for image recognition. In Proceedings of the IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2016.90"},{"key":"1649_CR15","doi-asserted-by":"publisher","first-page":"167","DOI":"10.1016\/j.neunet.2019.06.009","volume":"118","author":"K Kawaguchi","year":"2019","unstructured":"Kawaguchi, K., & Bengio, Y. (2019). Depth with nonlinearity creates no bad local minima in ResNets. Neural Networks, 118, 167\u2013174.","journal-title":"Neural Networks"},{"key":"1649_CR16","doi-asserted-by":"crossref","unstructured":"Kuehne, H., et al. (2013) Hmdb51: A large video database for human motion recognition. In High performance computing in science and engineering \u201812. (pp. 571-582). Berlin: Springer.","DOI":"10.1007\/978-3-642-33374-3_41"},{"key":"1649_CR17","doi-asserted-by":"crossref","unstructured":"Li, W., Xiatian, Z., & Shaogang, G. (2017). Harmonious attention network for person reidentification. In Proceedings of the IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2018.00243"},{"key":"1649_CR18","doi-asserted-by":"crossref","unstructured":"Oquab, M., et al. (2015) Is object localization for free?-weakly-supervised learning with convolutional neural networks. In Proceedings of the IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2015.7298668"},{"key":"1649_CR19","doi-asserted-by":"crossref","unstructured":"Preim, B., & Botha, C.P. (2013). Visual computing for medicine: Theory, algorithms, and applications. Newnes.","DOI":"10.1016\/B978-0-12-415873-3.00020-1"},{"issue":"3","key":"1649_CR20","doi-asserted-by":"publisher","first-page":"211","DOI":"10.1007\/s11263-015-0816-y","volume":"115","author":"O Russakovsky","year":"2015","unstructured":"Russakovsky, O., et al. (2015). Imagenet large scale visual recognition challenge. International Journal of Computer Vision, 115(3), 211\u2013252.","journal-title":"International Journal of Computer Vision"},{"key":"1649_CR21","doi-asserted-by":"publisher","first-page":"197","DOI":"10.1016\/j.media.2019.01.012","volume":"53","author":"J Schlemper","year":"2019","unstructured":"Schlemper, J., et al. (2019). Attention gated networks: Learning to leverage salient regions in medical images. Medical Image Analysis, 53, 197\u2013207.","journal-title":"Medical Image Analysis"},{"key":"1649_CR22","doi-asserted-by":"crossref","unstructured":"Selvaraju, R. R., et al.(2017) Grad-CAM: Visual explanations from deep networks via gradient-based localization. In ICCV.","DOI":"10.1109\/ICCV.2017.74"},{"key":"1649_CR23","unstructured":"Shamir, O. (2018). Are ResNets provably better than linear predictors?. Advances in neural information processing systems."},{"key":"1649_CR24","unstructured":"Shrikumar, A., Greenside, P., & Kundaje, A. (2017) Learning important features through propagating activation differences. In Proceedings of the 34th international conference on machine learning, (Volume 70. JMLR. org)."},{"key":"1649_CR25","unstructured":"Simonyan, K., & Zisserman, A. (2014). Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556"},{"key":"1649_CR26","unstructured":"Soomro, K., Zamir, A.R., Shah, M. (2012) UCF101: A dataset of 101 human actions classes from videos in the wild. arXiv preprint arXiv:1212.0402"},{"key":"1649_CR27","unstructured":"Sundararajan, M., Ankur, T., & Qiqi, Y. (2017). Axiomatic attribution for deep networks. In Proceedings of the 34th international conference on machine learning, (Volume 70. JMLR. org)."},{"key":"1649_CR28","doi-asserted-by":"crossref","unstructured":"Szegedy, C., et al. (2015) Going deeper with convolutions. In Proceedings of the IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"1649_CR29","doi-asserted-by":"crossref","unstructured":"Tran, D., et al. (2015) Learning spatiotemporal features with 3d convolutional networks. In Proceedings of the IEEE international conference on computer vision.","DOI":"10.1109\/ICCV.2015.510"},{"key":"1649_CR30","doi-asserted-by":"crossref","unstructured":"Xie, S., et al. (2017). Aggregated residual transformations for deep neural networks. In Proceedings of the IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2017.634"},{"key":"1649_CR31","doi-asserted-by":"crossref","unstructured":"Yudistira, N., & Kurita, T. (2017) Correlation Net: Spatio temporal multimodal deep learning for action recognition. arXiv preprint arXiv:1807.08291.","DOI":"10.1186\/s13640-017-0235-9"},{"key":"1649_CR32","volume-title":"Visualizing and understanding convolutional networks. European conference on computer vision","author":"MD Zeiler","year":"2014","unstructured":"Zeiler, M. D., & Fergus, R. (2014). Visualizing and understanding convolutional networks. European conference on computer vision. Cham: Springer."},{"issue":"10","key":"1649_CR33","doi-asserted-by":"publisher","first-page":"1084","DOI":"10.1007\/s11263-017-1059-x","volume":"126","author":"J Zhang","year":"2018","unstructured":"Zhang, J., et al. (2018). Top-down neural attention by excitation backprop. International Journal of Computer Vision, 126(10), 1084\u20131102.","journal-title":"International Journal of Computer Vision"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-022-01649-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-022-01649-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-022-01649-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,9,10]],"date-time":"2022-09-10T10:10:27Z","timestamp":1662804627000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-022-01649-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,8,1]]},"references-count":33,"journal-issue":{"issue":"10","published-print":{"date-parts":[[2022,10]]}},"alternative-id":["1649"],"URL":"https:\/\/doi.org\/10.1007\/s11263-022-01649-x","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,8,1]]},"assertion":[{"value":"22 January 2021","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 July 2022","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"1 August 2022","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}