{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,21]],"date-time":"2026-05-21T20:58:54Z","timestamp":1779397134878,"version":"3.53.1"},"reference-count":87,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2022,5,28]],"date-time":"2022-05-28T00:00:00Z","timestamp":1653696000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,5,28]],"date-time":"2022-05-28T00:00:00Z","timestamp":1653696000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Mach. Intell. Res."],"published-print":{"date-parts":[[2022,6]]},"DOI":"10.1007\/s11633-022-1333-4","type":"journal-article","created":{"date-parts":[[2022,5,28]],"date-time":"2022-05-28T14:03:32Z","timestamp":1653746612000},"page":"227-246","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":7,"title":["TwinNet: Twin Structured Knowledge Transfer Network for Weakly Supervised Action Localization"],"prefix":"10.1007","volume":"19","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-1630-6058","authenticated-orcid":false,"given":"Xiao-Yu","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8846-1853","authenticated-orcid":false,"given":"Hai-Chao","family":"Shi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chang-Sheng","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Li-Xin","family":"Duan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2022,5,28]]},"reference":[{"key":"1333_CR1","doi-asserted-by":"publisher","first-page":"568","DOI":"10.5555\/2968826.2968890","volume-title":"Two-stream convolutional networks for action recognition in videos","author":"K Simonyan","year":"2014","unstructured":"K. Simonyan, A. Zisserman. Two-stream convolutional networks for action recognition in videos. In Proceedings of the 27th International Conference on Neural Information Processing Systems, ACM, Montreal, Canada, pp. 568\u2013576, 2014. DOI: https:\/\/doi.org\/10.5555\/2968826.2968890."},{"key":"1333_CR2","doi-asserted-by":"publisher","first-page":"4489","DOI":"10.1109\/ICCV.2015.510","volume-title":"Learning spatiotemporal features with 3D convolutional networks","author":"D Tran","year":"2015","unstructured":"D. Tran, L. Bourdev, R. Fergus, L. Torresani, M. Paluri. Learning spatiotemporal features with 3D convolutional networks. In Proceedings of IEEE International Conference on Computer Vision, IEEE, Santiago, Chile, pp. 4489\u20134497, 2015. DOI: https:\/\/doi.org\/10.1109\/ICCV.2015.510."},{"key":"1333_CR3","doi-asserted-by":"publisher","first-page":"3169","DOI":"10.1109\/CVPR.2011.5995407","volume-title":"Action recognition by dense trajectories","author":"H Wang","year":"2011","unstructured":"H. Wang, A. Kl\u00e4ser, C. Schmid, C. L. Liu. Action recognition by dense trajectories. In Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, IEEE, Colorado Springs, USA, pp. 3169\u20133176, 2011. DOI: https:\/\/doi.org\/10.1109\/CVPR.2011.5995407."},{"key":"1333_CR4","doi-asserted-by":"publisher","first-page":"3551","DOI":"10.1109\/ICCV.2013.441","volume-title":"Action recognition with improved trajectories","author":"H Wang","year":"2013","unstructured":"H. Wang, C. Schmid. Action recognition with improved trajectories. In Proceedings of IEEE International Conference on Computer Vision, IEEE, Sydney, Australia, pp. 3551\u20133558, 2013. DOI: https:\/\/doi.org\/10.1109\/ICCV.2013.441."},{"key":"1333_CR5","doi-asserted-by":"publisher","first-page":"20","DOI":"10.1007\/978-3-319-46484-8_2","volume-title":"Temporal segment networks: Towards good practices for deep action recognition","author":"L M Wang","year":"2016","unstructured":"L. M. Wang, Y. J. Xiong, Z. Wang, Y. Qiao, D. H. Lin, X. O. Tang, L. Van Gool. Temporal segment networks: Towards good practices for deep action recognition. In Proceedings of the 14th European Conference on Computer Vision, Springer, Amsterdam, The Netherlands, pp. 20\u201336, 2016. DOI: https:\/\/doi.org\/10.1007\/978-3-319-46484-8_2."},{"key":"1333_CR6","doi-asserted-by":"publisher","first-page":"1933","DOI":"10.1109\/CVPR.2016.213","volume-title":"Convolutional two-stream network fusion for video action recognition","author":"C Feichtenhofer","year":"2016","unstructured":"C. Feichtenhofer, A. Pinz, A. Zisserman. Convolutional two-stream network fusion for video action recognition. In Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, IEEE, Las Vegas, USA, pp. 1933\u20131941, 2016. DOI: https:\/\/doi.org\/10.1109\/CVPR.2016.213."},{"key":"1333_CR7","doi-asserted-by":"publisher","first-page":"4724","DOI":"10.1109\/CVPR.2017.502","volume-title":"Quo vadis, action recognition? A new model and the kinetics dataset","author":"J Carreira","year":"2017","unstructured":"J. Carreira, A. Zisserman. Quo vadis, action recognition? A new model and the kinetics dataset. In Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, IEEE, Honolulu, USA, pp. 4724\u20134733, 2017. DOI: https:\/\/doi.org\/10.1109\/CVPR.2017.502."},{"key":"1333_CR8","doi-asserted-by":"publisher","first-page":"4694","DOI":"10.1109\/CVPR.2015.7299101","volume-title":"Beyond short snippets: Deep networks for video classification","author":"J Y H Ng","year":"2015","unstructured":"J. Y. H. Ng, M. Hausknecht, S. Vijayanarasimhan, O. Vinyals, R. Monga, G. Toderici. Beyond short snippets: Deep networks for video classification. In Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, IEEE, Boston, USA, pp. 4694\u20134702, 2015. DOI: https:\/\/doi.org\/10.1109\/CVPR.2015.7299101."},{"key":"1333_CR9","doi-asserted-by":"publisher","first-page":"2625","DOI":"10.1109\/CVPR.2015.7298878","volume-title":"Long-term recurrent convolutional networks for visual recognition and description","author":"J Donahue","year":"2015","unstructured":"J. Donahue, L. A. Hendricks, S. Guadarrama, M. Rohrbach, S. Venugopalan, T. Darrell, K. Saenko. Long-term recurrent convolutional networks for visual recognition and description. In Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, IEEE, Boston, USA, pp. 2625\u20132634, 2015. DOI: https:\/\/doi.org\/10.1109\/CVPR.2015.7298878."},{"key":"1333_CR10","series-title":"Technical Report CRCV-TR-12-01, Center for Research in Computer Vision","volume-title":"UCF101: A Dataset of 101 Human Actions Classes from Videos in the Wild","author":"K Soomro","year":"2012","unstructured":"K. Soomro, A. R. Zamir, M. Shah. UCF101: A Dataset of 101 Human Actions Classes from Videos in the Wild. Technical Report CRCV-TR-12-01, Center for Research in Computer Vision, University of Central Florida, USA, 2012."},{"key":"1333_CR11","doi-asserted-by":"publisher","first-page":"2556","DOI":"10.1109\/ICCV.2011.6126543","volume-title":"HMDB: A large video database for human motion recognition","author":"H Kuehne","year":"2011","unstructured":"H. Kuehne, H. Jhuang, E. Garrote, T. Poggio, T. Serre. HMDB: A large video database for human motion recognition. In Proceedings of International Conference on Computer Vision, IEEE, Barcelona, Spain, pp. 2556\u20132563, 2011. DOI: https:\/\/doi.org\/10.1109\/ICCV.2011.6126543."},{"key":"1333_CR12","doi-asserted-by":"publisher","first-page":"1725","DOI":"10.1109\/CVPR.2014.223","volume-title":"Large-scale video classification with convolutional neural networks","author":"A Karpathy","year":"2014","unstructured":"A. Karpathy, G. Toderici, S. Shetty, T. Leung, R. Sukthankar, F. F. Li. Large-scale video classification with convolutional neural networks. In Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, IEEE, Columbus, USA, pp. 1725\u20131732, 2014. DOI: https:\/\/doi.org\/10.1109\/CVPR.2014.223."},{"key":"1333_CR13","unstructured":"S. Karaman, L. Seidenari, A. Del Bimbo. Fast saliency based pooling of fisher encoded dense trajectories. In Proceedings of ECCV THUMOS Workshop, Zurich, Switzerland, 2014."},{"key":"1333_CR14","unstructured":"D. Oneata, J. Verbeek, C. Schmid. The LEAR submission at THUMOS 2014. 2014."},{"key":"1333_CR15","unstructured":"G. Singh, F. Cuzzolin. Untrimmed video classification for activity detection: Submission to activityNet challenge. [Online], Available: https:\/\/arxiv.org\/abs\/1607.01979, 2016."},{"key":"1333_CR16","unstructured":"L. M. Wang, Y. Qiao, X. O. Tang. Action recognition and detection by combining motion and appearance features. THUMOS14 Action Recognition Challenge, vol. 1, no. 2, Article number 2, 2014."},{"key":"1333_CR17","doi-asserted-by":"publisher","first-page":"768","DOI":"10.1007\/978-3-319-46487-9_47","volume-title":"DAPs: Deep action proposals for action understanding","author":"V Escorcia","year":"2016","unstructured":"V. Escorcia, F. C. Heilbron, J. C. Niebles, B. Ghanem. DAPs: Deep action proposals for action understanding. In Proceedings of the 14th European Conference on Computer Vision, Springer, Amsterdam, The Netherlands, pp. 768\u2013784, 2016. DOI: https:\/\/doi.org\/10.1007\/978-3-319-46487-9_47."},{"key":"1333_CR18","doi-asserted-by":"publisher","first-page":"437","DOI":"10.1007\/978-3-319-46454-1_27","volume-title":"Spot on: Action localization from pointly-supervised proposals","author":"P Mettes","year":"2016","unstructured":"P. Mettes, J. C. Van Gemert, C. G. M. Snoek. Spot on: Action localization from pointly-supervised proposals. In Proceedings of the 14th European Conference on Computer Vision, Springer, Amsterdam, The Netherlands, pp. 437\u2013453, 2016. DOI: https:\/\/doi.org\/10.1007\/978-3-319-46454-1_27."},{"key":"1333_CR19","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1007\/978-3-030-01225-0_1","volume-title":"BSN: Boundary sensitive network for temporal action proposal generation","author":"T W Lin","year":"2018","unstructured":"T. W. Lin, X. Zhao, H. S. Su, C. J. Wang, M. Yang. BSN: Boundary sensitive network for temporal action proposal generation. In Proceedings of the 15th European Conference on Computer Vision, Springer, Munich, Germany, pp. 3\u201321, 2018. DOI: https:\/\/doi.org\/10.1007\/978-3-030-01225-0_1."},{"key":"1333_CR20","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.214","volume-title":"Learning activity progression in LSTMs for activity detection and early detection","author":"S G Ma","year":"2016","unstructured":"S. G. Ma, L. Sigal, S. Sclaroff. Learning activity progression in LSTMs for activity detection and early detection. In Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, IEEE, Las Vegas, USA, pp. 1942\u20131950, 2016. DOI: https:\/\/doi.org\/10.1109\/CVPR.2016.214."},{"key":"1333_CR21","doi-asserted-by":"publisher","first-page":"1961","DOI":"10.1109\/CVPR.2016.216","volume-title":"A multi-stream bi-directional recurrent neural network for fine-grained action detection","author":"B Singh","year":"2016","unstructured":"B. Singh, T. K. Marks, M. Jones, O. Tuzel, M. Shao. A multi-stream bi-directional recurrent neural network for fine-grained action detection. In Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, IEEE, Las Vegas, USA, pp. 1961\u20131970, 2016. DOI: https:\/\/doi.org\/10.1109\/CVPR.2016.216."},{"key":"1333_CR22","doi-asserted-by":"publisher","first-page":"2678","DOI":"10.1109\/CVPR.2016.293","volume-title":"End-to-end learning of action detection from frame glimpses in videos","author":"S Yeung","year":"2016","unstructured":"S. Yeung, O. Russakovsky, G. Mori, F. F. Li. End-to-end learning of action detection from frame glimpses in videos. In Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, IEEE, Las Vegas, USA, pp. 2678\u20132687, 2016. DOI: https:\/\/doi.org\/10.1109\/CVPR.2016.293."},{"key":"1333_CR23","doi-asserted-by":"publisher","first-page":"3544","DOI":"10.1109\/ICCV.2017.381","volume-title":"Hide-and-seek: Forcing a network to be meticulous for weakly-supervised object and action localization","author":"K K Singh","year":"2017","unstructured":"K. K. Singh, Y. J. Lee. Hide-and-seek: Forcing a network to be meticulous for weakly-supervised object and action localization. In Proceedings of IEEE International Conference on Computer Vision, IEEE, Venice, Italy, pp. 3544\u20133553, 2017. DOI: https:\/\/doi.org\/10.1109\/ICCV.2017.381."},{"key":"1333_CR24","doi-asserted-by":"publisher","unstructured":"L. M. Wang, Y. J. Xiong, D. H. Lin, L. Van Gool. UntrimmedNets for weakly supervised action recognition and detection. In Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, Honolulu, USA, pp. 6402\u20136411, 2017. DOI: https:\/\/doi.org\/10.1109\/CVPR.2017.678.","DOI":"10.1109\/CVPR.2017.678"},{"key":"1333_CR25","doi-asserted-by":"publisher","first-page":"6752","DOI":"10.1109\/CVPR.2018.00706","volume-title":"Weakly supervised action localization by sparse temporal pooling network","author":"P Nguyen","year":"2018","unstructured":"P. Nguyen, B. Han, T. Liu, G. Prasad. Weakly supervised action localization by sparse temporal pooling network. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, IEEE, Salt Lake City, USA, pp. 6752\u20136761, 2018. DOI: https:\/\/doi.org\/10.1109\/CVPR.2018.00706."},{"key":"1333_CR26","doi-asserted-by":"publisher","first-page":"162","DOI":"10.1007\/978-3-030-01270-010","volume-title":"Autoloc: Weakly-supervised temporal action localization in untrimmed videos","author":"Z Shou","year":"2018","unstructured":"Z. Shou, H. Gao, L. Zhang, K. Miyazawa, S. F. Chang. Autoloc: Weakly-supervised temporal action localization in untrimmed videos. In Proceedings of the 15th European Conference on Computer Vision, Springer, Munich, Germany, pp. 162\u2013179, 2018. DOI: https:\/\/doi.org\/10.1007\/978-3-030-01270-010."},{"key":"1333_CR27","doi-asserted-by":"publisher","first-page":"588","DOI":"10.1007\/978-3-030-01225-0_35","volume-title":"W-TALC: Weakly-supervised temporal activity localization and classification","author":"S Paul","year":"2018","unstructured":"S. Paul, S. Roy, A. K. Roy-Chowdhury. W-TALC: Weakly-supervised temporal activity localization and classification. In Proceedings of the 15th European Conference on Computer Vision, Springer, Munich, Germany, pp. 588\u2013607, 2018. DOI: https:\/\/doi.org\/10.1007\/978-3-030-01225-0_35."},{"key":"1333_CR28","unstructured":"Y. G. Jiang, J. G. Liu, A. R. Zamir, G. Toderici. THUMOS Challenge: Action recognition with a large number of classes. In Proceedings of ECCV2014 International Workshop and Competition. [Online], Available: http:\/\/crcv.ucf.edu\/THUMOS14\/home.html, 2014."},{"key":"1333_CR29","doi-asserted-by":"publisher","first-page":"961","DOI":"10.1109\/CVPR.2015.7298698","volume-title":"ActivityNet: A large-scale video benchmark for human activity understanding","author":"F C Heilbron","year":"2015","unstructured":"F. C. Heilbron, V. Escorcia, B. Ghanem, J. C. Niebles. ActivityNet: A large-scale video benchmark for human activity understanding. In Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, IEEE, Boston, USA, pp. 961\u2013970, 2015. DOI: https:\/\/doi.org\/10.1109\/CVPR.2015.7298698."},{"key":"1333_CR30","unstructured":"M. Crucianu. MEXaction2: Action Detection and Localization Dataset.[Online], Available: http:\/\/mexculture.cnam.fr\/Datasets\/mex+action+dataset.html, July 15, 2015."},{"issue":"2","key":"1333_CR31","doi-asserted-by":"publisher","first-page":"373","DOI":"10.1007\/s10994-019-05855-6","volume":"109","author":"J E Van Engelen","year":"2020","unstructured":"J. E. Van Engelen, H. H. Hoos. A survey on semi-supervised learning. Machine Learning, vol. 109, no. 2, pp. 373\u2013440, 2020. DOI: https:\/\/doi.org\/10.1007\/s10994-019-05855-6.","journal-title":"Machine Learning"},{"issue":"3","key":"1333_CR32","doi-asserted-by":"publisher","first-page":"415","DOI":"10.1007\/s10115-009-0209-z","volume":"24","author":"Z H Zhou","year":"2010","unstructured":"Z. H. Zhou, M. Li. Semi-supervised learning by disagreement. Knowledge and Information Systems, vol. 24, no. 3, pp. 415\u2013439, 2010. DOI: https:\/\/doi.org\/10.1007\/s10115-009-0209-z.","journal-title":"Knowledge and Information Systems"},{"issue":"1","key":"1333_CR33","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1017\/S026988890999035X","volume":"25","author":"J Foulds","year":"2010","unstructured":"J. Foulds, E. Frank. A review of multi-instance learning assumptions. The Knowledge Engineering Review, vol. 25, no. 1, pp. 1\u201325, 2010. DOI: https:\/\/doi.org\/10.1017\/S026988890999035X.","journal-title":"The Knowledge Engineering Review"},{"issue":"5","key":"1333_CR34","doi-asserted-by":"publisher","first-page":"800","DOI":"10.1007\/s11390-006-0800-7","volume":"21","author":"Z H Zhou","year":"2006","unstructured":"Z. H. Zhou. Multi-instance learning from supervised view. Journal of Computer Science and Technology, vol. 21, no. 5, pp. 800\u2013809, 2006. DOI: https:\/\/doi.org\/10.1007\/s11390-006-0800-7.","journal-title":"Journal of Computer Science and Technology"},{"issue":"5","key":"1333_CR35","doi-asserted-by":"publisher","first-page":"845","DOI":"10.1109\/TNNLS.2013.2292894","volume":"25","author":"B Frenay","year":"2014","unstructured":"B. Frenay, M. Verleysen. Classification in the presence of label noise: A survey. IEEE Transactions on Neural Networks and Learning Systems, vol. 25, no. 5, pp. 845\u2013869, 2014. DOI: https:\/\/doi.org\/10.1109\/TNNLS.2013.2292894.","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"1333_CR36","doi-asserted-by":"publisher","first-page":"1575","DOI":"10.5555\/3016100.3016119","volume-title":"Risk minimization in the presence of label noise","author":"W Gao","year":"2016","unstructured":"W. Gao, L. Wang, Y. F. Li, Z. H. Zhou. Risk minimization in the presence of label noise. In Proceedings of the 30th AAAI Conference on Artificial Intelligence, AAAI Press, Phoenix, USA, pp. 1575\u20131581, 2016. DOI: https:\/\/doi.org\/10.5555\/3016100.3016119."},{"key":"1333_CR37","unstructured":"D. Bahdanau, K. Cho, Y. Bengio. Neural machine translation by jointly learning to align and translate. [Online], Available: https:\/\/arxiv.org\/abs\/1409.0473, 2014."},{"key":"1333_CR38","unstructured":"K. Gregor, I. Danihelka, A. Graves, D. Rezende, D. Wierstra. Draw: A recurrent neural network for image generation. In Proceedings of the 32nd International Conference on Machine Learning, Lille, France, pp. 1462\u20131471, 2015."},{"key":"1333_CR39","doi-asserted-by":"publisher","first-page":"2048","DOI":"10.5555\/3045118.3045336","volume-title":"Show, attend and tell: Neural image caption generation with visual attention","author":"K Xu","year":"2015","unstructured":"K. Xu, J. L. Ba, R. Kiros, K. Cho, A. Courville, R. Salakhutdinov, R. S. Zemel, Y. Bengio. Show, attend and tell: Neural image caption generation with visual attention. In Proceedings of the 32nd International Conference on Machine Learning, ACM, Lille, France, pp. 2048\u20132057, 2015. DOI: https:\/\/doi.org\/10.5555\/3045118.3045336."},{"key":"1333_CR40","doi-asserted-by":"publisher","first-page":"21","DOI":"10.1109\/CVPR.2016.10","volume-title":"Stacked attention networks for image question answering","author":"Z C Yang","year":"2016","unstructured":"Z. C. Yang, X. D. He, J. F. Gao, L. Deng, A. Smola. Stacked attention networks for image question answering. In Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, IEEE, Las Vegas, USA, pp. 21\u201329, 2016. DOI: https:\/\/doi.org\/10.1109\/CVPR.2016.10."},{"key":"1333_CR41","first-page":"551","volume-title":"Long short-term memory-networks for machine reading","author":"J P Cheng","year":"2016","unstructured":"J. P. Cheng, L. Dong, M. Lapata. Long short-term memory-networks for machine reading. In Proceedings of Conference on Empirical Methods in Natural Language Processing, ACL, Austin, USA, pp. 551\u2013561, 2016."},{"key":"1333_CR42","doi-asserted-by":"publisher","first-page":"2249","DOI":"10.18653\/vl\/D16-1244","volume-title":"A decomposable attention model for natural language inference","author":"A P Parikh","year":"2016","unstructured":"A. P. Parikh, O. T\u00e4ckstr\u00f6m, D. Das, J. Uszkoreit. A decomposable attention model for natural language inference. In Proceedings of Conference on Empirical Methods in Natural Language Processing, ACL, Austin, USA, pp. 2249\u20132255, 2016. DOI: https:\/\/doi.org\/10.18653\/vl\/D16-1244."},{"key":"1333_CR43","unstructured":"Z. H. Lin, M. W. Feng, C. N. Dos Santos, M. Yu, B. Xiang, B. W. Zhou, Y. Bengio. A structured self-attentive sentence embedding. In Proceedings of the 5th International Conference on Learning Representations, Toulon, France, 2017."},{"key":"1333_CR44","unstructured":"R. Paulus, C. M. Xiong, R. Socher. A deep reinforced model for abstractive summarization. In Proceedings of the 6th International Conference on Learning Representations, Vancouver, Canada, 2018."},{"key":"1333_CR45","doi-asserted-by":"publisher","first-page":"6000","DOI":"10.5555\/3295222.3295349","volume-title":"Attention is all you need","author":"A Vaswani","year":"2017","unstructured":"A. Vaswani, N. Shazeer, N. Parmar, J. Uszkoreit, L. Jones, A. N. Gomez, L. Kaiser, I. Polosukhin. Attention is all you need. In Proceedings of the 31st International Conference on Neural Information Processing Systems, ACM, Long Beach, USA, pp. 6000\u20136010, 2017. DOI: https:\/\/doi.org\/10.5555\/3295222.3295349."},{"key":"1333_CR46","first-page":"4055","volume-title":"Image transformer","author":"N Parmar","year":"2018","unstructured":"N. Parmar, A. Vaswani, J. Uszkoreit, L. Kaiser, N. Shazeer, A. Ku, D. Tran. Image transformer. In Proceedings of the 35th International Conference on Machine Learning, Stockholm, Sweden, pp. 4055\u20134064, 2018."},{"key":"1333_CR47","doi-asserted-by":"publisher","first-page":"7794","DOI":"10.1109\/CVPR.2018.00813","volume-title":"Non-local neural networks","author":"X L Wang","year":"2018","unstructured":"X. L. Wang, R. Girshick, A. Gupta, K. M. He. Non-local neural networks. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, IEEE, Salt Lake City, USA, pp. 7794\u20137803, 2018. DOI: https:\/\/doi.org\/10.1109\/CVPR.2018.00813."},{"key":"1333_CR48","doi-asserted-by":"publisher","first-page":"1316","DOI":"10.1109\/CVPR.2018.00143","volume-title":"AttnGAN: Fine-grained text to image generation with attentional generative adversarial networks","author":"T Xu","year":"2018","unstructured":"T. Xu, P. C. Zhang, Q. Y. Huang, H. Zhang, Z. Gan, X. L. Huang, X. D. He. AttnGAN: Fine-grained text to image generation with attentional generative adversarial networks. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, IEEE, Salt Lake City, USA, pp. 1316\u20131324, 2018. DOI: https:\/\/doi.org\/10.1109\/CVPR.2018.00143."},{"key":"1333_CR49","first-page":"7354","volume-title":"Self-attention generative adversarial networks","author":"H Zhang","year":"2019","unstructured":"H. Zhang, I. Goodfellow, D. Metaxas, A. Odena. Self-attention generative adversarial networks. In Proceedings of the 36th International Conference on Machine Learning, PMLR, Long Beach, USA, pp. 7354\u20137363, 2019."},{"key":"1333_CR50","unstructured":"Y. Liu, C. J. Sun, L. Lin, X. L. Wang. Learning natural language inference using bidirectional LSTM model and inner-attention. [Online], Available: https:\/\/arxiv.org\/abs\/1605.09090, 2016."},{"key":"1333_CR51","volume-title":"Learning to Learn","author":"S Thrun","year":"2012","unstructured":"S. Thrun, L. Pratt. Learning to Learn, New York, USA: Springer, 2012."},{"issue":"10","key":"1333_CR52","doi-asserted-by":"publisher","first-page":"1345","DOI":"10.1109\/TKDE.2009.191","volume":"22","author":"S J Pan","year":"2010","unstructured":"S. J. Pan, Q. Yang. A survey on transfer learning. IEEE Transactions on Knowledge and Data Engineering, vol. 22, no. 10, pp. 1345\u20131359, 2010. DOI: https:\/\/doi.org\/10.1109\/TKDE.2009.191.","journal-title":"IEEE Transactions on Knowledge and Data Engineering"},{"issue":"2","key":"1333_CR53","doi-asserted-by":"publisher","first-page":"199","DOI":"10.1109\/TNN.2010.2091281","volume":"22","author":"S J Pan","year":"2011","unstructured":"S. J. Pan, I. W. Tsang, J. T. Kwok, Q. Yang. Domain adaptation via transfer component analysis. IEEE Transactions on Neural Networks, vol. 22, no. 2, pp. 199\u2013210, 2011. DOI: https:\/\/doi.org\/10.1109\/TNN.2010.2091281.","journal-title":"IEEE Transactions on Neural Networks"},{"key":"1333_CR54","doi-asserted-by":"publisher","first-page":"I","DOI":"10.5555\/3042817.3042844","volume-title":"Connecting the dots with landmarks: Discriminatively learning domain-invariant features for unsupervised domain adaptation","author":"B Q Gong","year":"2013","unstructured":"B. Q. Gong, K. Grauman, F. Sha. Connecting the dots with landmarks: Discriminatively learning domain-invariant features for unsupervised domain adaptation. In Proceedings of the 30th International Conference on Machine Learning, ACM, Atlanta, USA, pp. I\u2013222\u2013I\u2013230, 2013. DOI: https:\/\/doi.org\/10.5555\/3042817.3042844."},{"key":"1333_CR55","doi-asserted-by":"publisher","first-page":"213","DOI":"10.1007\/978-3-642-15561-116","volume-title":"Adapting visual category models to new domains","author":"K Saenko","year":"2010","unstructured":"K. Saenko, B. Kulis, M. Fritz, T. Darrell. Adapting visual category models to new domains. In Proceedings of the 11th European Conference on Computer Vision, Springer, Heraklion, Greece, pp. 213\u2013226, 2010. DOI: https:\/\/doi.org\/10.1007\/978-3-642-15561-116."},{"key":"1333_CR56","doi-asserted-by":"publisher","first-page":"2200","DOI":"10.1109\/ICCV.2013.274","volume-title":"Transfer feature learning with joint distribution adaptation","author":"M S Long","year":"2013","unstructured":"M. S. Long, J. M. Wang, G. G. Ding, J. G. Sun, P. S. Yu. Transfer feature learning with joint distribution adaptation. In Proceedings of IEEE International Conference on Computer Vision, IEEE, Sydney, Australia, pp. 2200\u20132207, 2013. DOI: https:\/\/doi.org\/10.1109\/ICCV.2013.274."},{"issue":"1","key":"1333_CR57","doi-asserted-by":"publisher","first-page":"723","DOI":"10.5555\/2188385.2188410","volume":"13","author":"A Gretton","year":"2012","unstructured":"A. Gretton, K. M. Borgwardt, M. J. Rasch, B. Sch\u00f6lkopf. A kernel two-sample test. The Journal of Machine Learning Research, vol. 13, no. 1, pp. 723\u2013773, 2012. DOI: https:\/\/doi.org\/10.5555\/2188385.2188410.","journal-title":"The Journal of Machine Learning Research"},{"key":"1333_CR58","doi-asserted-by":"publisher","first-page":"2208","DOI":"10.5555\/3305890.3305909","volume-title":"Deep transfer learning with joint adaptation networks","author":"M S Long","year":"2017","unstructured":"M. S. Long, H. Zhu, J. M. Wang, M. I. Jordan. Deep transfer learning with joint adaptation networks. In Proceedings of the 34th International Conference on Machine Learning, ACM, Sydney, Australia, pp. 2208\u20132217, 2017. DOI: https:\/\/doi.org\/10.5555\/3305890.3305909."},{"key":"1333_CR59","doi-asserted-by":"publisher","first-page":"97","DOI":"10.5555\/3045118.3045130","volume-title":"Learning transferable features with deep adaptation networks","author":"M S Long","year":"2015","unstructured":"M. S. Long, Y. Cao, J. M. Wang, et al. Learning transferable features with deep adaptation networks. In Proceedings of the 32nd International Conference on Machine Learning, ACM, Lille, France, pp. 97\u2013105, 2015. DOI: https:\/\/doi.org\/10.5555\/3045118.3045130."},{"key":"1333_CR60","unstructured":"J. Yosinski, J. Clune, Y. Bengio, H. Lipson. How transferable are features in deep neural networks? In Proceedings of the 27th International Conference on Neural Information Processing Systems, Montreal, Canada, pp. 3320\u20133328, 2014."},{"key":"1333_CR61","doi-asserted-by":"publisher","first-page":"513","DOI":"10.5555\/3104482.3104547","volume-title":"Domain adaptation for large-scale sentiment classification: A deep learning approach","author":"X Glorot","year":"2011","unstructured":"X. Glorot, A. Bordes, Y. Bengio. Domain adaptation for large-scale sentiment classification: A deep learning approach. In Proceedings of the 28th International Conference on Machine Learning, ACM, Bellevue, USA, pp. 513\u2013520, 2011. DOI: https:\/\/doi.org\/10.5555\/3104482.3104547."},{"key":"1333_CR62","doi-asserted-by":"publisher","first-page":"1627","DOI":"10.5555\/3042573.3042781","volume-title":"Marginalized denoising autoencoders for domain adaptation","author":"M M Chen","year":"2012","unstructured":"M. M. Chen, Z. X. Xu, K. Q. Weinberger, F. Sha. Marginalized denoising autoencoders for domain adaptation. In Proceedings of the 29th International Conference on Machine Learning, ACM, Edinburgh, UK, pp. 1627\u20131634, 2012. DOI: https:\/\/doi.org\/10.5555\/3042573.3042781."},{"key":"1333_CR63","doi-asserted-by":"publisher","first-page":"766","DOI":"10.1145\/2487575.2487612","volume-title":"Multi-source deep learning for information trustworthiness estimation","author":"L Ge","year":"2013","unstructured":"L. Ge, J. Gao, X. Y. Li, A. D. Zhang. Multi-source deep learning for information trustworthiness estimation. In Proceedings of the 19th ACM SIGKDD International Conference on Knowledge Discovery and Data Mining, ACM, Chicago, USA, pp. 766\u2013774, 2013. DOI: https:\/\/doi.org\/10.1145\/2487575.2487612."},{"key":"1333_CR64","doi-asserted-by":"publisher","first-page":"689","DOI":"10.5555\/3104482.3104569","volume-title":"Multimodal deep learning","author":"J Ngiam","year":"2011","unstructured":"J. Ngiam, A. Khosla, M. Kim, J. Nam, H. Lee, A. Y. Ng. Multimodal deep learning. In Proceedings of the 28th International Conference on Machine Learning, ACM, Bellevue, USA, pp. 689\u2013696, 2011. DOI: https:\/\/doi.org\/10.5555\/3104482.3104569."},{"key":"1333_CR65","doi-asserted-by":"publisher","first-page":"2962","DOI":"10.1109\/CVPR.2017.316","volume-title":"Adversarial discriminative domain adaptation","author":"E Tzeng","year":"2017","unstructured":"E. Tzeng, J. Hoffman, K. Saenko, T. Darrell. Adversarial discriminative domain adaptation. In Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, IEEE, Honolulu, USA, pp. 2962\u20132971, 2017. DOI: https:\/\/doi.org\/10.1109\/CVPR.2017.316."},{"key":"1333_CR66","doi-asserted-by":"publisher","first-page":"4068","DOI":"10.1109\/ICCV.2015.463","volume-title":"Simultaneous deep transfer across domains and tasks","author":"E Tzeng","year":"2015","unstructured":"E. Tzeng, J. Hoffman, T. Darrell, K. Saenko. Simultaneous deep transfer across domains and tasks. In Proceedings of IEEE International Conference on Computer Vision, IEEE, Santiago, Chile, pp. 4068\u20134076, 2015. DOI: https:\/\/doi.org\/10.1109\/ICCV.2015.463."},{"key":"1333_CR67","doi-asserted-by":"publisher","first-page":"1180","DOI":"10.5555\/3045118.3045244","volume-title":"Unsupervised domain adaptation by backpropagation","author":"Y Ganin","year":"2015","unstructured":"Y. Ganin, V. Lempitsky. Unsupervised domain adaptation by backpropagation. In Proceedings of the 32nd International Conference on Machine Learning, ACM, Lille, France, pp. 1180\u20131189, 2015. DOI: https:\/\/doi.org\/10.5555\/3045118.3045244."},{"key":"1333_CR68","doi-asserted-by":"publisher","first-page":"136","DOI":"10.5555\/3157096.3157112","volume-title":"Unsupervised domain adaptation with residual transfer networks","author":"M S Long","year":"2016","unstructured":"M. S. Long, H. Zhu, J. M. Wang, M. I. Jordan. Unsupervised domain adaptation with residual transfer networks. In Proceedings of the 30th International Conference on Neural Information Processing Systems, ACM, Barcelona, Spain, pp. 136\u2013144, 2016. DOI: https:\/\/doi.org\/10.5555\/3157096.3157112."},{"key":"1333_CR69","volume-title":"Domain adaptation under target and conditional shift","author":"K Zhang","year":"2013","unstructured":"K. Zhang, B. Sch\u00f6lkopf, K. Muandet, Z. K. Wang. Domain adaptation under target and conditional shift. In Proceedings of the 30th International Conference on Machine Learning, ACM, Atlanta, USA, 2013."},{"key":"1333_CR70","doi-asserted-by":"publisher","first-page":"5001","DOI":"10.1109\/CVPR.2018.00525","volume-title":"Cross-domain weakly-supervised object detection through progressive domain adaptation","author":"N Inoue","year":"2018","unstructured":"N. Inoue, R. Furuta, T. Yamasaki, K. Aizawa. Cross-domain weakly-supervised object detection through progressive domain adaptation. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, IEEE, Salt Lake City, USA, pp. 5001\u20135009, 2018. DOI: https:\/\/doi.org\/10.1109\/CVPR.2018.00525."},{"key":"1333_CR71","doi-asserted-by":"publisher","first-page":"1431","DOI":"10.1109\/ICCV.2015.168","volume-title":"Webly supervised learning of convolutional networks","author":"X L Chen","year":"2015","unstructured":"X. L. Chen, A. Gupta. Webly supervised learning of convolutional networks. In Proceedings of IEEE International Conference on Computer Vision, IEEE, Santiago, Chile, pp. 1431\u20131439, 2015. DOI: https:\/\/doi.org\/10.1109\/ICCV.2015.168."},{"key":"1333_CR72","doi-asserted-by":"publisher","unstructured":"B. W. Zhang, L. M. Wang, Z. Wang, Y. Qiao, H. L. Wang. Real-time action recognition with enhanced motion vector CNNs. In Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, Las Vegas, USA, pp. 2718\u20132726, 2016. DOI: https:\/\/doi.org\/10.1109\/CVPR.2016.297.","DOI":"10.1109\/CVPR.2016.297"},{"key":"1333_CR73","doi-asserted-by":"publisher","first-page":"46","DOI":"10.1109\/CVPR.2015.7298599","volume-title":"What do 15, 000 object categories tell us about classifying and localizing actions?","author":"M Jain","year":"2015","unstructured":"M. Jain, J. C. Van Gemert, C. G. M. Snoek. What do 15, 000 object categories tell us about classifying and localizing actions?. In Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, IEEE, Boston, USA, pp. 46\u201355, 2015. DOI: https:\/\/doi.org\/10.1109\/CVPR.2015.7298599."},{"key":"1333_CR74","volume-title":"University of Amsterdam at THUMOS challenge 2014","author":"M Jain","year":"2014","unstructured":"M. Jain, J. Van Gemert, C. G. M. Snoek. University of Amsterdam at THUMOS challenge 2014. In Proceddings of the 14th ECCV, THUMOS Challenge, ECCV, Orlando, USA, 2014."},{"issue":"21","key":"1333_CR75","doi-asserted-by":"publisher","first-page":"8274","DOI":"10.1016\/j.eswa.2015.06.013","volume":"42","author":"G Varol","year":"2015","unstructured":"G. Varol, A. A. Salah. Efficient large-scale action recognition in videos using extreme learning machines. Expert Systems with Applications, vol. 42, no. 21, pp. 8274\u20138282, 2015. DOI: https:\/\/doi.org\/10.1016\/j.eswa.2015.06.013.","journal-title":"Expert Systems with Applications"},{"key":"1333_CR76","doi-asserted-by":"publisher","first-page":"3131","DOI":"10.1109\/CV-PR.2016.341","volume-title":"Temporal action detection using a statistical language model","author":"A Richard","year":"2016","unstructured":"A. Richard, J. Gall. Temporal action detection using a statistical language model. In Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, IEEE, Las Vegas, USA, pp. 3131\u20133140, 2016. DOI: https:\/\/doi.org\/10.1109\/CV-PR.2016.341."},{"key":"1333_CR77","doi-asserted-by":"publisher","first-page":"1049","DOI":"10.1109\/CVPR.2016.119","volume-title":"Temporal action localization in untrimmed videos via multi-stage CNNs","author":"Z Shou","year":"2016","unstructured":"Z. Shou, D. G. Wang, S. F. Chang. Temporal action localization in untrimmed videos via multi-stage CNNs. In Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, IEEE, Las Vegas, USA, pp. 1049\u20131058, 2016. DOI: https:\/\/doi.org\/10.1109\/CVPR.2016.119."},{"key":"1333_CR78","unstructured":"H. Alwassel, F. C. Heilbron, B. Ghanem. Action search: Learning to search for human activities in untrimmed videos. [Online], Available: https:\/\/arxiv.org\/abs\/1706.04269, 2017."},{"key":"1333_CR79","doi-asserted-by":"publisher","first-page":"988","DOI":"10.1145\/3123266.3123343","volume-title":"Single shot temporal action detection","author":"T W Lin","year":"2017","unstructured":"T. W. Lin, X. Zhao, Z. Shou. Single shot temporal action detection. In Proceedings of the 25th ACM International Conference on Multimedia, ACM, Mountain View, USA, pp. 988\u2013996, 2017. DOI: https:\/\/doi.org\/10.1145\/3123266.3123343."},{"key":"1333_CR80","doi-asserted-by":"publisher","first-page":"3093","DOI":"10.1109\/CVPR.2016.337","volume-title":"Temporal action localization with pyramid of score distribution features","author":"J Yuan","year":"2016","unstructured":"J. Yuan, B. B. Ni, X. K. Yang, A. A. Kassim. Temporal action localization with pyramid of score distribution features. In Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, IEEE, Las Vegas, USA, pp. 3093\u20133102, 2016. DOI: https:\/\/doi.org\/10.1109\/CVPR.2016.337."},{"key":"1333_CR81","doi-asserted-by":"publisher","first-page":"1417","DOI":"10.1109\/CVPR.2017.155","volume-title":"CDC: Convolutional-de-convolutional networks for precise temporal action localization in untrimmed videos","author":"Z Shou","year":"2017","unstructured":"Z. Shou, J. Chan, A. Zareian, K. Miyazawa, S. F. Chang. CDC: Convolutional-de-convolutional networks for precise temporal action localization in untrimmed videos. In Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, IEEE, Honolulu, USA, pp. 1417\u20131426, 2017. DOI: https:\/\/doi.org\/10.1109\/CVPR.2017.155."},{"key":"1333_CR82","doi-asserted-by":"publisher","first-page":"5794","DOI":"10.1109\/ICCV.2017.617","volume-title":"R-C3D: Region convolutional 3D network for temporal activity detection","author":"H J Xu","year":"2017","unstructured":"H. J. Xu, A. Das, K. Saenko. R-C3D: Region convolutional 3D network for temporal activity detection. In Proceedings of IEEE International Conference on Computer Vision, IEEE, Venice, Italy, pp. 5794\u20135803, 2017. DOI: https:\/\/doi.org\/10.1109\/ICCV.2017.617."},{"key":"1333_CR83","unstructured":"H. J. Xu, B. Y. Kang, X. M. Sun, J. S. Feng, K. Saenko, T. Darrell. Similarity R-C3D for few-shot temporal activity detection. [Online], Available: https:\/\/arxiv.org\/abs\/1812.10000, 2018."},{"key":"1333_CR84","doi-asserted-by":"publisher","first-page":"2933","DOI":"10.1109\/ICCV.2017.317","volume-title":"Temporal action detection with structured segment networks","author":"Y Zhao","year":"2017","unstructured":"Y. Zhao, Y. J. Xiong, L. M. Wang, Z. R. Wu, X. O. Tang, D. H. Lin. Temporal action detection with structured segment networks. In Proceedings of IEEE International Conference on Computer Vision, IEEE, Venice, Italy, pp. 2933\u20132942, 2017. DOI: https:\/\/doi.org\/10.1109\/ICCV.2017.317."},{"key":"1333_CR85","doi-asserted-by":"publisher","first-page":"3175","DOI":"10.1109\/CVPR.2017.338","volume-title":"SCC: Semantic context cascade for efficient action detection","author":"F C Heilbron","year":"2017","unstructured":"F. C. Heilbron, W. Barrios, V. Escorcia, B. Ghanem. SCC: Semantic context cascade for efficient action detection. In Proceedings of IEEE Conference on Computer Vision and Pattern Recognition, IEEE, Honolulu, USA, pp. 3175\u20133184, 2017. DOI: https:\/\/doi.org\/10.1109\/CVPR.2017.338."},{"key":"1333_CR86","doi-asserted-by":"publisher","first-page":"1130","DOI":"10.1109\/CVPR.2018.00124","volume-title":"Rethinking the faster R-CNN architecture for temporal action localization","author":"Y W Chao","year":"2018","unstructured":"Y. W. Chao, S. Vijayanarasimhan, B. Seybold, D. A. Ross, J. Deng, R. Sukthankar. Rethinking the faster R-CNN architecture for temporal action localization. In Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition, IEEE, Salt Lake City, USA, pp. 1130\u20131139, 2018. DOI: https:\/\/doi.org\/10.1109\/CVPR.2018.00124."},{"key":"1333_CR87","unstructured":"Y. J. Xiong, Y. Zhao, L. M. Wang, D. H. Lin, X. O. Tang. A pursuit of temporal accuracy in general activity detection. [Online], Available: https:\/\/arxiv.org\/abs\/1703.02716, 2017."}],"container-title":["Machine Intelligence Research"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11633-022-1333-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11633-022-1333-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11633-022-1333-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,5,28]],"date-time":"2022-05-28T15:12:29Z","timestamp":1653750749000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11633-022-1333-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,5,28]]},"references-count":87,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2022,6]]}},"alternative-id":["1333"],"URL":"https:\/\/doi.org\/10.1007\/s11633-022-1333-4","relation":{},"ISSN":["2731-538X","2731-5398"],"issn-type":[{"value":"2731-538X","type":"print"},{"value":"2731-5398","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,5,28]]},"assertion":[{"value":"21 January 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 April 2022","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 May 2022","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}