{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,7]],"date-time":"2026-04-07T13:04:39Z","timestamp":1775567079797,"version":"3.50.1"},"reference-count":99,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2025,4]]},"DOI":"10.1007\/s11263-024-02279-1","type":"journal-article","created":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T21:22:31Z","timestamp":1730150551000},"page":"1940-1963","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["A Memory-Assisted Knowledge Transferring Framework with Curriculum Anticipation for Weakly Supervised Online Activity Detection"],"prefix":"10.1007","volume":"133","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3831-8893","authenticated-orcid":false,"given":"Tianshan","family":"Liu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kin-Man","family":"Lam","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bing-Kun","family":"Bao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"issue":"6","key":"2279_CR1","doi-asserted-by":"crossref","first-page":"1550","DOI":"10.1007\/s11263-023-01771-4","volume":"131","author":"Y Bai","year":"2023","unstructured":"Bai, Y., Zou, Q., Chen, X., Li, L., Ding, Z., & Chen, L. (2023). Extreme low-resolution action recognition with confident spatial-temporal attention transfer. International Journal of Computer Vision, 131(6), 1550\u20131565.","journal-title":"International Journal of Computer Vision"},{"key":"2279_CR2","doi-asserted-by":"crossref","unstructured":"Bengio, Y., Louradour, J., Collobert, R., & Weston, J. (2009). Curriculum learning. In Proceedings of the 26th Annual International Conference on Machine Learning, pp. 41\u201348.","DOI":"10.1145\/1553374.1553380"},{"key":"2279_CR3","doi-asserted-by":"publisher","unstructured":"Carreira, J., & Zisserman, A. (2017). Quo vadis, action recognition? a new model and the kinetics dataset. In 2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 4724\u20134733. https:\/\/doi.org\/10.1109\/CVPR.2017.502","DOI":"10.1109\/CVPR.2017.502"},{"key":"2279_CR4","unstructured":"Chen, G., Choi, W., Yu, X., Han, T., & Chandraker, M. (2017). Learning efficient object detection models with knowledge distillation. In Advances in Neural Information Processing Systems, vol. 30."},{"key":"2279_CR5","doi-asserted-by":"publisher","unstructured":"Chen, J., Mittal, G., Yu, Y., Kong, Y., & Chen, M. (2022). Gatehub: Gated history unit with background suppression for online action detection. In 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 19893\u201319902. https:\/\/doi.org\/10.1109\/CVPR52688.2022.01930","DOI":"10.1109\/CVPR52688.2022.01930"},{"key":"2279_CR6","unstructured":"Chung, J., Gulcehre, C., Cho, K., & Bengio, Y. (2014). Empirical evaluation of gated recurrent neural networks on sequence modeling. arXiv:1412.3555"},{"key":"2279_CR7","doi-asserted-by":"publisher","unstructured":"Dai, R., Das, S., & Bremond, F. (2021). Learning an augmented RGB representation with cross-modal knowledge distillation for action detection. In 2021 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 13033\u201313044. https:\/\/doi.org\/10.1109\/ICCV48922.2021.01281","DOI":"10.1109\/ICCV48922.2021.01281"},{"key":"2279_CR8","doi-asserted-by":"publisher","unstructured":"De\u00a0Geest, R., & Tuytelaars, T. (2018). Modeling temporal structure with lstm for online action detection. In 2018 IEEE Winter Conference on Applications of Computer Vision (WACV), pp. 1549\u20131557. https:\/\/doi.org\/10.1109\/WACV.2018.00173","DOI":"10.1109\/WACV.2018.00173"},{"key":"2279_CR9","doi-asserted-by":"publisher","unstructured":"Eun, H., Moon, J., Park, J., Jung, C., & Kim, C (2020). Learning to discriminate information for online action detection. In 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 806\u2013815. https:\/\/doi.org\/10.1109\/CVPR42600.2020.00089","DOI":"10.1109\/CVPR42600.2020.00089"},{"key":"2279_CR10","doi-asserted-by":"publisher","first-page":"6985","DOI":"10.1109\/TIP.2021.3101158","volume":"30","author":"Z Feng","year":"2021","unstructured":"Feng, Z., Lai, J., & Xie, X. (2021). Resolution-aware knowledge distillation for efficient inference. IEEE Transactions on Image Processing, 30, 6985\u20136996. https:\/\/doi.org\/10.1109\/TIP.2021.3101158","journal-title":"IEEE Transactions on Image Processing"},{"key":"2279_CR11","doi-asserted-by":"publisher","unstructured":"Gao, M., Xu, M., Davis, L., Socher, R., & Xiong, C. (2019). Startnet: Online detection of action start in untrimmed videos. In 2019 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 5541\u20135550. https:\/\/doi.org\/10.1109\/ICCV.2019.00564","DOI":"10.1109\/ICCV.2019.00564"},{"key":"2279_CR12","doi-asserted-by":"crossref","unstructured":"Gao, J., Yang, Z., & Nevatia, R. (2017). Red: Reinforced encoder-decoder networks for action anticipation. arXiv:1707.04818","DOI":"10.5244\/C.31.92"},{"key":"2279_CR13","doi-asserted-by":"publisher","unstructured":"Gao, M., Zhou, Y., Xu, R., Socher, R., & Xiong, C. (2021). Woad: Weakly supervised online action detection in untrimmed videos. In 2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 1915\u20131923. https:\/\/doi.org\/10.1109\/CVPR46437.2021.00195","DOI":"10.1109\/CVPR46437.2021.00195"},{"issue":"10","key":"2279_CR14","doi-asserted-by":"publisher","first-page":"2581","DOI":"10.1109\/TPAMI.2019.2929038","volume":"42","author":"NC Garcia","year":"2020","unstructured":"Garcia, N. C., Morerio, P., & Murino, V. (2020). Learning with privileged information via adversarial discriminative modality distillation. IEEE Transactions on Pattern Analysis and Machine Intelligence, 42(10), 2581\u20132593. https:\/\/doi.org\/10.1109\/TPAMI.2019.2929038","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2279_CR15","doi-asserted-by":"crossref","unstructured":"Geest, R.D., Gavves, E., Ghodrati, A., Li, Z., Snoek, C., & Tuytelaars, T. (2016). Online action detection. In European Conference on Computer Vision (ECCV), pp. 269\u2013284.","DOI":"10.1007\/978-3-319-46454-1_17"},{"issue":"4","key":"2279_CR16","doi-asserted-by":"publisher","first-page":"2051","DOI":"10.1109\/TIP.2018.2883743","volume":"28","author":"S Ge","year":"2019","unstructured":"Ge, S., Zhao, S., Li, C., & Li, J. (2019). Low-resolution face recognition in the wild via selective knowledge distillation. IEEE Transactions on Image Processing, 28(4), 2051\u20132062. https:\/\/doi.org\/10.1109\/TIP.2018.2883743","journal-title":"IEEE Transactions on Image Processing"},{"key":"2279_CR17","doi-asserted-by":"publisher","unstructured":"Gong, D., Liu, L., Le, V., Saha, B., Mansour, M.R., Venkatesh, S., & Van Den\u00a0Hengel, A. (2019). Memorizing normality to detect anomaly: Memory-augmented deep autoencoder for unsupervised anomaly detection. In 2019 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 1705\u20131714. https:\/\/doi.org\/10.1109\/ICCV.2019.00179","DOI":"10.1109\/ICCV.2019.00179"},{"issue":"7","key":"2279_CR18","doi-asserted-by":"publisher","first-page":"3249","DOI":"10.1109\/TIP.2016.2563981","volume":"25","author":"C Gong","year":"2016","unstructured":"Gong, C., Tao, D., Maybank, S. J., Liu, W., Kang, G., & Yang, J. (2016). Multi-modal curriculum learning for semi-supervised image classification. IEEE Transactions on Image Processing, 25(7), 3249\u20133260. https:\/\/doi.org\/10.1109\/TIP.2016.2563981","journal-title":"IEEE Transactions on Image Processing"},{"issue":"5","key":"2279_CR19","doi-asserted-by":"publisher","first-page":"7099","DOI":"10.1109\/TII.2022.3209672","volume":"19","author":"J Gou","year":"2023","unstructured":"Gou, J., Sun, L., Yu, B., Wan, S., Ou, W., & Yi, Z. (2023). Multilevel attention-based sample correlations for knowledge distillation. IEEE Transactions on Industrial Informatics, 19(5), 7099\u20137109. https:\/\/doi.org\/10.1109\/TII.2022.3209672","journal-title":"IEEE Transactions on Industrial Informatics"},{"key":"2279_CR20","first-page":"1","volume":"2023","author":"J Gou","year":"2023","unstructured":"Gou, J., Xiong, X., Yu, B., Du, L., Zhan, Y., & Tao, D. (2023). Multi-target knowledge distillation via student self-reflection. International Journal of Computer Vision, 2023, 1\u201318.","journal-title":"International Journal of Computer Vision"},{"issue":"6","key":"2279_CR21","doi-asserted-by":"crossref","first-page":"1789","DOI":"10.1007\/s11263-021-01453-z","volume":"129","author":"J Gou","year":"2021","unstructured":"Gou, J., Yu, B., Maybank, S. J., & Tao, D. (2021). Knowledge distillation: A survey. International Journal of Computer Vision, 129(6), 1789\u20131819.","journal-title":"International Journal of Computer Vision"},{"key":"2279_CR22","doi-asserted-by":"crossref","unstructured":"Grauman, K., Westbury, A., Byrne, E., Chavis, Z., Furnari, A., Girdhar, R., Hamburger, J., Jiang, H., Liu, M., & Liu, X., et\u00a0al. (2022). Ego4d: Around the world in 3,000 hours of egocentric video. In 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 18973\u201318990.","DOI":"10.1109\/CVPR52688.2022.01842"},{"key":"2279_CR23","unstructured":"Graves, A., Bellemare, M.G., Menick, J., Munos, R., & Kavukcuoglu, K. (2017). Automated curriculum learning for neural networks. In International Conference on Machine Learning (ICML), pp. 1311\u20131320."},{"key":"2279_CR24","doi-asserted-by":"crossref","unstructured":"Guo, H., Ren, Z., Wu, Y., Hua, G., & Ji, Q. (2022). Uncertainty-based spatial-temporal attention for online action detection. In European Conference on Computer Vision (ECCV), pp. 69\u201386.","DOI":"10.1007\/978-3-031-19772-7_5"},{"key":"2279_CR25","unstructured":"Hacohen, G., & Weinshall, D. (2019). On the power of curriculum learning in training deep networks. In International Conference on Machine Learning (ICML), pp. 2535\u20132544."},{"key":"2279_CR26","doi-asserted-by":"crossref","unstructured":"Han, T., Xie, W., & Zisserman, A. (2020). Memory-augmented dense predictive coding for video representation learning. In European Conference on Computer Vision (ECCV), pp. 312\u2013329.","DOI":"10.1007\/978-3-030-58580-8_19"},{"key":"2279_CR27","doi-asserted-by":"publisher","unstructured":"He, B., Yang, X., Kang, L., Cheng, Z., Zhou, X., & Shrivastava, A.: ASM-Loc: Action-aware segment modeling for weakly-supervised temporal action localization. In 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 13915\u201313925 (2022). https:\/\/doi.org\/10.1109\/CVPR52688.2022.01355","DOI":"10.1109\/CVPR52688.2022.01355"},{"key":"2279_CR28","doi-asserted-by":"publisher","unstructured":"Heilbron, F.C., Escorcia, V., Ghanem, B., & Niebles, J.C. (2015). Activitynet: A large-scale video benchmark for human activity understanding. In 2015 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 961\u2013970. https:\/\/doi.org\/10.1109\/CVPR.2015.7298698","DOI":"10.1109\/CVPR.2015.7298698"},{"key":"2279_CR29","unstructured":"Hinton, G., Vinyals, O., & Dean, J. (2015). Distilling the knowledge in a neural network. arXiv:1503.02531"},{"issue":"8","key":"2279_CR30","doi-asserted-by":"crossref","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter, S., & Schmidhuber, J. (1997). Long short-term memory. Neural Computation, 9(8), 1735\u20131780.","journal-title":"Neural Computation"},{"key":"2279_CR31","doi-asserted-by":"publisher","unstructured":"Hoffman, J., Gupta, S., & Darrell, T. (2016). Learning with side information through modality hallucination. In 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 826\u2013834. https:\/\/doi.org\/10.1109\/CVPR.2016.96","DOI":"10.1109\/CVPR.2016.96"},{"key":"2279_CR32","doi-asserted-by":"crossref","unstructured":"Huang, L., Huang, Y., Ouyang, W., & Wang, L. (2020). Relational prototypical network for weakly supervised temporal action localization. In Proceedings of the AAAI Conference on Artificial Intelligence, vol. 34, pp. 11053\u201311060.","DOI":"10.1609\/aaai.v34i07.6760"},{"key":"2279_CR33","doi-asserted-by":"publisher","unstructured":"Huang, L., Wang, L., & Li, H. (2021). Foreground-action consistency network for weakly supervised temporal action localization. In 2021 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 7982\u20137991. https:\/\/doi.org\/10.1109\/ICCV48922.2021.00790","DOI":"10.1109\/ICCV48922.2021.00790"},{"key":"2279_CR34","doi-asserted-by":"publisher","unstructured":"Huang, L., Wang, L., & Li, H. (2022). Weakly supervised temporal action localization via representative snippet knowledge propagation. In 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3262\u20133271. https:\/\/doi.org\/10.1109\/CVPR52688.2022.00327","DOI":"10.1109\/CVPR52688.2022.00327"},{"key":"2279_CR35","doi-asserted-by":"publisher","first-page":"5154","DOI":"10.1109\/TIP.2021.3078324","volume":"30","author":"L Huang","year":"2021","unstructured":"Huang, L., Huang, Y., Ouyang, W., & Wang, L. (2021). Modeling sub-actions for weakly supervised temporal action localization. IEEE Transactions on Image Processing, 30, 5154\u20135167. https:\/\/doi.org\/10.1109\/TIP.2021.3078324","journal-title":"IEEE Transactions on Image Processing"},{"issue":"9","key":"2279_CR36","doi-asserted-by":"publisher","first-page":"5729","DOI":"10.1109\/TPAMI.2021.3076172","volume":"44","author":"L Huang","year":"2022","unstructured":"Huang, L., Huang, Y., Ouyang, W., & Wang, L. (2022). Two-branch relational prototypical network for weakly supervised temporal action localization. IEEE Transactions on Pattern Analysis and Machine Intelligence, 44(9), 5729\u20135746. https:\/\/doi.org\/10.1109\/TPAMI.2021.3076172","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2279_CR37","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1016\/j.cviu.2016.10.018","volume":"155","author":"H Idrees","year":"2017","unstructured":"Idrees, H., Zamir, A. R., Jiang, Y.-G., Gorban, A., Laptev, I., Sukthankar, R., & Shah, M. (2017). The thumos challenge on action recognition for videos \u201cin the wild\u2019\u2019. Computer Vision and Image Understanding, 155, 1\u201323.","journal-title":"Computer Vision and Image Understanding"},{"key":"2279_CR38","doi-asserted-by":"crossref","unstructured":"Islam, A., Long, C., & Radke, R. (2021). A hybrid attention mechanism for weakly-supervised temporal action localization. In Proceedings of the AAAI Conference on Artificial Intelligence, vol. 35, pp. 1637\u20131645.","DOI":"10.1609\/aaai.v35i2.16256"},{"key":"2279_CR39","doi-asserted-by":"publisher","unstructured":"Ji, M., Shin, S., Hwang, S., Park, G., & Moon, I.-C. (2021). Refine myself by teaching myself: Feature refinement via self-knowledge distillation. In 2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 10659\u201310668. https:\/\/doi.org\/10.1109\/CVPR46437.2021.01052","DOI":"10.1109\/CVPR46437.2021.01052"},{"key":"2279_CR40","unstructured":"Kay, W., Carreira, J., Simonyan, K., Zhang, B., Hillier, C., Vijayanarasimhan, S., Viola, F., Green, T., Back, T., & Natsev, P., et al. (2017). The kinetics human action video dataset. arXiv:1705.06950"},{"key":"2279_CR41","unstructured":"Kingma, D.P., & Ba, J. (2015). Adam: A method for stochastic optimization. In International Conference on Learning Representations (ICLR), pp. 1\u201311."},{"issue":"5","key":"2279_CR42","doi-asserted-by":"crossref","first-page":"1366","DOI":"10.1007\/s11263-022-01594-9","volume":"130","author":"Y Kong","year":"2022","unstructured":"Kong, Y., & Fu, Y. (2022). Human action recognition and prediction: A survey. International Journal of Computer Vision, 130(5), 1366\u20131401.","journal-title":"International Journal of Computer Vision"},{"key":"2279_CR43","unstructured":"Kumar, M., Packer, B., & Koller, D. (2010). Self-paced learning for latent variable models. In Advances in Neural Information Processing Systems, vol. 23."},{"key":"2279_CR44","doi-asserted-by":"publisher","unstructured":"Lee, S., Kim, H.G., Hwi\u00a0Choi, D., Kim, H.-I., & Ro, Y.M. (2021). Video prediction recalling long-term motion context via memory alignment learning. In 2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3053\u20133062. https:\/\/doi.org\/10.1109\/CVPR46437.2021.00307","DOI":"10.1109\/CVPR46437.2021.00307"},{"key":"2279_CR45","doi-asserted-by":"crossref","unstructured":"Lee, S., Park, S., & Ro, Y.M. (2022). Audio-visual mismatch-aware video retrieval via association and adjustment. In European Conference on Computer Vision (ECCV), pp. 497\u2013514.","DOI":"10.1007\/978-3-031-19781-9_29"},{"key":"2279_CR46","doi-asserted-by":"crossref","unstructured":"Lee, P., Uh, Y., & Byun, H. (2020). Background suppression network for weakly-supervised temporal action localization. In Proceedings of the AAAI Conference on Artificial Intelligence, vol. 34, pp. 11320\u201311327.","DOI":"10.1609\/aaai.v34i07.6793"},{"key":"2279_CR47","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1109\/TPAMI.2022.3204808","volume":"2022","author":"S Lee","year":"2022","unstructured":"Lee, S., Eun, H., Moon, J., Choi, S., Kim, Y., Jung, C., & Kim, C. (2022). Learning to discriminate information for online action detection: Analysis and application. IEEE Transactions on Pattern Analysis and Machine Intelligence, 2022, 1\u201317. https:\/\/doi.org\/10.1109\/TPAMI.2022.3204808","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2279_CR48","doi-asserted-by":"publisher","unstructured":"Li, J., Yang, T., Ji, W., Wang, J., & Cheng, L. (2022). Exploring denoised cross-video contrast for weakly-supervised temporal action localization. In 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 19882\u201319892. https:\/\/doi.org\/10.1109\/CVPR52688.2022.01929","DOI":"10.1109\/CVPR52688.2022.01929"},{"key":"2279_CR49","doi-asserted-by":"crossref","unstructured":"Li, S., Zhu, X., Huang, Q., Xu, H., & Kuo, C.-C.J. (2017). Multiple instance curriculum learning for weakly supervised object detection. In British Machine Vision Conference (BMVC).","DOI":"10.5244\/C.31.29"},{"key":"2279_CR50","doi-asserted-by":"publisher","unstructured":"Liu, T., & Lam, K.-M. (2022). A hybrid egocentric activity anticipation framework via memory-augmented recurrent and one-shot representation forecasting. In 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 13894\u201313903. https:\/\/doi.org\/10.1109\/CVPR52688.2022.01353","DOI":"10.1109\/CVPR52688.2022.01353"},{"key":"2279_CR51","doi-asserted-by":"publisher","unstructured":"Liu, D., Jiang, T., & Wang, Y. (2019). Completeness modeling and context separation for weakly supervised temporal action localization. In 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 1298\u20131307. https:\/\/doi.org\/10.1109\/CVPR.2019.00139","DOI":"10.1109\/CVPR.2019.00139"},{"key":"2279_CR52","doi-asserted-by":"publisher","unstructured":"Liu, Z., Nie, Y., Long, C., Zhang, Q., & Li, G. (2021). A hybrid video anomaly detection framework via memory-augmented flow reconstruction and flow-guided frame prediction. In 2021 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 13568\u201313577. https:\/\/doi.org\/10.1109\/ICCV48922.2021.01333","DOI":"10.1109\/ICCV48922.2021.01333"},{"key":"2279_CR53","doi-asserted-by":"crossref","unstructured":"Liu, Z., Wang, L., Zhang, Q., Tang, W., Yuan, J., Zheng, N., & Hua, G. (2021). Acsnet: Action-context separation network for weakly supervised temporal action localization. In Proceedings of the AAAI Conference on Artificial Intelligence, vol. 35, pp. 2233\u20132241.","DOI":"10.1609\/aaai.v35i3.16322"},{"issue":"1","key":"2279_CR54","doi-asserted-by":"publisher","first-page":"315","DOI":"10.1109\/TCSVT.2021.3060162","volume":"32","author":"T Liu","year":"2022","unstructured":"Liu, T., Lam, K.-M., Zhao, R., & Qiu, G. (2022). Deep cross-modal representation learning and distillation for illumination-invariant pedestrian detection. IEEE Transactions on Circuits and Systems for Video Technology, 32(1), 315\u2013329. https:\/\/doi.org\/10.1109\/TCSVT.2021.3060162","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"2279_CR55","unstructured":"Lopez-Paz, D., Bottou, L., Sch\u00f6lkopf, B., & Vapnik, V. (2016). Unifying distillation and privileged information. In International Conference on Learning Representations (ICLR), pp. 1\u201310"},{"key":"2279_CR56","doi-asserted-by":"crossref","unstructured":"Luo, Z., Guillory, D., Shi, B., Ke, W., Wan, F., Darrell, T., & Xu, H. (2020). Weakly-supervised action localization with expectation-maximization multi-instance learning. In Proceedings of the European Conference on Computer Vision (ECCV), pp. 729\u2013745.","DOI":"10.1007\/978-3-030-58526-6_43"},{"key":"2279_CR57","doi-asserted-by":"crossref","unstructured":"Luo, Z., Hsieh, J.-T., Jiang, L., Niebles, J.C., & Fei-Fei, L. (2018). Graph distillation for action detection with privileged modalities. In Proceedings of the European Conference on Computer Vision (ECCV), pp. 166\u2013183.","DOI":"10.1007\/978-3-030-01264-9_11"},{"key":"2279_CR58","doi-asserted-by":"publisher","unstructured":"Luo, W., Zhang, T., Yang, W., Liu, J., Mei, T., Wu, F., & Zhang, Y. (2021). Action unit memory network for weakly supervised temporal action localization. In 2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 9964\u20139974. https:\/\/doi.org\/10.1109\/CVPR46437.2021.00984","DOI":"10.1109\/CVPR46437.2021.00984"},{"issue":"11","key":"2279_CR59","first-page":"2579","volume":"9","author":"L Maaten","year":"2008","unstructured":"Maaten, L., & Hinton, G. (2008). Visualizing data using t-sne. Journal of Machine Learning Research, 9(11), 2579\u20132605.","journal-title":"Journal of Machine Learning Research"},{"key":"2279_CR60","first-page":"1","volume":"2023","author":"J Mao","year":"2023","unstructured":"Mao, J., Shi, S., Wang, X., & Li, H. (2023). 3d object detection for autonomous driving: A comprehensive survey. International Journal of Computer Vision, 2023, 1\u201355.","journal-title":"International Journal of Computer Vision"},{"issue":"6","key":"2279_CR61","doi-asserted-by":"publisher","first-page":"6688","DOI":"10.1109\/TPAMI.2020.3008558","volume":"45","author":"F Marchetti","year":"2020","unstructured":"Marchetti, F., Becattini, F., Seidenari, L., & Del Bimbo, A. (2020). Multiple trajectory prediction of moving agents with memory augmented networks. IEEE Transactions on Pattern Analysis and Machine Intelligence, 45(6), 6688\u20136702. https:\/\/doi.org\/10.1109\/TPAMI.2020.3008558","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2279_CR62","doi-asserted-by":"crossref","unstructured":"Min, K., & Corso, J.J. (2020). Adversarial background-aware loss for weakly-supervised temporal activity localization. In Proceedings of the European Conference on Computer Vision (ECCV), pp. 283\u2013299.","DOI":"10.1007\/978-3-030-58568-6_17"},{"key":"2279_CR63","doi-asserted-by":"publisher","unstructured":"Narayan, S., Cholakkal, H., Hayat, M., Khan, F.S., Yang, M.-H., & Shao, L. (2021). D2-net: Weakly-supervised action localization via discriminative embeddings and denoised activations. In 2021 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 13588\u201313597. https:\/\/doi.org\/10.1109\/ICCV48922.2021.01335","DOI":"10.1109\/ICCV48922.2021.01335"},{"key":"2279_CR64","doi-asserted-by":"publisher","unstructured":"Narayan, S., Cholakkal, H., Khan, F.S., & Shao, L. (2019). 3c-net: Category count and center loss for weakly-supervised action localization. In 2019 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 8678\u20138686. https:\/\/doi.org\/10.1109\/ICCV.2019.00877","DOI":"10.1109\/ICCV.2019.00877"},{"key":"2279_CR65","unstructured":"Oord, A.v.d., Dieleman, S., Zen, H., Simonyan, K., Vinyals, O., Graves, A., Kalchbrenner, N., Senior, A., & Kavukcuoglu, K. (2016). Wavenet: A generative model for raw audio. arXiv:1609.03499"},{"key":"2279_CR66","doi-asserted-by":"publisher","unstructured":"Pardo, A., Alwassel, H., Heilbron, F.C., Thabet, A., & Ghanem, B. (2021). Refineloc: Iterative refinement for weakly-supervised action localization. In 2021 IEEE Winter Conference on Applications of Computer Vision (WACV), pp. 3318\u20133327. https:\/\/doi.org\/10.1109\/WACV48630.2021.00336","DOI":"10.1109\/WACV48630.2021.00336"},{"key":"2279_CR67","doi-asserted-by":"crossref","unstructured":"Paul, S., Roy, S., & Roy-Chowdhury, A.K. (2018). W-talc: Weakly-supervised temporal activity localization and classification. In Proceedings of the European Conference on Computer Vision (ECCV), pp. 563\u2013579.","DOI":"10.1007\/978-3-030-01225-0_35"},{"key":"2279_CR68","doi-asserted-by":"publisher","unstructured":"Shi, Z., & Kim, T.-K. (2017). Learning and refining of privileged information-based RNNs for action recognition from depth sequences. In 2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 4684\u20134693. https:\/\/doi.org\/10.1109\/CVPR.2017.498","DOI":"10.1109\/CVPR.2017.498"},{"key":"2279_CR69","doi-asserted-by":"crossref","unstructured":"Shou, Z., Gao, H., Zhang, L., Miyazawa, K., & Chang, S.-F. (2018). Autoloc: Weakly-supervised temporal action localization in untrimmed videos. In Proceedings of the European Conference on Computer Vision (ECCV), pp. 154\u2013171.","DOI":"10.1007\/978-3-030-01270-0_10"},{"key":"2279_CR70","doi-asserted-by":"crossref","unstructured":"Shou, Z., Pan, J., Chan, J., Miyazawa, K., Mansour, H., Vetro, A., Nieto, X.G., & Chang, S.-F. (2018). Online action detection in untrimmed, streaming videos-modeling and evaluation. In European Conference on Computer Vision (ECCV).","DOI":"10.1007\/978-3-030-01219-9_33"},{"key":"2279_CR71","doi-asserted-by":"publisher","unstructured":"Shu, T., Xie, D., Rothrock, B., Todorovic, S., & Zhu, S.-C. (2015). Joint inference of groups, events and human roles in aerial videos. In 2015 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 4576\u20134584. https:\/\/doi.org\/10.1109\/CVPR.2015.7299088","DOI":"10.1109\/CVPR.2015.7299088"},{"key":"2279_CR72","doi-asserted-by":"publisher","unstructured":"Singh, K.K., & Lee, Y.J. (2017). Hide-and-seek: Forcing a network to be meticulous for weakly-supervised object and action localization. In 2017 IEEE International Conference on Computer Vision (ICCV), pp. 3544\u20133553. https:\/\/doi.org\/10.1109\/ICCV.2017.381","DOI":"10.1109\/ICCV.2017.381"},{"issue":"6","key":"2279_CR73","doi-asserted-by":"crossref","first-page":"1526","DOI":"10.1007\/s11263-022-01611-x","volume":"130","author":"P Soviany","year":"2022","unstructured":"Soviany, P., Ionescu, R. T., Rota, P., & Sebe, N. (2022). Curriculum learning: A survey. International Journal of Computer Vision, 130(6), 1526\u20131565.","journal-title":"International Journal of Computer Vision"},{"key":"2279_CR74","first-page":"10078","volume":"35","author":"Z Tong","year":"2022","unstructured":"Tong, Z., Song, Y., Wang, J., & Wang, L. (2022). Videomae: Masked autoencoders are data-efficient learners for self-supervised video pre-training. Advances in Neural Information Processing Systems, 35, 10078\u201310093.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2279_CR75","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, \u0141., & Polosukhin, I. (2017). Attention is all you need. In Advances in Neural Information Processing Systems, pp. 5998\u20136008."},{"key":"2279_CR76","doi-asserted-by":"publisher","unstructured":"Wang, Y., Gan, W., Yang, J., Wu, W., & Yan, J. (2019). Dynamic curriculum learning for imbalanced data classification. In 2019 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 5016\u20135025. https:\/\/doi.org\/10.1109\/ICCV.2019.00512","DOI":"10.1109\/ICCV.2019.00512"},{"key":"2279_CR77","doi-asserted-by":"publisher","unstructured":"Wang, X., Hu, J.-F., Lai, J.-H., Zhang, J., Zheng, & W.-S. (2019). Progressive teacher-student learning for early action prediction. In 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3551\u20133560. https:\/\/doi.org\/10.1109\/CVPR.2019.00367","DOI":"10.1109\/CVPR.2019.00367"},{"key":"2279_CR78","doi-asserted-by":"publisher","unstructured":"Wang, L., Xiong, Y., Lin, D., & Van\u00a0Gool, L. (2017). Untrimmednets for weakly supervised action recognition and detection. In 2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 6402\u20136411. https:\/\/doi.org\/10.1109\/CVPR.2017.678","DOI":"10.1109\/CVPR.2017.678"},{"key":"2279_CR79","doi-asserted-by":"publisher","unstructured":"Wang, T., Yuan, L., Zhang, X., & Feng, J. (2019). Distilling object detectors with fine-grained feature imitation. In 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 4928\u20134937. https:\/\/doi.org\/10.1109\/CVPR.2019.00507","DOI":"10.1109\/CVPR.2019.00507"},{"key":"2279_CR80","doi-asserted-by":"publisher","unstructured":"Wang, X., Zhang, S., Qing, Z., Shao, Y., Zuo, Z., Gao, C., & Sang, N. (2021). Oadtr: Online action detection with transformers. In 2021 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 7545\u20137555 (2021). https:\/\/doi.org\/10.1109\/ICCV48922.2021.00747","DOI":"10.1109\/ICCV48922.2021.00747"},{"key":"2279_CR81","unstructured":"Weinshall, D., Cohen, G., & Amir, D. (2018). Curriculum learning by transfer learning: Theory and experiments with deep networks. In International Conference on Machine Learning (ICML), pp. 5238\u20135246."},{"key":"2279_CR82","unstructured":"Weston, J., Chopra, S., & Bordes, A. (2014). Memory networks. arXiv:1410.3916"},{"key":"2279_CR83","doi-asserted-by":"crossref","unstructured":"Wu, C.-Y., Feichtenhofer, C., Fan, H., He, K., Krahenbuhl, P., & Girshick, R. (2019). Long-term feature banks for detailed video understanding. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 284\u2013293.","DOI":"10.1109\/CVPR.2019.00037"},{"key":"2279_CR84","doi-asserted-by":"publisher","first-page":"3513","DOI":"10.1109\/TIP.2021.3062192","volume":"30","author":"P Wu","year":"2021","unstructured":"Wu, P., & Liu, J. (2021). Learning causal temporal relation and feature discrimination for anomaly detection. IEEE Transactions on Image Processing, 30, 3513\u20133527. https:\/\/doi.org\/10.1109\/TIP.2021.3062192","journal-title":"IEEE Transactions on Image Processing"},{"key":"2279_CR85","doi-asserted-by":"publisher","unstructured":"Xu, M., Gao, M., Chen, Y.-T., Davis, L., & Crandall, D. (2019). Temporal recurrent networks for online action detection. In 2019 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 5531\u20135540. https:\/\/doi.org\/10.1109\/ICCV.2019.00563","DOI":"10.1109\/ICCV.2019.00563"},{"key":"2279_CR86","doi-asserted-by":"publisher","unstructured":"Xu, X., Li, Y.-L., & Lu, C. (2022). Learning to anticipate future with dynamic context removal. In 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 12724\u201312734. https:\/\/doi.org\/10.1109\/CVPR52688.2022.01240","DOI":"10.1109\/CVPR52688.2022.01240"},{"key":"2279_CR87","doi-asserted-by":"publisher","unstructured":"Xu, C., Mao, W., Zhang, W., & Chen, S. (2022). Remember intentions: Retrospective-memory-based trajectory prediction. In 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 6478\u20136487. https:\/\/doi.org\/10.1109\/CVPR52688.2022.00638","DOI":"10.1109\/CVPR52688.2022.00638"},{"key":"2279_CR88","unstructured":"Xu, M., Xiong, Y., Chen, H., Li, X., Xia, W., Tu, Z., & Soatto, S. (2021). Long short-term transformer for online action detection. In Advances in Neural Information Processing Systems, vol. 34."},{"key":"2279_CR89","doi-asserted-by":"publisher","unstructured":"Yang, L., Han, J., & Zhang, D. (2022). Colar: Effective and efficient online action detection by consulting exemplars. In 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3150\u20133159. https:\/\/doi.org\/10.1109\/CVPR52688.2022.00316","DOI":"10.1109\/CVPR52688.2022.00316"},{"key":"2279_CR90","doi-asserted-by":"publisher","unstructured":"Yang, W., Zhang, T., Yu, X., Qi, T., Zhang, Y., & Feng, W. (2021). Uncertainty guided collaborative training for weakly supervised temporal action detection. In 2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 53\u201363. https:\/\/doi.org\/10.1109\/CVPR46437.2021.00012","DOI":"10.1109\/CVPR46437.2021.00012"},{"issue":"10","key":"2279_CR91","doi-asserted-by":"crossref","first-page":"2349","DOI":"10.1007\/s11263-022-01649-x","volume":"130","author":"N Yudistira","year":"2022","unstructured":"Yudistira, N., Kavitha, M. S., & Kurita, T. (2022). Weakly-supervised action localization, and action recognition using global-local attention of 3d CNN. International Journal of Computer Vision, 130(10), 2349\u20132363.","journal-title":"International Journal of Computer Vision"},{"issue":"4","key":"2279_CR92","doi-asserted-by":"publisher","first-page":"5656","DOI":"10.1109\/TNNLS.2022.3208605","volume":"35","author":"H Yu","year":"2022","unstructured":"Yu, H., Zhu, P., Zhang, K., Wang, Y., Zhao, S., Wang, L., Zhang, T., & Hu, Q. (2022). Learning dynamic compact memory embedding for deformable visual object tracking. IEEE Transactions on Neural Networks and Learning Systems, 35(4), 5656\u20135670. https:\/\/doi.org\/10.1109\/TNNLS.2022.3208605","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"2279_CR93","unstructured":"Zhang, Y., Abbeel, P., & Pinto, L. (2020). Automatic curriculum learning through value disagreement. In Advances in Neural Information Processing Systems, vol. 33, pp. 7648\u20137659."},{"key":"2279_CR94","doi-asserted-by":"publisher","unstructured":"Zhang, L., Song, J., Gao, A., Chen, J., Bao, C., & Ma, K. (2019). Be your own teacher: Improve the performance of convolutional neural networks via self distillation. In 2019 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 3712\u20133721. https:\/\/doi.org\/10.1109\/ICCV.2019.00381","DOI":"10.1109\/ICCV.2019.00381"},{"key":"2279_CR95","doi-asserted-by":"crossref","unstructured":"Zhang, C., Xu, Y., Cheng, Z., Niu, Y., Pu, S., Wu, F., & Zou, F. (2019). Adversarial seeded sequence growing for weakly-supervised temporal action localization. In Proceedings of the 27th ACM International Conference on Multimedia, pp. 738\u2013746.","DOI":"10.1145\/3343031.3351044"},{"key":"2279_CR96","doi-asserted-by":"crossref","unstructured":"Zhao, Y., & Kr\u00e4henb\u00fchl, P. (2022). Real-time online video detection with temporal smoothing transformers. In European Conference on Computer Vision (ECCV), pp. 485\u2013502.","DOI":"10.1007\/978-3-031-19830-4_28"},{"issue":"8","key":"2279_CR97","doi-asserted-by":"crossref","first-page":"2474","DOI":"10.1007\/s11263-021-01473-9","volume":"129","author":"T Zhao","year":"2021","unstructured":"Zhao, T., Han, J., Yang, L., Wang, B., & Zhang, D. (2021). Soda: Weakly supervised temporal action localization based on astute background response and self-distillation learning. International Journal of Computer Vision, 129(8), 2474\u20132498.","journal-title":"International Journal of Computer Vision"},{"key":"2279_CR98","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2022.108741","volume":"129","author":"P Zhao","year":"2022","unstructured":"Zhao, P., Xie, L., Wang, J., Zhang, Y., & Tian, Q. (2022). Progressive privileged knowledge distillation for online action detection. Pattern Recognition, 129, 108741.","journal-title":"Pattern Recognition"},{"key":"2279_CR99","doi-asserted-by":"crossref","unstructured":"Zhong, J.-X., Li, N., Kong, W., Zhang, T., Li, T.H., & Li, G. (2018). Step-by-step erasion, one-by-one collection: a weakly supervised temporal action detector. In Proceedings of the 26th ACM International Conference on Multimedia, pp. 35\u201344.","DOI":"10.1145\/3240508.3240511"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-024-02279-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-024-02279-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-024-02279-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,3,31]],"date-time":"2025-03-31T09:13:32Z","timestamp":1743412412000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-024-02279-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":99,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2025,4]]}},"alternative-id":["2279"],"URL":"https:\/\/doi.org\/10.1007\/s11263-024-02279-1","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"19 July 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"10 October 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 October 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}