{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,18]],"date-time":"2026-01-18T00:59:52Z","timestamp":1768697992923,"version":"3.49.0"},"publisher-location":"New York, NY, USA","reference-count":46,"publisher":"ACM","license":[{"start":{"date-parts":[[2021,10,17]],"date-time":"2021-10-17T00:00:00Z","timestamp":1634428800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2021,10,17]]},"DOI":"10.1145\/3474085.3475197","type":"proceedings-article","created":{"date-parts":[[2021,10,18]],"date-time":"2021-10-18T10:23:20Z","timestamp":1634552600000},"page":"487-495","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":5,"title":["Annotation-Efficient Untrimmed Video Action Recognition"],"prefix":"10.1145","author":[{"given":"Yixiong","family":"Zou","sequence":"first","affiliation":[{"name":"Peking University &amp; Carnegie Mellon University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shanghang","family":"Zhang","sequence":"additional","affiliation":[{"name":"University of California, Berkeley, Berkeley, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guangyao","family":"Chen","sequence":"additional","affiliation":[{"name":"Peking University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yonghong","family":"Tian","sequence":"additional","affiliation":[{"name":"Peking University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kurt","family":"Keutzer","sequence":"additional","affiliation":[{"name":"University of California, Berkeley, Berkeley, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jos\u00e9 M. F.","family":"Moura","sequence":"additional","affiliation":[{"name":"Carnegie Mellon University, Pittsburgh, PA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2021,10,17]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.5555\/3026877.3026899"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.173"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01063"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"key":"e_1_3_2_2_5_1","volume-title":"Learning Open Set Network with Discriminative Reciprocal Points. ECCV","author":"Chen Guangyao","year":"2020","unstructured":"Guangyao Chen , Limeng Qiao , Yemin Shi , Peixi Peng , Jia Li , Tiejun Huang , Shiliang Pu , and Yonghong Tian . 2020. Learning Open Set Network with Discriminative Reciprocal Points. ECCV ( 2020 ). Guangyao Chen, Limeng Qiao, Yemin Shi, Peixi Peng, Jia Li, Tiejun Huang, Shiliang Pu, and Yonghong Tian. 2020. Learning Open Set Network with Discriminative Reciprocal Points. ECCV (2020)."},{"key":"e_1_3_2_2_6_1","volume-title":"Yu-Chiang Frank Wang, and Jia-Bin Huang","author":"Chen Wei-Yu","year":"2019","unstructured":"Wei-Yu Chen , Yen-Cheng Liu , Zsolt Kira , Yu-Chiang Frank Wang, and Jia-Bin Huang . 2019 . A Closer Look at Few-shot Classification. CoRR , Vol. abs\/ 1904 .04232 (2019). arxiv: 1904.04232 http:\/\/arxiv.org\/abs\/1904.04232 Wei-Yu Chen, Yen-Cheng Liu, Zsolt Kira, Yu-Chiang Frank Wang, and Jia-Bin Huang. 2019. A Closer Look at Few-shot Classification. CoRR, Vol. abs\/1904.04232 (2019). arxiv: 1904.04232 http:\/\/arxiv.org\/abs\/1904.04232"},{"key":"e_1_3_2_2_7_1","volume-title":"Imagenet: A large-scale hierarchical image database. In CVPR. Ieee, 248--255.","author":"Deng Jia","year":"2009","unstructured":"Jia Deng , Wei Dong , Richard Socher , Li-Jia Li , Kai Li , and Li Fei-Fei . 2009 . Imagenet: A large-scale hierarchical image database. In CVPR. Ieee, 248--255. Jia Deng, Wei Dong, Richard Socher, Li-Jia Li, Kai Li, and Li Fei-Fei. 2009. Imagenet: A large-scale hierarchical image database. In CVPR. Ieee, 248--255."},{"key":"e_1_3_2_2_8_1","unstructured":"Akshay Raj Dhamija Manuel G\u00fcnther and Terrance Boult. 2018. Reducing network agnostophobia. In Advances in Neural Information Processing Systems. 9157--9168.  Akshay Raj Dhamija Manuel G\u00fcnther and Terrance Boult. 2018. Reducing network agnostophobia. In Advances in Neural Information Processing Systems. 9157--9168."},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1016\/S0004-3702(96)00034-3"},{"key":"e_1_3_2_2_10_1","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. 961--970","author":"Fabian Caba Heilbron Bernard Ghanem","year":"2015","unstructured":"Bernard Ghanem Fabian Caba Heilbron , Victor Escorcia and Juan Carlos Niebles . 2015 . ActivityNet: A Large-Scale Video Benchmark for Human Activity Understanding . In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. 961--970 . Bernard Ghanem Fabian Caba Heilbron, Victor Escorcia and Juan Carlos Niebles. 2015. ActivityNet: A Large-Scale Video Benchmark for Human Activity Understanding. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. 961--970."},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"crossref","unstructured":"Bharath Hariharan and Ross Girshick. 2017. Low-shot visual recognition by shrinking and hallucinating features. In ICCV. 3018--3027.  Bharath Hariharan and Ross Girshick. 2017. Low-shot visual recognition by shrinking and hallucinating features. In ICCV. 3018--3027.","DOI":"10.1109\/ICCV.2017.328"},{"key":"e_1_3_2_2_12_1","unstructured":"Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Deep residual learning for image recognition. In CVPR .  Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Deep residual learning for image recognition. In CVPR ."},{"key":"e_1_3_2_2_13_1","volume-title":"A baseline for detecting misclassified and out-of-distribution examples in neural networks. ICLR","author":"Hendrycks Dan","year":"2017","unstructured":"Dan Hendrycks and Kevin Gimpel . 2017. A baseline for detecting misclassified and out-of-distribution examples in neural networks. ICLR ( 2017 ). Dan Hendrycks and Kevin Gimpel. 2017. A baseline for detecting misclassified and out-of-distribution examples in neural networks. ICLR (2017)."},{"key":"e_1_3_2_2_14_1","unstructured":"Will Kay Joao Carreira Karen Simonyan Brian Zhang Chloe Hillier Sudheendra Vijayanarasimhan Fabio Viola Tim Green Trevor Back Paul Natsev etal 2017a. The kinetics human action video dataset. arXiv preprint arXiv:1705.06950 (2017).  Will Kay Joao Carreira Karen Simonyan Brian Zhang Chloe Hillier Sudheendra Vijayanarasimhan Fabio Viola Tim Green Trevor Back Paul Natsev et al. 2017a. The kinetics human action video dataset. arXiv preprint arXiv:1705.06950 (2017)."},{"key":"e_1_3_2_2_15_1","unstructured":"Will Kay Joao Carreira Karen Simonyan Brian Zhang Chloe Hillier Sudheendra Vijayanarasimhan Fabio Viola Tim Green Trevor Back Paul Natsev etal 2017b. The Kinetics Human Action Video Dataset. arXiv:1705.06950 (2017).  Will Kay Joao Carreira Karen Simonyan Brian Zhang Chloe Hillier Sudheendra Vijayanarasimhan Fabio Viola Tim Green Trevor Back Paul Natsev et al. 2017b. The Kinetics Human Action Video Dataset. arXiv:1705.06950 (2017)."},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"crossref","unstructured":"Pilhyeon Lee Youngjung Uh and Hyeran Byun. 2020 a. Background Suppression Network for Weakly-Supervised Temporal Action Localization.. In AAAI. 11320--11327.  Pilhyeon Lee Youngjung Uh and Hyeran Byun. 2020 a. Background Suppression Network for Weakly-Supervised Temporal Action Localization.. In AAAI. 11320--11327.","DOI":"10.1609\/aaai.v34i07.6793"},{"key":"e_1_3_2_2_17_1","volume-title":"2020 b. Background Modeling via Uncertainty Estimation for Weakly-supervised Action Localization. arXiv preprint arXiv:2006.07006","author":"Lee Pilhyeon","year":"2020","unstructured":"Pilhyeon Lee , Jinglu Wang , Yan Lu , and Hyeran Byun . 2020 b. Background Modeling via Uncertainty Estimation for Weakly-supervised Action Localization. arXiv preprint arXiv:2006.07006 ( 2020 ). Pilhyeon Lee, Jinglu Wang, Yan Lu, and Hyeran Byun. 2020 b. Background Modeling via Uncertainty Estimation for Weakly-supervised Action Localization. arXiv preprint arXiv:2006.07006 (2020)."},{"key":"e_1_3_2_2_18_1","unstructured":"Aoxue Li Tiange Luo Zhiwu Lu Tao Xiang and Liwei Wang. 2019. Large-Scale Few-Shot Learning: Knowledge Transfer With Class Hierarchy. In CVPR. 7212--7220.  Aoxue Li Tiange Luo Zhiwu Lu Tao Xiang and Liwei Wang. 2019. Large-Scale Few-Shot Learning: Knowledge Transfer With Class Hierarchy. In CVPR. 7212--7220."},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01225-0_1"},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00139"},{"key":"e_1_3_2_2_21_1","volume-title":"Adversarial Background-Aware Loss for Weakly-supervised Temporal Activity Localization. arXiv preprint arXiv:2007.06643","author":"Min Kyle","year":"2020","unstructured":"Kyle Min and Jason J Corso . 2020. Adversarial Background-Aware Loss for Weakly-supervised Temporal Activity Localization. arXiv preprint arXiv:2007.06643 ( 2020 ). Kyle Min and Jason J Corso. 2020. Adversarial Background-Aware Loss for Weakly-supervised Temporal Activity Localization. arXiv preprint arXiv:2007.06643 (2020)."},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00706"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00560"},{"key":"e_1_3_2_2_24_1","volume-title":"Representation learning with contrastive predictive coding. arXiv preprint arXiv:1807.03748","author":"van den Oord Aaron","year":"2018","unstructured":"Aaron van den Oord , Yazhe Li , and Oriol Vinyals . 2018. Representation learning with contrastive predictive coding. arXiv preprint arXiv:1807.03748 ( 2018 ). Aaron van den Oord, Yazhe Li, and Oriol Vinyals. 2018. Representation learning with contrastive predictive coding. arXiv preprint arXiv:1807.03748 (2018)."},{"key":"e_1_3_2_2_25_1","volume-title":"Few-shot image recognition by predicting parameters from activations. CoRR, abs\/1706.03466","author":"Qiao Siyuan","year":"2017","unstructured":"Siyuan Qiao , Chenxi Liu , Wei Shen , and Alan L Yuille . 2017. Few-shot image recognition by predicting parameters from activations. CoRR, abs\/1706.03466 , Vol. 1 ( 2017 ). Siyuan Qiao, Chenxi Liu, Wei Shen, and Alan L Yuille. 2017. Few-shot image recognition by predicting parameters from activations. CoRR, abs\/1706.03466, Vol. 1 (2017)."},{"key":"e_1_3_2_2_26_1","volume-title":"Meta-learning with latent embedding optimization. ICLR","author":"Rusu Andrei A","year":"2019","unstructured":"Andrei A Rusu , Dushyant Rao , Jakub Sygnowski , Oriol Vinyals , Razvan Pascanu , Simon Osindero , and Raia Hadsell . 2019. Meta-learning with latent embedding optimization. ICLR ( 2019 ). Andrei A Rusu, Dushyant Rao, Jakub Sygnowski, Oriol Vinyals, Razvan Pascanu, Simon Osindero, and Raia Hadsell. 2019. Meta-learning with latent embedding optimization. ICLR (2019)."},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.155"},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.119"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.5555\/3294996.3295163"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00678"},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.5555\/3042817.3043064"},{"key":"e_1_3_2_2_32_1","volume-title":"Contrastive multiview coding. arXiv preprint arXiv:1906.05849","author":"Tian Yonglong","year":"2019","unstructured":"Yonglong Tian , Dilip Krishnan , and Phillip Isola . 2019. Contrastive multiview coding. arXiv preprint arXiv:1906.05849 ( 2019 ). Yonglong Tian, Dilip Krishnan, and Phillip Isola. 2019. Contrastive multiview coding. arXiv preprint arXiv:1906.05849 (2019)."},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"crossref","unstructured":"Pavel Tokmakov Yu-Xiong Wang and Martial Hebert. 2019. Learning compositional representations for few-shot recognition. In ICCV. 6372--6381.  Pavel Tokmakov Yu-Xiong Wang and Martial Hebert. 2019. Learning compositional representations for few-shot recognition. In ICCV. 6372--6381.","DOI":"10.1109\/ICCV.2019.00647"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.5555\/3157382.3157504"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2012.256"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.678"},{"key":"e_1_3_2_2_37_1","volume-title":"Temporal segment networks for action recognition in videos","author":"Wang Limin","year":"2018","unstructured":"Limin Wang , Yuanjun Xiong , Zhe Wang , Yu Qiao , Dahua Lin , Xiaoou Tang , and Luc Van Gool . 2018. Temporal segment networks for action recognition in videos . IEEE transactions on pattern analysis and machine intelligence, Vol. 41 , 11 ( 2018 ), 2740--2755. Limin Wang, Yuanjun Xiong, Zhe Wang, Yu Qiao, Dahua Lin, Xiaoou Tang, and Luc Van Gool. 2018. Temporal segment networks for action recognition in videos. IEEE transactions on pattern analysis and machine intelligence, Vol. 41, 11 (2018), 2740--2755."},{"key":"e_1_3_2_2_38_1","volume-title":"Revisiting Few-shot Activity Detection with Class Similarity Control. arXiv preprint arXiv:2004.00137","author":"Xu Huijuan","year":"2020","unstructured":"Huijuan Xu , Ximeng Sun , Eric Tzeng , Abir Das , Kate Saenko , and Trevor Darrell . 2020. Revisiting Few-shot Activity Detection with Class Similarity Control. arXiv preprint arXiv:2004.00137 ( 2020 ). Huijuan Xu, Ximeng Sun, Eric Tzeng, Abir Das, Kate Saenko, and Trevor Darrell. 2020. Revisiting Few-shot Activity Detection with Class Similarity Control. arXiv preprint arXiv:2004.00137 (2020)."},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00157"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00394"},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.317"},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"crossref","unstructured":"Bolei Zhou Aditya Khosla Agata Lapedriza Aude Oliva and Antonio Torralba. 2016. Learning deep features for discriminative localization. In CVPR. 2921--2929.  Bolei Zhou Aditya Khosla Agata Lapedriza Aude Oliva and Antonio Torralba. 2016. Learning deep features for discriminative localization. In CVPR. 2921--2929.","DOI":"10.1109\/CVPR.2016.319"},{"key":"e_1_3_2_2_43_1","unstructured":"Linchao Zhu and Yi Yang. 2018. Compound memory networks for few-shot video classification. In ECCV. 751--766.  Linchao Zhu and Yi Yang. 2018. Compound memory networks for few-shot video classification. In ECCV. 751--766."},{"key":"e_1_3_2_2_44_1","volume-title":"2020 a. Adaptation-Oriented Feature Projection for One-shot Action Recognition","author":"Zou Yixiong","year":"2020","unstructured":"Yixiong Zou , Yemin Shi , Daochen Shi , Yaowei Wang , Yongsheng Liang , and Yonghong Tian . 2020 a. Adaptation-Oriented Feature Projection for One-shot Action Recognition . IEEE Transactions on Multimedia ( 2020 ). Yixiong Zou, Yemin Shi, Daochen Shi, Yaowei Wang, Yongsheng Liang, and Yonghong Tian. 2020 a. Adaptation-Oriented Feature Projection for One-shot Action Recognition. IEEE Transactions on Multimedia (2020)."},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICME.2018.8486447"},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413849"}],"event":{"name":"MM '21: ACM Multimedia Conference","location":"Virtual Event China","acronym":"MM '21","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 29th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3474085.3475197","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3474085.3475197","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T20:48:47Z","timestamp":1750193327000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3474085.3475197"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,10,17]]},"references-count":46,"alternative-id":["10.1145\/3474085.3475197","10.1145\/3474085"],"URL":"https:\/\/doi.org\/10.1145\/3474085.3475197","relation":{},"subject":[],"published":{"date-parts":[[2021,10,17]]},"assertion":[{"value":"2021-10-17","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}