{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,6]],"date-time":"2026-06-06T16:18:50Z","timestamp":1780762730905,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":43,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,10,10]],"date-time":"2022-10-10T00:00:00Z","timestamp":1665360000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61871378, U2003111, 62122013, U2001211"],"award-info":[{"award-number":["61871378, U2003111, 62122013, U2001211"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Defense Industrial Technology Development Program","award":["JCKY2021906A001"],"award-info":[{"award-number":["JCKY2021906A001"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,10,10]]},"DOI":"10.1145\/3503161.3548077","type":"proceedings-article","created":{"date-parts":[[2022,10,10]],"date-time":"2022-10-10T15:42:46Z","timestamp":1665416566000},"page":"3820-3828","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":30,"title":["Dynamic Graph Modeling for Weakly-Supervised Temporal Action Localization"],"prefix":"10.1145","author":[{"given":"Haichao","family":"Shi","sequence":"first","affiliation":[{"name":"Institute of Information Engineering, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiao-Yu","family":"Zhang","sequence":"additional","affiliation":[{"name":"Institute of Information Engineering, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Changsheng","family":"Li","sequence":"additional","affiliation":[{"name":"Beijing Institute of Technology, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lixing","family":"Gong","sequence":"additional","affiliation":[{"name":"JD.com, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yong","family":"Li","sequence":"additional","affiliation":[{"name":"JD.com, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yongjun","family":"Bao","sequence":"additional","affiliation":[{"name":"JD.com, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2022,10,10]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"Victor Ponce-L\u00f3pez, Xavier Bar\u00f3, Isabelle Guyon, Shohreh Kasaei, and Sergio Escalera.","author":"Asadi-Aghbolaghi Maryam","year":"2017","unstructured":"Maryam Asadi-Aghbolaghi , Albert Clap\u00e9s , Marco Bellantonio , Hugo Jair Escalante , Victor Ponce-L\u00f3pez, Xavier Bar\u00f3, Isabelle Guyon, Shohreh Kasaei, and Sergio Escalera. 2017 . A Survey on Deep Learning Based Approaches for Action and Gesture Recognition in Image Sequences. In FG. 476--483. Maryam Asadi-Aghbolaghi, Albert Clap\u00e9s, Marco Bellantonio, Hugo Jair Escalante, Victor Ponce-L\u00f3pez, Xavier Bar\u00f3, Isabelle Guyon, Shohreh Kasaei, and Sergio Escalera. 2017. A Survey on Deep Learning Based Approaches for Action and Gesture Recognition in Image Sequences. In FG. 476--483."},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"crossref","unstructured":"Jo a o Carreira and Andrew Zisserman. 2017. Quo Vadis Action Recognition? A New Model and the Kinetics Dataset. In CVPR. 4724--4733.  Jo a o Carreira and Andrew Zisserman. 2017. Quo Vadis Action Recognition? A New Model and the Kinetics Dataset. In CVPR. 4724--4733.","DOI":"10.1109\/CVPR.2017.502"},{"key":"e_1_3_2_2_3_1","unstructured":"Jie Chen Tengfei Ma and Cao Xiao. 2018. FastGCN: Fast Learning with Graph Convolutional Networks via Importance Sampling. In ICLR. OpenReview.net. https:\/\/openreview.net\/forum?id=rytstxWAW  Jie Chen Tengfei Ma and Cao Xiao. 2018. FastGCN: Fast Learning with Graph Convolutional Networks via Importance Sampling. In ICLR. OpenReview.net. https:\/\/openreview.net\/forum?id=rytstxWAW"},{"key":"e_1_3_2_2_4_1","volume-title":"Victor Escorcia and Juan Carlos Niebles","author":"Fabian Caba Heilbron Bernard Ghanem","year":"2015","unstructured":"Bernard Ghanem Fabian Caba Heilbron , Victor Escorcia and Juan Carlos Niebles . 2015 . ActivityNet: A Large-Scale Video Benchmark for Human Activity Understanding. In CVPR. 961--970. Bernard Ghanem Fabian Caba Heilbron, Victor Escorcia and Juan Carlos Niebles. 2015. ActivityNet: A Large-Scale Video Benchmark for Human Activity Understanding. In CVPR. 961--970."},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"crossref","unstructured":"Junyu Gao Mengyuan Chen and Changsheng Xu. 2022. Fine-grained Temporal Contrastive Learning for Weakly-supervised Temporal Action Localization. In CVPR. 19999--20009.  Junyu Gao Mengyuan Chen and Changsheng Xu. 2022. Fine-grained Temporal Contrastive Learning for Weakly-supervised Temporal Action Localization. In CVPR. 19999--20009.","DOI":"10.1109\/CVPR52688.2022.01937"},{"key":"e_1_3_2_2_6_1","unstructured":"William L. Hamilton Zhitao Ying and Jure Leskovec. 2017. Inductive Representation Learning on Large Graphs. In NeurIPS. 1024--1034.  William L. Hamilton Zhitao Ying and Jure Leskovec. 2017. Inductive Representation Learning on Large Graphs. In NeurIPS. 1024--1034."},{"key":"e_1_3_2_2_7_1","unstructured":"Wen-bing Huang Tong Zhang Yu Rong and Junzhou Huang. 2018. Adaptive Sampling Towards Fast Graph Representation Learning. In NeurIPS. 4563--4572.  Wen-bing Huang Tong Zhang Yu Rong and Junzhou Huang. 2018. Adaptive Sampling Towards Fast Graph Representation Learning. In NeurIPS. 4563--4572."},{"key":"e_1_3_2_2_8_1","volume":"202","author":"Islam Ashraful","unstructured":"Ashraful Islam , Chengjiang Long , and Richard J. Radke. 202 1. A Hybrid Attention Mechanism for Weakly-Supervised Temporal Action Localization. In AAAI. 1637--1645. Ashraful Islam, Chengjiang Long, and Richard J. Radke. 2021. A Hybrid Attention Mechanism for Weakly-Supervised Temporal Action Localization. In AAAI. 1637--1645.","journal-title":"Richard J. Radke."},{"key":"e_1_3_2_2_9_1","unstructured":"Y.-G. Jiang J. Liu A. Roshan Zamir G. Toderici I. Laptev M. Shah and R. Sukthankar. 2014. THUMOS Challenge: Action Recognition with a Large Number of Classes. http:\/\/crcv.ucf.edu\/THUMOS14\/.  Y.-G. Jiang J. Liu A. Roshan Zamir G. Toderici I. Laptev M. Shah and R. Sukthankar. 2014. THUMOS Challenge: Action Recognition with a Large Number of Classes. http:\/\/crcv.ucf.edu\/THUMOS14\/."},{"key":"e_1_3_2_2_10_1","volume-title":"Kipf and Max Welling","author":"Thomas","year":"2017","unstructured":"Thomas N. Kipf and Max Welling . 2017 . Semi-Supervised Classification with Graph Convolutional Networks. In ICLR. OpenReview .net. https:\/\/openreview.net\/forum?id=SJU4ayYgl Thomas N. Kipf and Max Welling. 2017. Semi-Supervised Classification with Graph Convolutional Networks. In ICLR. OpenReview.net. https:\/\/openreview.net\/forum?id=SJU4ayYgl"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"crossref","unstructured":"Pilhyeon Lee Youngjung Uh and Hyeran Byun. 2020. Background Suppression Network for Weakly-Supervised Temporal Action Localization. In AAAI. 11320--11327.  Pilhyeon Lee Youngjung Uh and Hyeran Byun. 2020. Background Suppression Network for Weakly-Supervised Temporal Action Localization. In AAAI. 11320--11327.","DOI":"10.1609\/aaai.v34i07.6793"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"crossref","unstructured":"Yong Jae Lee Joydeep Ghosh and Kristen Grauman. 2012. Discovering important people and objects for egocentric video summarization. In CVPR. 1346--1353.  Yong Jae Lee Joydeep Ghosh and Kristen Grauman. 2012. Discovering important people and objects for egocentric video summarization. In CVPR. 1346--1353.","DOI":"10.1109\/CVPR.2012.6247820"},{"key":"e_1_3_2_2_13_1","unstructured":"Tianwei Lin Xu Zhao and Zheng Shou. 2017. Single Shot Temporal Action Detection. In MM. 988--996.  Tianwei Lin Xu Zhao and Zheng Shou. 2017. Single Shot Temporal Action Detection. In MM. 988--996."},{"key":"e_1_3_2_2_14_1","volume-title":"BSN: Boundary Sensitive Network for Temporal Action Proposal Generation. In ECCV. 3--21.","author":"Lin Tianwei","year":"2018","unstructured":"Tianwei Lin , Xu Zhao , Haisheng Su , Chongjing Wang , and Ming Yang . 2018 . BSN: Boundary Sensitive Network for Temporal Action Proposal Generation. In ECCV. 3--21. Tianwei Lin, Xu Zhao, Haisheng Su, Chongjing Wang, and Ming Yang. 2018. BSN: Boundary Sensitive Network for Temporal Action Proposal Generation. In ECCV. 3--21."},{"key":"e_1_3_2_2_15_1","unstructured":"Daochang Liu Tingting Jiang and Yizhou Wang. 2019a. Completeness Modeling and Context Separation for Weakly Supervised Temporal Action Localization. In CVPR. 1298--1307.  Daochang Liu Tingting Jiang and Yizhou Wang. 2019a. Completeness Modeling and Context Separation for Weakly Supervised Temporal Action Localization. In CVPR. 1298--1307."},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"crossref","unstructured":"Ziyi Liu Le Wang Qilin Zhang Zhanning Gao Zhenxing Niu Nanning Zheng and Gang Hua. 2019b. Weakly Supervised Temporal Action Localization Through Contrast Based Evaluation Networks. In ICCV. 3898--3907.  Ziyi Liu Le Wang Qilin Zhang Zhanning Gao Zhenxing Niu Nanning Zheng and Gang Hua. 2019b. Weakly Supervised Temporal Action Localization Through Contrast Based Evaluation Networks. In ICCV. 3898--3907.","DOI":"10.1109\/ICCV.2019.00400"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"crossref","unstructured":"Ziyi Liu Le Wang Qilin Zhang Wei Tang Junsong Yuan Nanning Zheng and Gang Hua. 2021. ACSNet: Action-Context Separation Network for Weakly Supervised Temporal Action Localization. In AAAI. 2233--2241.  Ziyi Liu Le Wang Qilin Zhang Wei Tang Junsong Yuan Nanning Zheng and Gang Hua. 2021. ACSNet: Action-Context Separation Network for Weakly Supervised Temporal Action Localization. In AAAI. 2233--2241.","DOI":"10.1609\/aaai.v35i3.16322"},{"key":"e_1_3_2_2_18_1","volume":"202","author":"Min Kyle","unstructured":"Kyle Min and Jason J. Corso. 202 0. Adversarial Background-Aware Loss for Weakly-Supervised Temporal Activity Localization. In ECCV. 283--299. Kyle Min and Jason J. Corso. 2020. Adversarial Background-Aware Loss for Weakly-Supervised Temporal Activity Localization. In ECCV. 283--299.","journal-title":"Jason J. Corso."},{"key":"e_1_3_2_2_19_1","volume-title":"Fahad Shahbaz Khan, and Ling Shao","author":"Narayan Sanath","year":"2019","unstructured":"Sanath Narayan , Hisham Cholakkal , Fahad Shahbaz Khan, and Ling Shao . 2019 . 3C-Net: Category Count and Center Loss for Weakly-Supervised Action Localization . In ICCV. 8678--8686. Sanath Narayan, Hisham Cholakkal, Fahad Shahbaz Khan, and Ling Shao. 2019. 3C-Net: Category Count and Center Loss for Weakly-Supervised Action Localization. In ICCV. 8678--8686."},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"crossref","unstructured":"Phuc Nguyen Ting Liu Gautam Prasad and Bohyung Han. 2018. Weakly Supervised Action Localization by Sparse Temporal Pooling Network. In CVPR. 6752--6761.  Phuc Nguyen Ting Liu Gautam Prasad and Bohyung Han. 2018. Weakly Supervised Action Localization by Sparse Temporal Pooling Network. In CVPR. 6752--6761.","DOI":"10.1109\/CVPR.2018.00706"},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"crossref","unstructured":"Phuc Xuan Nguyen Deva Ramanan and Charless C Fowlkes. 2019. Weakly-supervised action localization with background modeling. In ICCV. 5502--5511.  Phuc Xuan Nguyen Deva Ramanan and Charless C Fowlkes. 2019. Weakly-supervised action localization with background modeling. In ICCV. 5502--5511.","DOI":"10.1109\/ICCV.2019.00560"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"crossref","unstructured":"Dan Oneata Jakob J. Verbeek and Cordelia Schmid. 2013. Action and Event Recognition with Fisher Vectors on a Compact Feature Set. In ICCV. 1817--1824.  Dan Oneata Jakob J. Verbeek and Cordelia Schmid. 2013. Action and Event Recognition with Fisher Vectors on a Compact Feature Set. In ICCV. 1817--1824.","DOI":"10.1109\/ICCV.2013.228"},{"key":"e_1_3_2_2_23_1","volume-title":"Roy-Chowdhury","author":"Paul Sujoy","year":"2018","unstructured":"Sujoy Paul , Sourya Roy , and Amit K . Roy-Chowdhury . 2018 . W-TALC: Weakly- Supervised Temporal Activity Localization and Classification. In ECCV. 588--607. Sujoy Paul, Sourya Roy, and Amit K. Roy-Chowdhury. 2018. W-TALC: Weakly-Supervised Temporal Activity Localization and Classification. In ECCV. 588--607."},{"key":"e_1_3_2_2_24_1","volume-title":"Action Graphs: Weakly-supervised Action Localization with Graph Convolution Networks","author":"Rashid Maheen","year":"2020","unstructured":"Maheen Rashid , Hedvig Kjellstr\u00f6m , and Yong Jae Lee . 2020 . Action Graphs: Weakly-supervised Action Localization with Graph Convolution Networks . In WACV. IEEE , 604--613. Maheen Rashid, Hedvig Kjellstr\u00f6m, and Yong Jae Lee. 2020. Action Graphs: Weakly-supervised Action Localization with Graph Convolution Networks. In WACV. IEEE, 604--613."},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"crossref","unstructured":"Yantao Shen Hongsheng Li Shuai Yi Dapeng Chen and Xiaogang Wang. 2018. Person Re-identification with Deep Similarity-Guided Graph Neural Network. In ECCV. 508--526.  Yantao Shen Hongsheng Li Shuai Yi Dapeng Chen and Xiaogang Wang. 2018. Person Re-identification with Deep Similarity-Guided Graph Neural Network. In ECCV. 508--526.","DOI":"10.1007\/978-3-030-01267-0_30"},{"key":"e_1_3_2_2_26_1","unstructured":"Baifeng Shi Qi Dai Yadong Mu and Jingdong Wang. 2020. Weakly-Supervised Action Localization by Generative Attention Modeling. In CVPR. 1006--1016.  Baifeng Shi Qi Dai Yadong Mu and Jingdong Wang. 2020. Weakly-Supervised Action Localization by Generative Attention Modeling. In CVPR. 1006--1016."},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"crossref","unstructured":"Zheng Shou Hang Gao Lei Zhang Kazuyuki Miyazawa and Shih-Fu Chang. 2018. AutoLoc: Weakly-Supervised Temporal Action Localization in Untrimmed Videos. In ECCV. 162--179.  Zheng Shou Hang Gao Lei Zhang Kazuyuki Miyazawa and Shih-Fu Chang. 2018. AutoLoc: Weakly-Supervised Temporal Action Localization in Untrimmed Videos. In ECCV. 162--179.","DOI":"10.1007\/978-3-030-01270-0_10"},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"crossref","unstructured":"Zheng Shou Dongang Wang and Shih-Fu Chang. 2016. Temporal Action Localization in Untrimmed Videos via Multi-stage CNNs. In CVPR. 1049--1058.  Zheng Shou Dongang Wang and Shih-Fu Chang. 2016. Temporal Action Localization in Untrimmed Videos via Multi-stage CNNs. In CVPR. 1049--1058.","DOI":"10.1109\/CVPR.2016.119"},{"key":"e_1_3_2_2_29_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N. Gomez Lukasz Kaiser and Illia Polosukhin. 2017. Attention is All you Need. In NeurIPS. 5998--6008.  Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N. Gomez Lukasz Kaiser and Illia Polosukhin. 2017. Attention is All you Need. In NeurIPS. 5998--6008."},{"key":"e_1_3_2_2_30_1","unstructured":"Petar Velickovic Guillem Cucurull Arantxa Casanova Adriana Romero Pietro Li\u00f2 and Yoshua Bengio. 2018. Graph Attention Networks. In ICLR. OpenReview.net. https:\/\/openreview.net\/forum?id=rJXMpikCZ  Petar Velickovic Guillem Cucurull Arantxa Casanova Adriana Romero Pietro Li\u00f2 and Yoshua Bengio. 2018. Graph Attention Networks. In ICLR. OpenReview.net. https:\/\/openreview.net\/forum?id=rJXMpikCZ"},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"crossref","unstructured":"Limin Wang Yuanjun Xiong Dahua Lin and Luc Van Gool. 2017. UntrimmedNets for Weakly Supervised Action Recognition and Detection. In CVPR. 6402--6411.  Limin Wang Yuanjun Xiong Dahua Lin and Luc Van Gool. 2017. UntrimmedNets for Weakly Supervised Action Recognition and Detection. In CVPR. 6402--6411.","DOI":"10.1109\/CVPR.2017.678"},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"crossref","unstructured":"Xiaolong Wang Ross B. Girshick Abhinav Gupta and Kaiming He. 2018. Non-Local Neural Networks. In CVPR. 7794--7803.  Xiaolong Wang Ross B. Girshick Abhinav Gupta and Kaiming He. 2018. Non-Local Neural Networks. In CVPR. 7794--7803.","DOI":"10.1109\/CVPR.2018.00813"},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"crossref","unstructured":"Xiaolong Wang and Abhinav Gupta. 2018. Videos as Space-Time Region Graphs. In ECCV. 413--431.  Xiaolong Wang and Abhinav Gupta. 2018. Videos as Space-Time Region Graphs. In ECCV. 413--431.","DOI":"10.1007\/978-3-030-01228-1_25"},{"key":"e_1_3_2_2_34_1","unstructured":"Mengmeng Xu Chen Zhao David S. Rojas Ali K. Thabet and Bernard Ghanem. 2020. G-TAD: Sub-Graph Localization for Temporal Action Detection. In CVPR. 10153--10162.  Mengmeng Xu Chen Zhao David S. Rojas Ali K. Thabet and Bernard Ghanem. 2020. G-TAD: Sub-Graph Localization for Temporal Action Detection. In CVPR. 10153--10162."},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"crossref","unstructured":"Sijie Yan Yuanjun Xiong and Dahua Lin. 2018. Spatial Temporal Graph Convolutional Networks for Skeleton-Based Action Recognition. In AAAI. 7444--7452.  Sijie Yan Yuanjun Xiong and Dahua Lin. 2018. Spatial Temporal Graph Convolutional Networks for Skeleton-Based Action Recognition. In AAAI. 7444--7452.","DOI":"10.1609\/aaai.v32i1.12328"},{"key":"e_1_3_2_2_36_1","unstructured":"Yuan Yuan Yueming Lyu Xi Shen Ivor W. Tsang and Dit-Yan Yeung. 2019. Marginalized Average Attentional Network for Weakly-Supervised Learning. In ICLR. OpenReview.net. https:\/\/openreview.net\/forum?id=HkljioCcFQ  Yuan Yuan Yueming Lyu Xi Shen Ivor W. Tsang and Dit-Yan Yeung. 2019. Marginalized Average Attentional Network for Weakly-Supervised Learning. In ICLR. OpenReview.net. https:\/\/openreview.net\/forum?id=HkljioCcFQ"},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"crossref","unstructured":"Runhao Zeng Wenbing Huang Mingkui Tan Yu Rong Peilin Zhao Junzhou Huang and Chuang Gan. 2019. Graph convolutional networks for temporal action localization. In ICCV. 7094--7103.  Runhao Zeng Wenbing Huang Mingkui Tan Yu Rong Peilin Zhao Junzhou Huang and Chuang Gan. 2019. Graph convolutional networks for temporal action localization. In ICCV. 7094--7103.","DOI":"10.1109\/ICCV.2019.00719"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"crossref","unstructured":"Yuanhao Zhai Le Wang Wei Tang Qilin Zhang Junsong Yuan and Gang Hua. 2020. Two-Stream Consensus Network for Weakly-Supervised Temporal Action Localization. In ECCV. 37--54.  Yuanhao Zhai Le Wang Wei Tang Qilin Zhang Junsong Yuan and Gang Hua. 2020. Two-Stream Consensus Network for Weakly-Supervised Temporal Action Localization. In ECCV. 37--54.","DOI":"10.1007\/978-3-030-58539-6_3"},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"crossref","unstructured":"Can Zhang Meng Cao Dongming Yang Jie Chen and Yuexian Zou. 2021. CoLA: Weakly-Supervised Temporal Action Localization With Snippet Contrastive Learning. In CVPR. 16010--16019.  Can Zhang Meng Cao Dongming Yang Jie Chen and Yuexian Zou. 2021. CoLA: Weakly-Supervised Temporal Action Localization With Snippet Contrastive Learning. In CVPR. 16010--16019.","DOI":"10.1109\/CVPR46437.2021.01575"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"crossref","unstructured":"Xiao-Yu Zhang Haichao Shi Changsheng Li and Peng Li. 2020. Multi-Instance Multi-Label Action Recognition and Localization Based on Spatio-Temporal Pre-Trimming for Untrimmed Videos. In AAAI. 12886--12893.  Xiao-Yu Zhang Haichao Shi Changsheng Li and Peng Li. 2020. Multi-Instance Multi-Label Action Recognition and Localization Based on Spatio-Temporal Pre-Trimming for Untrimmed Videos. In AAAI. 12886--12893.","DOI":"10.1609\/aaai.v34i07.6986"},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"crossref","unstructured":"Xiaoyu Zhang Haichao Shi Changsheng Li Kai Zheng Xiaobin Zhu and Lixin Duan. 2019. Learning Transferable Self-Attentive Representations for Action Recognition in Untrimmed Videos with Weak Supervision. In AAAI. 9227--9234.  Xiaoyu Zhang Haichao Shi Changsheng Li Kai Zheng Xiaobin Zhu and Lixin Duan. 2019. Learning Transferable Self-Attentive Representations for Action Recognition in Untrimmed Videos with Weak Supervision. In AAAI. 9227--9234.","DOI":"10.1609\/aaai.v33i01.33019227"},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2022.3185485"},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"crossref","unstructured":"Yue Zhao Yuanjun Xiong Limin Wang Zhirong Wu Xiaoou Tang and Dahua Lin. 2017. Temporal Action Detection with Structured Segment Networks. In ICCV. 2933--2942.  Yue Zhao Yuanjun Xiong Limin Wang Zhirong Wu Xiaoou Tang and Dahua Lin. 2017. Temporal Action Detection with Structured Segment Networks. In ICCV. 2933--2942.","DOI":"10.1109\/ICCV.2017.317"}],"event":{"name":"MM '22: The 30th ACM International Conference on Multimedia","location":"Lisboa Portugal","acronym":"MM '22","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 30th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3503161.3548077","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3503161.3548077","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T19:02:30Z","timestamp":1750186950000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3503161.3548077"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,10,10]]},"references-count":43,"alternative-id":["10.1145\/3503161.3548077","10.1145\/3503161"],"URL":"https:\/\/doi.org\/10.1145\/3503161.3548077","relation":{},"subject":[],"published":{"date-parts":[[2022,10,10]]},"assertion":[{"value":"2022-10-10","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}