{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,23]],"date-time":"2025-08-23T05:23:44Z","timestamp":1755926624438,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":49,"publisher":"ACM","license":[{"start":{"date-parts":[[2021,10,17]],"date-time":"2021-10-17T00:00:00Z","timestamp":1634428800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61976049, 62072080, 61632007, 61976116 and U20B2063"],"award-info":[{"award-number":["61976049, 62072080, 61632007, 61976116 and U20B2063"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Fundamental Research Funds for the Central Universities under Project","award":["ZYGX2019Z015, 30920021135"],"award-info":[{"award-number":["ZYGX2019Z015, 30920021135"]}]},{"name":"Sichuan Science and Technology Program","award":["2018GZDZX0032, 2019ZDZX0008, 2019YFG0003, 2019YFG0533 and 2020YFS0057"],"award-info":[{"award-number":["2018GZDZX0032, 2019ZDZX0008, 2019YFG0003, 2019YFG0533 and 2020YFS0057"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2021,10,17]]},"DOI":"10.1145\/3474085.3475616","type":"proceedings-article","created":{"date-parts":[[2021,10,18]],"date-time":"2021-10-18T06:57:34Z","timestamp":1634540254000},"page":"4930-4938","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["CAA"],"prefix":"10.1145","author":[{"given":"Yifan","family":"Ren","sequence":"first","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xing","family":"Xu","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fumin","family":"Shen","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yazhou","family":"Yao","sequence":"additional","affiliation":[{"name":"Nanjing University of Science and Technology, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Huimin","family":"Lu","sequence":"additional","affiliation":[{"name":"Kyushu Institute of Technology, Kitakyushu, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2021,10,17]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Boundary Content Graph Neural Network for Temporal Action Proposal Generation. In European Conference on Computer Vision","volume":"12373","author":"Bai Yueran","year":"2020"},{"volume-title":"Soft-NMS - Improving Object Detection with One Line of Code. In IEEE International Conference on Computer Vision. 5562--5570","author":"Bodla Navaneeth","key":"e_1_3_2_1_2_1"},{"volume-title":"SST: Single-Stream Temporal Action Proposals. In IEEE Conference on Computer Vision and Pattern Recognition. 6373--6382","year":"2017","author":"Buch Shyamal","key":"e_1_3_2_1_3_1"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"volume-title":"Rethinking the Faster R-CNN Architecture for Temporal Action Localization. In IEEE Conference on Computer Vision and Pattern Recognition. 1130--1139","year":"2018","author":"Chao Yu-Wei","key":"e_1_3_2_1_5_1"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.5555\/3157096.3157139"},{"volume-title":"Temporal Context Network for Activity Localization in Videos. In IEEE International Conference on Computer Vision. 5727--5736","year":"2017","author":"Dai Xiyang","key":"e_1_3_2_1_7_1"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2005.177"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1007\/11744047_33"},{"volume-title":"SlowFast Networks for Video Recognition. In IEEE International Conference on Computer Vision. 6201--6210","year":"2019","author":"Feichtenhofer Christoph","key":"e_1_3_2_1_10_1"},{"volume-title":"Accurate Temporal Action Proposal Generation with Relation-Aware Pyramid Network. In AAAI Conference on Artificial Intelligence. 10810--10817","year":"2020","author":"Gao Jialin","key":"e_1_3_2_1_11_1"},{"volume-title":"TURN TAP: Temporal Unit Regression Network for Temporal Action Proposals. In IEEE International Conference on Computer Vision. 3648--3656","year":"2017","author":"Gao Jiyang","key":"e_1_3_2_1_12_1"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.169"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.81"},{"volume-title":"Deep Residual Learning for Image Recognition. In 2016 IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2016","year":"2016","author":"He Kaiming","key":"e_1_3_2_1_15_1"},{"volume-title":"SCC: Semantic Context Cascade for Efficient Action Detection. In IEEE Conference on Computer Vision and Pattern Recognition. 3175--3184","year":"2017","author":"Heilbron Fabian Caba","key":"e_1_3_2_1_16_1"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298698"},{"volume-title":"Strip Pooling: Rethinking Spatial Pooling for Scene Parsing. In IEEE Conference on Computer Vision and Pattern Recognition. 4002--4011","year":"2020","author":"Hou Qibin","key":"e_1_3_2_1_18_1"},{"volume-title":"Squeeze-and-Excitation Networks. In 2018 IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2018","year":"2018","author":"Hu Jie","key":"e_1_3_2_1_19_1"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2019.2952088"},{"volume-title":"STM: Spatio Temporal and Motion Encoding for Action Recognition. In IEEE International Conference on Computer Vision. 2000--2009","year":"2019","author":"Jiang Boyuan","key":"e_1_3_2_1_21_1"},{"key":"e_1_3_2_1_22_1","unstructured":"Y.-G. Jiang J. Liu A. Roshan Zamir G. Toderici I. Laptev M. Shah and R. Sukthankar. 2014. THUMOS Challenge: Action Recognition with a Large Number of Classes. http:\/\/crcv.ucf.edu\/THUMOS14\/.  Y.-G. Jiang J. Liu A. Roshan Zamir G. Toderici I. Laptev M. Shah and R. Sukthankar. 2014. THUMOS Challenge: Action Recognition with a Large Number of Classes. http:\/\/crcv.ucf.edu\/THUMOS14\/."},{"volume-title":"Karen Simonyan, Brian Zhang, Chloe Hillier, Sudheendra Vijayanarasimhan, Fabio Viola, Tim Green, Trevor Back, Paul Natsev, Mustafa Suleyman, and Andrew Zisserman.","year":"2017","author":"Kay Will","key":"e_1_3_2_1_23_1"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.5555\/2999134.2999257"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i04.5893"},{"volume-title":"TEA: Temporal Excitation and Aggregation for Action Recognition. In IEEE Conference on Computer Vision and Pattern Recognition. 906--915","year":"2020","author":"Li Yan","key":"e_1_3_2_1_26_1"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6815"},{"volume-title":"BMN: Boundary-Matching Network for Temporal Action Proposal Generation. In IEEE International Conference on Computer Vision. 3888--3897","year":"2019","author":"Lin Tianwei","key":"e_1_3_2_1_28_1"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123343"},{"key":"e_1_3_2_1_30_1","volume-title":"BSN: Boundary Sensitive Network for Temporal Action Proposal Generation. In European Conference on Computer Vision","volume":"11208","author":"Lin Tianwei","year":"2018"},{"key":"e_1_3_2_1_31_1","volume-title":"TSI: Temporal Scale Invariant Network for Action Proposal Generation. In Asian Conference on Computer Vision","volume":"12626","author":"Liu Shuming","year":"2020"},{"volume":"9905","volume-title":"SSD: Single Shot MultiBox Detector. In European Conference on Computer Vision","author":"Liu Wei","key":"e_1_3_2_1_32_1"},{"volume-title":"Multi-Granularity Generator for Temporal Action Proposal. In IEEE Conference on Computer Vision and Pattern Recognition. 3604--3613","year":"2019","author":"Liu Yuan","key":"e_1_3_2_1_33_1"},{"volume-title":"Real-Time Object Detection. In IEEE Conference on Computer Vision and Pattern Recognition. 779--788","year":"2016","author":"Redmon Joseph","key":"e_1_3_2_1_34_1"},{"volume-title":"Stronger. In IEEE Conference on Computer Vision and Pattern Recognition. 6517--6525","year":"2017","author":"Redmon Joseph","key":"e_1_3_2_1_35_1"},{"volume-title":"YOLOv3: An Incremental Improvement","year":"2018","author":"Redmon Joseph","key":"e_1_3_2_1_36_1"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.5555\/2969239.2969250"},{"volume-title":"CDC: Convolutional-De-Convolutional Networks for Precise Temporal Action Localization in Untrimmed Videos. In IEEE Conference on Computer Vision and Pattern Recognition. 1417--1426","year":"2017","author":"Shou Zheng","key":"e_1_3_2_1_38_1"},{"volume-title":"3rd International Conference on Learning Representations, ICLR","year":"2015","author":"Simonyan Karen","key":"e_1_3_2_1_39_1"},{"volume-title":"BSN: Complementary Boundary Regressor with Scale-Balanced Relation Modeling for Temporal Action Proposal Generation. In AAAI Conference on Artificial Intelligence.","year":"2021","author":"Su Haisheng","key":"e_1_3_2_1_40_1"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.510"},{"volume-title":"UntrimmedNets for Weakly Supervised Action Recognition and Detection. In IEEE Conference on Computer Vision and Pattern Recognition. 6402--6411","year":"2017","author":"Wang Limin","key":"e_1_3_2_1_42_1"},{"key":"e_1_3_2_1_43_1","volume-title":"Temporal Segment Networks: Towards Good Practices for Deep Action Recognition. In European Conference on Computer Vision","volume":"9912","author":"Wang Limin","year":"2016"},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/3326362"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.617"},{"volume-title":"G-TAD: Sub-Graph Localization for Temporal Action Detection. In IEEE Conference on Computer Vision and Pattern Recognition. 10153--10162","year":"2020","author":"Xu Mengmeng","key":"e_1_3_2_1_46_1"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2017.2676345"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2020.2967597"},{"volume-title":"Temporal Action Detection with Structured Segment Networks. In IEEE International Conference on Computer Vision. 2933--2942","year":"2017","author":"Zhao Yue","key":"e_1_3_2_1_49_1"}],"event":{"name":"MM '21: ACM Multimedia Conference","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Virtual Event China","acronym":"MM '21"},"container-title":["Proceedings of the 29th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3474085.3475616","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3474085.3475616","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T20:48:24Z","timestamp":1750193304000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3474085.3475616"}},"subtitle":["Candidate-Aware Aggregation for Temporal Action Detection"],"short-title":[],"issued":{"date-parts":[[2021,10,17]]},"references-count":49,"alternative-id":["10.1145\/3474085.3475616","10.1145\/3474085"],"URL":"https:\/\/doi.org\/10.1145\/3474085.3475616","relation":{},"subject":[],"published":{"date-parts":[[2021,10,17]]},"assertion":[{"value":"2021-10-17","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}