{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,11]],"date-time":"2025-12-11T07:40:28Z","timestamp":1765438828410,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":85,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,10,17]],"date-time":"2022-10-17T00:00:00Z","timestamp":1665964800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"name":"Korea government(MSIT)","award":["B0101-15-0266"],"award-info":[{"award-number":["B0101-15-0266"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,10,17]]},"DOI":"10.1145\/3511808.3557463","type":"proceedings-article","created":{"date-parts":[[2022,10,16]],"date-time":"2022-10-16T01:22:22Z","timestamp":1665883342000},"page":"982-992","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["SWAG-Net: Semantic Word-Aware Graph Network for Temporal Video Grounding"],"prefix":"10.1145","author":[{"given":"Sunoh","family":"Kim","sequence":"first","affiliation":[{"name":"Department of ECE, ASRI, Seoul National University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Taegil","family":"Ha","sequence":"additional","affiliation":[{"name":"Department of ECE, ASRI, Seoul National University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kimin","family":"Yun","sequence":"additional","affiliation":[{"name":"Electronics and Telecommunications Research Institute, Daejeon, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jin Young","family":"Choi","sequence":"additional","affiliation":[{"name":"Department of ECE, ASRI, Seoul National University, Seoul, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2022,10,17]]},"reference":[{"key":"#cr-split#-e_1_3_2_2_1_1.1","doi-asserted-by":"crossref","unstructured":"Lisa Anne Hendricks Oliver Wang Eli Shechtman Josef Sivic Trevor Darrell and Bryan Russell. 2017. Localizing moments in video with natural language. In ICCV. 5803--5812. https:\/\/doi.org\/10.1109\/ICCV.2017.618 10.1109\/ICCV.2017.618","DOI":"10.1109\/ICCV.2017.618"},{"key":"#cr-split#-e_1_3_2_2_1_1.2","doi-asserted-by":"crossref","unstructured":"Lisa Anne Hendricks Oliver Wang Eli Shechtman Josef Sivic Trevor Darrell and Bryan Russell. 2017. Localizing moments in video with natural language. In ICCV. 5803--5812. https:\/\/doi.org\/10.1109\/ICCV.2017.618","DOI":"10.1109\/ICCV.2017.618"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.3115\/1225403.1225421"},{"key":"e_1_3_2_2_3_1","unstructured":"Joan Bruna Wojciech Zaremba Arthur Szlam and Yann LeCun. 2014. Spectral networks and locally connected networks on graphs. In ICLR. https:\/\/openreview.net\/forum?id=DQNsQf-UsoDBa  Joan Bruna Wojciech Zaremba Arthur Szlam and Yann LeCun. 2014. Spectral networks and locally connected networks on graphs. In ICLR. https:\/\/openreview.net\/forum?id=DQNsQf-UsoDBa"},{"key":"#cr-split#-e_1_3_2_2_4_1.1","doi-asserted-by":"crossref","unstructured":"Joao Carreira and Andrew Zisserman. 2017. Quo vadis action recognition? a new model and the kinetics dataset. In CVPR. 6299--6308. https:\/\/doi.org\/10.1109\/CVPR.2017.502 10.1109\/CVPR.2017.502","DOI":"10.1109\/CVPR.2017.502"},{"key":"#cr-split#-e_1_3_2_2_4_1.2","doi-asserted-by":"crossref","unstructured":"Joao Carreira and Andrew Zisserman. 2017. Quo vadis action recognition? a new model and the kinetics dataset. In CVPR. 6299--6308. https:\/\/doi.org\/10.1109\/CVPR.2017.502","DOI":"10.1109\/CVPR.2017.502"},{"key":"#cr-split#-e_1_3_2_2_5_1.1","doi-asserted-by":"crossref","unstructured":"Shaoxiang Chen and Yu-Gang Jiang. 2019. Semantic proposal for activity localization in videos via sentence query. In AAAI. 8199--8206. https:\/\/doi.org\/10.1609\/aaai.v33i01.33018199 10.1609\/aaai.v33i01.33018199","DOI":"10.1609\/aaai.v33i01.33018199"},{"key":"#cr-split#-e_1_3_2_2_5_1.2","doi-asserted-by":"crossref","unstructured":"Shaoxiang Chen and Yu-Gang Jiang. 2019. Semantic proposal for activity localization in videos via sentence query. In AAAI. 8199--8206. https:\/\/doi.org\/10.1609\/aaai.v33i01.33018199","DOI":"10.1609\/aaai.v33i01.33018199"},{"key":"e_1_3_2_2_6_1","unstructured":"Micha\u00ebl Defferrard Xavier Bresson and Pierre Vandergheynst. 2016. Convolutional neural networks on graphs with fast localized spectral filtering. In NeurIPS. https:\/\/arxiv.org\/abs\/1606.09375  Micha\u00ebl Defferrard Xavier Bresson and Pierre Vandergheynst. 2016. Convolutional neural networks on graphs with fast localized spectral filtering. In NeurIPS. https:\/\/arxiv.org\/abs\/1606.09375"},{"key":"e_1_3_2_2_7_1","volume-title":"Bert: Pre-training of deep bidirectional transformers for language understanding. In NAACL-HLT. https:\/\/doi.org\/10.18653\/v1\/N19--1423","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin , Ming-Wei Chang , Kenton Lee , and Kristina Toutanova . 2019 . Bert: Pre-training of deep bidirectional transformers for language understanding. In NAACL-HLT. https:\/\/doi.org\/10.18653\/v1\/N19--1423 10.18653\/v1 Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. Bert: Pre-training of deep bidirectional transformers for language understanding. In NAACL-HLT. https:\/\/doi.org\/10.18653\/v1\/N19--1423"},{"key":"#cr-split#-e_1_3_2_2_8_1.1","doi-asserted-by":"crossref","unstructured":"Jianfeng Dong Xirong Li Chaoxi Xu Shouling Ji Yuan He Gang Yang and Xun Wang. 2019. Dual encoding for zero-example video retrieval. In CVPR. 9346--9355. https:\/\/doi.org\/10.1109\/CVPR.2019.00957 10.1109\/CVPR.2019.00957","DOI":"10.1109\/CVPR.2019.00957"},{"key":"#cr-split#-e_1_3_2_2_8_1.2","doi-asserted-by":"crossref","unstructured":"Jianfeng Dong Xirong Li Chaoxi Xu Shouling Ji Yuan He Gang Yang and Xun Wang. 2019. Dual encoding for zero-example video retrieval. In CVPR. 9346--9355. https:\/\/doi.org\/10.1109\/CVPR.2019.00957","DOI":"10.1109\/CVPR.2019.00957"},{"key":"e_1_3_2_2_9_1","volume-title":"Proceedings of the 30th ACM International Conference on Information and Knowledge Management (CIKM). 453--463","author":"Kamhoua Barakeel Fanseu","year":"2021","unstructured":"Barakeel Fanseu Kamhoua , Lin Zhang , Kaili Ma , James Cheng , Bo Li , and Bo Han . 2021 . HyperGraph Convolution Based Attributed HyperGraph Clustering . In Proceedings of the 30th ACM International Conference on Information and Knowledge Management (CIKM). 453--463 . https:\/\/doi.org\/10.1145\/3459637.3482437 10.1145\/3459637.3482437 Barakeel Fanseu Kamhoua, Lin Zhang, Kaili Ma, James Cheng, Bo Li, and Bo Han. 2021. HyperGraph Convolution Based Attributed HyperGraph Clustering. In Proceedings of the 30th ACM International Conference on Information and Knowledge Management (CIKM). 453--463. https:\/\/doi.org\/10.1145\/3459637.3482437"},{"key":"e_1_3_2_2_10_1","volume-title":"Tall: Temporal activity localization via language query. In ICCV. 5267--5275. https:\/\/doi.org\/10.1109\/ICCV.2017.563","author":"Gao Jiyang","year":"2017","unstructured":"Jiyang Gao , Chen Sun , Zhenheng Yang , and Ram Nevatia . 2017 . Tall: Temporal activity localization via language query. In ICCV. 5267--5275. https:\/\/doi.org\/10.1109\/ICCV.2017.563 10.1109\/ICCV.2017.563 Jiyang Gao, Chen Sun, Zhenheng Yang, and Ram Nevatia. 2017. Tall: Temporal activity localization via language query. In ICCV. 5267--5275. https:\/\/doi.org\/10.1109\/ICCV.2017.563"},{"key":"#cr-split#-e_1_3_2_2_11_1.1","unstructured":"Soham Ghosh Anuva Agarwal Zarana Parekh and Alexander G Hauptmann. 2019. ExCL: Extractive Clip Localization Using Natural Language Descriptions. In NAACL-HLT. 1984--1990. https:\/\/doi.org\/10.18653\/v1\/N19--1198 10.18653\/v1"},{"key":"#cr-split#-e_1_3_2_2_11_1.2","unstructured":"Soham Ghosh Anuva Agarwal Zarana Parekh and Alexander G Hauptmann. 2019. ExCL: Extractive Clip Localization Using Natural Language Descriptions. In NAACL-HLT. 1984--1990. https:\/\/doi.org\/10.18653\/v1\/N19--1198"},{"key":"e_1_3_2_2_12_1","unstructured":"Ross Girshick. 2015. Fast r-cnn. In ICCV. 1440--1448. https:\/\/doi.org\/10.1109\/ ICCV.2015.169  Ross Girshick. 2015. Fast r-cnn. In ICCV. 1440--1448. https:\/\/doi.org\/10.1109\/ ICCV.2015.169"},{"key":"#cr-split#-e_1_3_2_2_13_1.1","doi-asserted-by":"crossref","unstructured":"Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Deep residual learning for image recognition. In CVPR. 770--778. https:\/\/doi.org\/10.1109\/CVPR.2016.90 10.1109\/CVPR.2016.90","DOI":"10.1109\/CVPR.2016.90"},{"key":"#cr-split#-e_1_3_2_2_13_1.2","doi-asserted-by":"crossref","unstructured":"Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Deep residual learning for image recognition. In CVPR. 770--778. https:\/\/doi.org\/10.1109\/CVPR.2016.90","DOI":"10.1109\/CVPR.2016.90"},{"key":"#cr-split#-e_1_3_2_2_14_1.1","doi-asserted-by":"crossref","unstructured":"Ronghang Hu Marcus Rohrbach Jacob Andreas Trevor Darrell and Kate Saenko. 2017. Modeling relationships in referential expressions with compositional modular networks. In CVPR. 1115--1124. https:\/\/doi.org\/10.1109\/CVPR.2017.470 10.1109\/CVPR.2017.470","DOI":"10.1109\/CVPR.2017.470"},{"key":"#cr-split#-e_1_3_2_2_14_1.2","doi-asserted-by":"crossref","unstructured":"Ronghang Hu Marcus Rohrbach Jacob Andreas Trevor Darrell and Kate Saenko. 2017. Modeling relationships in referential expressions with compositional modular networks. In CVPR. 1115--1124. https:\/\/doi.org\/10.1109\/CVPR.2017.470","DOI":"10.1109\/CVPR.2017.470"},{"key":"e_1_3_2_2_15_1","unstructured":"Drew A Hudson and Christopher D Manning. 2018. Compositional Attention Networks for Machine Reasoning. In ICLR. https:\/\/openreview.net\/forum?id=S1Euwz-Rb  Drew A Hudson and Christopher D Manning. 2018. Compositional Attention Networks for Machine Reasoning. In ICLR. https:\/\/openreview.net\/forum?id=S1Euwz-Rb"},{"key":"e_1_3_2_2_16_1","unstructured":"Jin-Hwa Kim Kyoung-Woon On Woosang Lim Jeonghee Kim Jung-Woo Ha and Byoung-Tak Zhang. 2017. Hadamard product for low-rank bilinear pooling. In ICLR. https:\/\/openreview.net\/forum?id=r1rhWnZkg  Jin-Hwa Kim Kyoung-Woon On Woosang Lim Jeonghee Kim Jung-Woo Ha and Byoung-Tak Zhang. 2017. Hadamard product for low-rank bilinear pooling. In ICLR. https:\/\/openreview.net\/forum?id=r1rhWnZkg"},{"key":"#cr-split#-e_1_3_2_2_17_1.1","doi-asserted-by":"crossref","unstructured":"Sunoh Kim Kimin Yun and Jin Young Choi. 2021. Position-aware Location Regression Network for Temporal Video Grounding. In AVSS. 1--8. https:\/\/doi.org\/10.1109\/AVSS52988.2021.9663815 10.1109\/AVSS52988.2021.9663815","DOI":"10.1109\/AVSS52988.2021.9663815"},{"key":"#cr-split#-e_1_3_2_2_17_1.2","doi-asserted-by":"crossref","unstructured":"Sunoh Kim Kimin Yun and Jin Young Choi. 2021. Position-aware Location Regression Network for Temporal Video Grounding. In AVSS. 1--8. https:\/\/doi.org\/10.1109\/AVSS52988.2021.9663815","DOI":"10.1109\/AVSS52988.2021.9663815"},{"key":"e_1_3_2_2_18_1","volume-title":"Skeletonbased action recognition of people handling objects","author":"Kim Sunoh","year":"2019","unstructured":"Sunoh Kim , Kimin Yun , Jongyoul Park , and Jin Young Choi . 2019. Skeletonbased action recognition of people handling objects . In WACV. IEEE , 61--70. https:\/\/doi.org\/10.1109\/WACV. 2019 .00014 10.1109\/WACV.2019.00014 Sunoh Kim, Kimin Yun, Jongyoul Park, and Jin Young Choi. 2019. Skeletonbased action recognition of people handling objects. In WACV. IEEE, 61--70. https:\/\/doi.org\/10.1109\/WACV.2019.00014"},{"key":"e_1_3_2_2_19_1","volume-title":"Adam: A method for stochastic optimization. In ICLR. https:\/\/arxiv.org\/abs\/1412.6980","author":"Kingma Diederik P","year":"2015","unstructured":"Diederik P Kingma and Jimmy Ba . 2015 . Adam: A method for stochastic optimization. In ICLR. https:\/\/arxiv.org\/abs\/1412.6980 Diederik P Kingma and Jimmy Ba. 2015. Adam: A method for stochastic optimization. In ICLR. https:\/\/arxiv.org\/abs\/1412.6980"},{"key":"e_1_3_2_2_20_1","unstructured":"Thomas N Kipf and MaxWelling. 2017. Semi-supervised classification with graph convolutional networks. In ICLR. https:\/\/openreview.net\/forum?id=SJU4ayYgl  Thomas N Kipf and MaxWelling. 2017. Semi-supervised classification with graph convolutional networks. In ICLR. https:\/\/openreview.net\/forum?id=SJU4ayYgl"},{"key":"#cr-split#-e_1_3_2_2_21_1.1","doi-asserted-by":"crossref","unstructured":"Ranjay Krishna Kenji Hata Frederic Ren Li Fei-Fei and Juan Carlos Niebles. 2017. Dense-captioning events in videos. In ICCV. 706--715. https:\/\/doi.org\/10.1109\/ICCV.2017.83 10.1109\/ICCV.2017.83","DOI":"10.1109\/ICCV.2017.83"},{"key":"#cr-split#-e_1_3_2_2_21_1.2","doi-asserted-by":"crossref","unstructured":"Ranjay Krishna Kenji Hata Frederic Ren Li Fei-Fei and Juan Carlos Niebles. 2017. Dense-captioning events in videos. In ICCV. 706--715. https:\/\/doi.org\/10.1109\/ICCV.2017.83","DOI":"10.1109\/ICCV.2017.83"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"crossref","unstructured":"Kun Li Dan Guo and Meng Wang. 2021. Proposal-Free Video Grounding with Contextual Pyramid Network. In AAAI. 1902--1910. https:\/\/ojs.aaai.org\/index. php\/AAAI\/article\/view\/16285  Kun Li Dan Guo and Meng Wang. 2021. Proposal-Free Video Grounding with Contextual Pyramid Network. In AAAI. 1902--1910. https:\/\/ojs.aaai.org\/index. php\/AAAI\/article\/view\/16285","DOI":"10.1609\/aaai.v35i3.16285"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3459637.3482417"},{"key":"e_1_3_2_2_24_1","volume-title":"Steven Bethard, and David McClosky.","author":"Manning Christopher D","year":"2014","unstructured":"Christopher D Manning , Mihai Surdeanu , John Bauer , Jenny Rose Finkel , Steven Bethard, and David McClosky. 2014 . The Stanford CoreNLP natural language processing toolkit. In ACL. 55--60. https:\/\/doi.org\/10.3115\/v1\/P14--5010 10.3115\/v1 Christopher D Manning, Mihai Surdeanu, John Bauer, Jenny Rose Finkel, Steven Bethard, and David McClosky. 2014. The Stanford CoreNLP natural language processing toolkit. In ACL. 55--60. https:\/\/doi.org\/10.3115\/v1\/P14--5010"},{"key":"#cr-split#-e_1_3_2_2_25_1.1","doi-asserted-by":"crossref","unstructured":"Jonghwan Mun Minsu Cho and Bohyung Han. 2020. Local-global video-text interactions for temporal grounding. In CVPR. 10810--10819. https:\/\/doi.org\/10.1109\/CVPR42600.2020.01082 10.1109\/CVPR42600.2020.01082","DOI":"10.1109\/CVPR42600.2020.01082"},{"key":"#cr-split#-e_1_3_2_2_25_1.2","doi-asserted-by":"crossref","unstructured":"Jonghwan Mun Minsu Cho and Bohyung Han. 2020. Local-global video-text interactions for temporal grounding. In CVPR. 10810--10819. https:\/\/doi.org\/10.1109\/CVPR42600.2020.01082","DOI":"10.1109\/CVPR42600.2020.01082"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3340531.3412014"},{"key":"#cr-split#-e_1_3_2_2_27_1.1","unstructured":"Jeffrey Pennington Richard Socher and Christopher Manning. 2014. GloVe: Global Vectors for Word Representation. In EMNLP. 1532--1543. https:\/\/doi.org\/10.3115\/v1\/D14--1162 10.3115\/v1"},{"key":"#cr-split#-e_1_3_2_2_27_1.2","doi-asserted-by":"crossref","unstructured":"Jeffrey Pennington Richard Socher and Christopher Manning. 2014. GloVe: Global Vectors for Word Representation. In EMNLP. 1532--1543. https:\/\/doi.org\/10.3115\/v1\/D14--1162","DOI":"10.3115\/v1\/D14-1162"},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3357384.3358010"},{"key":"e_1_3_2_2_29_1","volume-title":"VLG-Net: Video-Language Graph Matching Network for Video Grounding. In ICCV workshop. https:\/\/doi.org\/10","author":"Qu Sisi","year":"2021","unstructured":"Sisi Qu , Mattia Soldan , Mengmeng Xu , Jesper Tegner , and Bernard Ghanem . 2021 . VLG-Net: Video-Language Graph Matching Network for Video Grounding. In ICCV workshop. https:\/\/doi.org\/10 .1109\/ICCVW54120.2021.00361 10.1109\/ICCVW54120.2021.00361 Sisi Qu, Mattia Soldan, Mengmeng Xu, Jesper Tegner, and Bernard Ghanem. 2021. VLG-Net: Video-Language Graph Matching Network for Video Grounding. In ICCV workshop. https:\/\/doi.org\/10.1109\/ICCVW54120.2021.00361"},{"key":"e_1_3_2_2_30_1","volume-title":"NeurIPS","volume":"28","author":"Ren Shaoqing","year":"2015","unstructured":"Shaoqing Ren , Kaiming He , Ross Girshick , and Jian Sun . 2015 . Faster R-CNN: Towards Real-Time Object Detection with Region Proposal Networks . In NeurIPS , Vol. 28 . https:\/\/arxiv.org\/abs\/1506.01497 Shaoqing Ren, Kaiming He, Ross Girshick, and Jian Sun. 2015. Faster R-CNN: Towards Real-Time Object Detection with Region Proposal Networks. In NeurIPS, Vol. 28. https:\/\/arxiv.org\/abs\/1506.01497"},{"key":"#cr-split#-e_1_3_2_2_31_1.1","doi-asserted-by":"crossref","unstructured":"Cristian Rodriguez-Opazo Edison Marrese-Taylor Basura Fernando Hongdong Li and Stephen Gould. 2021. DORi: Discovering Object Relationships for Moment Localization of a Natural Language Query in a Video. In WACV. 1079--1088. https:\/\/doi.org\/10.1109\/WACV48630.2021.00112 10.1109\/WACV48630.2021.00112","DOI":"10.1109\/WACV48630.2021.00112"},{"key":"#cr-split#-e_1_3_2_2_31_1.2","doi-asserted-by":"crossref","unstructured":"Cristian Rodriguez-Opazo Edison Marrese-Taylor Basura Fernando Hongdong Li and Stephen Gould. 2021. DORi: Discovering Object Relationships for Moment Localization of a Natural Language Query in a Video. In WACV. 1079--1088. https:\/\/doi.org\/10.1109\/WACV48630.2021.00112","DOI":"10.1109\/WACV48630.2021.00112"},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/78.650093"},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3459637.3482264"},{"volume-title":"Hollywood in homes: Crowdsourcing data collection for activity understanding","author":"Sigurdsson Gunnar A","key":"e_1_3_2_2_34_1","unstructured":"Gunnar A Sigurdsson , G\u00fcl Varol , Xiaolong Wang , Ali Farhadi , Ivan Laptev , and Abhinav Gupta . 2016. Hollywood in homes: Crowdsourcing data collection for activity understanding . In ECCV. Springer , 510--526. https:\/\/doi.org\/10.1007\/978--3--319--46448-0_31 10.1007\/978--3--319--46448-0_31 Gunnar A Sigurdsson, G\u00fcl Varol, Xiaolong Wang, Ali Farhadi, Ivan Laptev, and Abhinav Gupta. 2016. Hollywood in homes: Crowdsourcing data collection for activity understanding. In ECCV. Springer, 510--526. https:\/\/doi.org\/10.1007\/978--3--319--46448-0_31"},{"key":"e_1_3_2_2_35_1","volume-title":"Proceedings of the 29th ACM International Conference on Information and Knowledge Management (CIKM). 1435--1444","author":"Tang Xianfeng","year":"2020","unstructured":"Xianfeng Tang , Huaxiu Yao , Yiwei Sun , YiqiWang, Jiliang Tang , Charu Aggarwal , Prasenjit Mitra , and Suhang Wang . 2020 . Investigating and mitigating degreerelated biases in graph convoltuional networks . In Proceedings of the 29th ACM International Conference on Information and Knowledge Management (CIKM). 1435--1444 . https:\/\/doi.org\/10.1145\/3340531.3411872 10.1145\/3340531.3411872 Xianfeng Tang, Huaxiu Yao, Yiwei Sun, YiqiWang, Jiliang Tang, Charu Aggarwal, Prasenjit Mitra, and Suhang Wang. 2020. Investigating and mitigating degreerelated biases in graph convoltuional networks. In Proceedings of the 29th ACM International Conference on Information and Knowledge Management (CIKM). 1435--1444. https:\/\/doi.org\/10.1145\/3340531.3411872"},{"key":"#cr-split#-e_1_3_2_2_36_1.1","doi-asserted-by":"crossref","unstructured":"Du Tran Lubomir Bourdev Rob Fergus Lorenzo Torresani and Manohar Paluri. 2015. Learning spatiotemporal features with 3d convolutional networks. In ICCV. 4489--4497. https:\/\/doi.org\/10.1109\/ICCV.2015.510 10.1109\/ICCV.2015.510","DOI":"10.1109\/ICCV.2015.510"},{"key":"#cr-split#-e_1_3_2_2_36_1.2","doi-asserted-by":"crossref","unstructured":"Du Tran Lubomir Bourdev Rob Fergus Lorenzo Torresani and Manohar Paluri. 2015. Learning spatiotemporal features with 3d convolutional networks. In ICCV. 4489--4497. https:\/\/doi.org\/10.1109\/ICCV.2015.510","DOI":"10.1109\/ICCV.2015.510"},{"key":"e_1_3_2_2_37_1","volume-title":"andWenhao Jiang","author":"Ma Lin","year":"2020","unstructured":"JingwenWang, Lin Ma , andWenhao Jiang . 2020 . Temporally grounding language queries in videos by contextual boundary-aware prediction. In AAAI. https:\/\/doi.org\/10.1609\/AAAI.V34I07.6897 10.1609\/AAAI.V34I07.6897 JingwenWang, Lin Ma, andWenhao Jiang. 2020. Temporally grounding language queries in videos by contextual boundary-aware prediction. In AAAI. https:\/\/doi.org\/10.1609\/AAAI.V34I07.6897"},{"key":"#cr-split#-e_1_3_2_2_38_1.1","doi-asserted-by":"crossref","unstructured":"Lei Wang Yuchun Huang Yaolin Hou Shenman Zhang and Jie Shan. 2019. Graph attention convolution for point cloud semantic segmentation. In CVPR. 10296--10305. https:\/\/doi.org\/10.1109\/CVPR.2019.01054 10.1109\/CVPR.2019.01054","DOI":"10.1109\/CVPR.2019.01054"},{"key":"#cr-split#-e_1_3_2_2_38_1.2","doi-asserted-by":"crossref","unstructured":"Lei Wang Yuchun Huang Yaolin Hou Shenman Zhang and Jie Shan. 2019. Graph attention convolution for point cloud semantic segmentation. In CVPR. 10296--10305. https:\/\/doi.org\/10.1109\/CVPR.2019.01054","DOI":"10.1109\/CVPR.2019.01054"},{"key":"#cr-split#-e_1_3_2_2_39_1.1","doi-asserted-by":"crossref","unstructured":"XiaolongWang Ross Girshick Abhinav Gupta and Kaiming He. 2018. Non-local neural networks. In CVPR. 7794--7803. https:\/\/doi.org\/10.1109\/CVPR.2018.00813 10.1109\/CVPR.2018.00813","DOI":"10.1109\/CVPR.2018.00813"},{"key":"#cr-split#-e_1_3_2_2_39_1.2","doi-asserted-by":"crossref","unstructured":"XiaolongWang Ross Girshick Abhinav Gupta and Kaiming He. 2018. Non-local neural networks. In CVPR. 7794--7803. https:\/\/doi.org\/10.1109\/CVPR.2018.00813","DOI":"10.1109\/CVPR.2018.00813"},{"key":"#cr-split#-e_1_3_2_2_40_1.1","unstructured":"Xiaolong Wang and Abhinav Gupta. 2018. Videos as space-time region graphs. In ECCV. 399--417. https:\/\/doi.org\/10.1007\/978--3-030-01228--1_25 10.1007\/978--3-030-01228--1_25"},{"key":"#cr-split#-e_1_3_2_2_40_1.2","unstructured":"Xiaolong Wang and Abhinav Gupta. 2018. Videos as space-time region graphs. In ECCV. 399--417. https:\/\/doi.org\/10.1007\/978--3-030-01228--1_25"},{"key":"e_1_3_2_2_41_1","volume-title":"Proceedings of the 28th ACM International Conference on Information and Knowledge Management (CIKM). 159--168","author":"Zhang Junsan","year":"2019","unstructured":"XiaominWang, Junsan Zhang , LeiquanWang, Philip S Yu , Jie Zhu , and Haisheng Li . 2019 . Video-level Multi-model Fusion for Action Recognition . In Proceedings of the 28th ACM International Conference on Information and Knowledge Management (CIKM). 159--168 . https:\/\/doi.org\/10.1145\/3357384.3357935 10.1145\/3357384.3357935 XiaominWang, Junsan Zhang, LeiquanWang, Philip S Yu, Jie Zhu, and Haisheng Li. 2019. Video-level Multi-model Fusion for Action Recognition. In Proceedings of the 28th ACM International Conference on Information and Knowledge Management (CIKM). 159--168. https:\/\/doi.org\/10.1145\/3357384.3357935"},{"key":"e_1_3_2_2_42_1","volume-title":"Dynamic graph cnn for learning on point clouds. ACM Transactions On Graphics (tog) 38, 5","author":"Wang Yue","year":"2019","unstructured":"Yue Wang , Yongbin Sun , Ziwei Liu , Sanjay E Sarma , Michael M Bronstein , and Justin M Solomon . 2019. Dynamic graph cnn for learning on point clouds. ACM Transactions On Graphics (tog) 38, 5 ( 2019 ), 1--12. https:\/\/doi.org\/10.1145\/3326362 10.1145\/3326362 Yue Wang, Yongbin Sun, Ziwei Liu, Sanjay E Sarma, Michael M Bronstein, and Justin M Solomon. 2019. Dynamic graph cnn for learning on point clouds. ACM Transactions On Graphics (tog) 38, 5 (2019), 1--12. https:\/\/doi.org\/10.1145\/3326362"},{"key":"#cr-split#-e_1_3_2_2_43_1.1","doi-asserted-by":"crossref","unstructured":"Jie Wu Guanbin Li Si Liu and Liang Lin. 2020. Tree-structured policy based progressive reinforcement learning for temporally language grounding in video. In AAAI. 12386--12393. https:\/\/doi.org\/10.1609\/aaai.v34i07.6924 10.1609\/aaai.v34i07.6924","DOI":"10.1609\/aaai.v34i07.6924"},{"key":"#cr-split#-e_1_3_2_2_43_1.2","doi-asserted-by":"crossref","unstructured":"Jie Wu Guanbin Li Si Liu and Liang Lin. 2020. Tree-structured policy based progressive reinforcement learning for temporally language grounding in video. In AAAI. 12386--12393. https:\/\/doi.org\/10.1609\/aaai.v34i07.6924","DOI":"10.1609\/aaai.v34i07.6924"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/3459637.3482354"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"crossref","unstructured":"Shaoning Xiao Long Chen Songyang Zhang Wei Ji Jian Shao Lu Ye and Jun Xiao. 2021. Boundary Proposal Network for Two-Stage Natural Language Video Localization. In AAAI. 2986--2994. https:\/\/ojs.aaai.org\/index.php\/AAAI\/article\/view\/16406  Shaoning Xiao Long Chen Songyang Zhang Wei Ji Jian Shao Lu Ye and Jun Xiao. 2021. Boundary Proposal Network for Two-Stage Natural Language Video Localization. In AAAI. 2986--2994. https:\/\/ojs.aaai.org\/index.php\/AAAI\/article\/view\/16406","DOI":"10.1609\/aaai.v35i4.16406"},{"key":"#cr-split#-e_1_3_2_2_46_1.1","doi-asserted-by":"crossref","unstructured":"Saining Xie Ross Girshick Piotr Doll\u00e1r Zhuowen Tu and Kaiming He. 2017. Aggregated residual transformations for deep neural networks. In CVPR. 1492--1500. https:\/\/doi.org\/10.1109\/CVPR.2017.634 10.1109\/CVPR.2017.634","DOI":"10.1109\/CVPR.2017.634"},{"key":"#cr-split#-e_1_3_2_2_46_1.2","doi-asserted-by":"crossref","unstructured":"Saining Xie Ross Girshick Piotr Doll\u00e1r Zhuowen Tu and Kaiming He. 2017. Aggregated residual transformations for deep neural networks. In CVPR. 1492--1500. https:\/\/doi.org\/10.1109\/CVPR.2017.634","DOI":"10.1109\/CVPR.2017.634"},{"key":"#cr-split#-e_1_3_2_2_47_1.1","doi-asserted-by":"crossref","unstructured":"Huijuan Xu Kun He Bryan A Plummer Leonid Sigal Stan Sclaroff and Kate Saenko. 2019. Multilevel language and vision integration for text-to-clip retrieval. In AAAI. 9062--9069. https:\/\/doi.org\/10.1609\/aaai.v33i01.33019062 10.1609\/aaai.v33i01.33019062","DOI":"10.1609\/aaai.v33i01.33019062"},{"key":"#cr-split#-e_1_3_2_2_47_1.2","doi-asserted-by":"crossref","unstructured":"Huijuan Xu Kun He Bryan A Plummer Leonid Sigal Stan Sclaroff and Kate Saenko. 2019. Multilevel language and vision integration for text-to-clip retrieval. In AAAI. 9062--9069. https:\/\/doi.org\/10.1609\/aaai.v33i01.33019062","DOI":"10.1609\/aaai.v33i01.33019062"},{"key":"e_1_3_2_2_48_1","volume-title":"G-tad: Sub-graph localization for temporal action detection. In CVPR. 10156--10165. https:\/\/doi.org\/10.1109\/cvpr42600.2020.01017","author":"Xu Mengmeng","year":"2020","unstructured":"Mengmeng Xu , Chen Zhao , David S Rojas , Ali Thabet , and Bernard Ghanem . 2020 . G-tad: Sub-graph localization for temporal action detection. In CVPR. 10156--10165. https:\/\/doi.org\/10.1109\/cvpr42600.2020.01017 10.1109\/cvpr42600.2020.01017 Mengmeng Xu, Chen Zhao, David S Rojas, Ali Thabet, and Bernard Ghanem. 2020. G-tad: Sub-graph localization for temporal action detection. In CVPR. 10156--10165. https:\/\/doi.org\/10.1109\/cvpr42600.2020.01017"},{"key":"e_1_3_2_2_49_1","unstructured":"Sibei Yang Guanbin Li and Yizhou Yu. 2019. Dynamic graph attention for referring expression comprehension. In ICCV. 4644--4653. https:\/\/doi.org\/10. 1109\/ICCV.2019.00474  Sibei Yang Guanbin Li and Yizhou Yu. 2019. Dynamic graph attention for referring expression comprehension. In ICCV. 4644--4653. https:\/\/doi.org\/10. 1109\/ICCV.2019.00474"},{"key":"#cr-split#-e_1_3_2_2_50_1.1","doi-asserted-by":"crossref","unstructured":"Xun Yang Fuli Feng Wei Ji MengWang and Tat-Seng Chua. 2021. Deconfounded Video Moment Retrieval with Causal Intervention. In ACM SIGIR. https:\/\/doi.org\/10.1145\/3404835.3462823 10.1145\/3404835.3462823","DOI":"10.1145\/3404835.3462823"},{"key":"#cr-split#-e_1_3_2_2_50_1.2","doi-asserted-by":"crossref","unstructured":"Xun Yang Fuli Feng Wei Ji MengWang and Tat-Seng Chua. 2021. Deconfounded Video Moment Retrieval with Causal Intervention. In ACM SIGIR. https:\/\/doi.org\/10.1145\/3404835.3462823","DOI":"10.1145\/3404835.3462823"},{"key":"e_1_3_2_2_51_1","volume-title":"Mattnet: Modular attention network for referring expression comprehension. In CVPR. 1307--1315. https:\/\/doi.org\/10.1109\/CVPR.2018.00142","author":"Yu Licheng","year":"2018","unstructured":"Licheng Yu , Zhe Lin , Xiaohui Shen , Jimei Yang , Xin Lu , Mohit Bansal , and Tamara L Berg . 2018 . Mattnet: Modular attention network for referring expression comprehension. In CVPR. 1307--1315. https:\/\/doi.org\/10.1109\/CVPR.2018.00142 10.1109\/CVPR.2018.00142 Licheng Yu, Zhe Lin, Xiaohui Shen, Jimei Yang, Xin Lu, Mohit Bansal, and Tamara L Berg. 2018. Mattnet: Modular attention network for referring expression comprehension. In CVPR. 1307--1315. https:\/\/doi.org\/10.1109\/CVPR.2018.00142"},{"key":"#cr-split#-e_1_3_2_2_52_1.1","doi-asserted-by":"crossref","unstructured":"Yitian Yuan Tao Mei and Wenwu Zhu. 2019. To find where you talk: Temporal sentence localization in video with attention based location regression. In AAAI. 9159--9166. https:\/\/doi.org\/10.1609\/aaai.v33i01.33019159 10.1609\/aaai.v33i01.33019159","DOI":"10.1609\/aaai.v33i01.33019159"},{"key":"#cr-split#-e_1_3_2_2_52_1.2","doi-asserted-by":"crossref","unstructured":"Yitian Yuan Tao Mei and Wenwu Zhu. 2019. To find where you talk: Temporal sentence localization in video with attention based location regression. In AAAI. 9159--9166. https:\/\/doi.org\/10.1609\/aaai.v33i01.33019159","DOI":"10.1609\/aaai.v33i01.33019159"},{"key":"e_1_3_2_2_53_1","volume-title":"Vision-based garbage dumping action detection for real-world surveillance platform. ETRI Journal 39 (July","author":"Yun Kimin","year":"2019","unstructured":"Kimin Yun , Yongjin Kwon , Sungchan Oh , Jinyoung Moon , and Jongyoul Park . 2019. Vision-based garbage dumping action detection for real-world surveillance platform. ETRI Journal 39 (July 2019 ), 21--12. https:\/\/doi.org\/10.4218\/etrij.2018-0520 10.4218\/etrij.2018-0520 Kimin Yun, Yongjin Kwon, Sungchan Oh, Jinyoung Moon, and Jongyoul Park. 2019. Vision-based garbage dumping action detection for real-world surveillance platform. ETRI Journal 39 (July 2019), 21--12. https:\/\/doi.org\/10.4218\/etrij.2018-0520"},{"key":"#cr-split#-e_1_3_2_2_54_1.1","doi-asserted-by":"crossref","unstructured":"Runhao Zeng Haoming Xu Wenbing Huang Peihao Chen Mingkui Tan and Chuang Gan. 2020. Dense regression network for video grounding. In CVPR. 10287--10296. https:\/\/doi.org\/10.1109\/cvpr42600.2020.01030 10.1109\/cvpr42600.2020.01030","DOI":"10.1109\/CVPR42600.2020.01030"},{"key":"#cr-split#-e_1_3_2_2_54_1.2","doi-asserted-by":"crossref","unstructured":"Runhao Zeng Haoming Xu Wenbing Huang Peihao Chen Mingkui Tan and Chuang Gan. 2020. Dense regression network for video grounding. In CVPR. 10287--10296. https:\/\/doi.org\/10.1109\/cvpr42600.2020.01030","DOI":"10.1109\/CVPR42600.2020.01030"},{"key":"e_1_3_2_2_55_1","volume-title":"Man: Moment alignment network for natural language moment retrieval via iterative graph adjustment. In CVPR. 1247--1257. https:\/\/doi.org\/10.1109\/CVPR.2019.00134","author":"Zhang Da","year":"2019","unstructured":"Da Zhang , Xiyang Dai , XinWang, Yuan-FangWang, and Larry S Davis . 2019 . Man: Moment alignment network for natural language moment retrieval via iterative graph adjustment. In CVPR. 1247--1257. https:\/\/doi.org\/10.1109\/CVPR.2019.00134 10.1109\/CVPR.2019.00134 Da Zhang, Xiyang Dai, XinWang, Yuan-FangWang, and Larry S Davis. 2019. Man: Moment alignment network for natural language moment retrieval via iterative graph adjustment. In CVPR. 1247--1257. https:\/\/doi.org\/10.1109\/CVPR.2019.00134"},{"key":"#cr-split#-e_1_3_2_2_56_1.1","doi-asserted-by":"crossref","unstructured":"Songyang Zhang Houwen Peng Jianlong Fu and Jiebo Luo. 2020. Learning 2d temporal adjacent networks for moment localization with natural language. In AAAI. 12870--12877. https:\/\/doi.org\/10.1609\/aaai.v34i07.6984 10.1609\/aaai.v34i07.6984","DOI":"10.1609\/aaai.v34i07.6984"},{"key":"#cr-split#-e_1_3_2_2_56_1.2","doi-asserted-by":"crossref","unstructured":"Songyang Zhang Houwen Peng Jianlong Fu and Jiebo Luo. 2020. Learning 2d temporal adjacent networks for moment localization with natural language. In AAAI. 12870--12877. https:\/\/doi.org\/10.1609\/aaai.v34i07.6984","DOI":"10.1609\/aaai.v34i07.6984"},{"key":"#cr-split#-e_1_3_2_2_57_1.1","doi-asserted-by":"crossref","unstructured":"Zhu Zhang Zhijie Lin Zhou Zhao and Zhenxin Xiao. 2019. Cross-modal interaction networks for query-based moment retrieval in videos. In ACM SIGIR. 655--664. https:\/\/doi.org\/10.1145\/3331184.3331235 10.1145\/3331184.3331235","DOI":"10.1145\/3331184.3331235"},{"key":"#cr-split#-e_1_3_2_2_57_1.2","doi-asserted-by":"crossref","unstructured":"Zhu Zhang Zhijie Lin Zhou Zhao and Zhenxin Xiao. 2019. Cross-modal interaction networks for query-based moment retrieval in videos. In ACM SIGIR. 655--664. https:\/\/doi.org\/10.1145\/3331184.3331235","DOI":"10.1145\/3331184.3331235"},{"key":"#cr-split#-e_1_3_2_2_58_1.1","doi-asserted-by":"crossref","unstructured":"Yang Zhao Zhou Zhao Zhu Zhang and Zhijie Lin. 2021. Cascaded Prediction Network via Segment Tree for Temporal Video Grounding. In CVPR. 4197--4206. https:\/\/doi.org\/10.1109\/CVPR46437.2021.00418 10.1109\/CVPR46437.2021.00418","DOI":"10.1109\/CVPR46437.2021.00418"},{"key":"#cr-split#-e_1_3_2_2_58_1.2","doi-asserted-by":"crossref","unstructured":"Yang Zhao Zhou Zhao Zhu Zhang and Zhijie Lin. 2021. Cascaded Prediction Network via Segment Tree for Temporal Video Grounding. In CVPR. 4197--4206. https:\/\/doi.org\/10.1109\/CVPR46437.2021.00418","DOI":"10.1109\/CVPR46437.2021.00418"},{"key":"#cr-split#-e_1_3_2_2_59_1.1","doi-asserted-by":"crossref","unstructured":"Fengda Zhu Yi Zhu Xiaojun Chang and Xiaodan Liang. 2020. Vision-language navigation with self-supervised auxiliary reasoning tasks. In CVPR. 10012--10022. https:\/\/doi.org\/10.1109\/CVPR42600.2020.01003 10.1109\/CVPR42600.2020.01003","DOI":"10.1109\/CVPR42600.2020.01003"},{"key":"#cr-split#-e_1_3_2_2_59_1.2","doi-asserted-by":"crossref","unstructured":"Fengda Zhu Yi Zhu Xiaojun Chang and Xiaodan Liang. 2020. Vision-language navigation with self-supervised auxiliary reasoning tasks. In CVPR. 10012--10022. https:\/\/doi.org\/10.1109\/CVPR42600.2020.01003","DOI":"10.1109\/CVPR42600.2020.01003"}],"event":{"name":"CIKM '22: The 31st ACM International Conference on Information and Knowledge Management","sponsor":["SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web","SIGIR ACM Special Interest Group on Information Retrieval"],"location":"Atlanta GA USA","acronym":"CIKM '22"},"container-title":["Proceedings of the 31st ACM International Conference on Information &amp; Knowledge Management"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3511808.3557463","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3511808.3557463","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T17:48:55Z","timestamp":1750182535000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3511808.3557463"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,10,17]]},"references-count":85,"alternative-id":["10.1145\/3511808.3557463","10.1145\/3511808"],"URL":"https:\/\/doi.org\/10.1145\/3511808.3557463","relation":{},"subject":[],"published":{"date-parts":[[2022,10,17]]},"assertion":[{"value":"2022-10-17","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}