{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T14:51:50Z","timestamp":1784904710613,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":44,"publisher":"ACM","license":[{"start":{"date-parts":[[2020,10,12]],"date-time":"2020-10-12T00:00:00Z","timestamp":1602460800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Fundamental Research Funds for the Central Universities","award":["D2191240"],"award-info":[{"award-number":["D2191240"]}]},{"name":"Key-Area Research and Development Program of Guangdong Province","award":["2019B010155002; 2018B010108002"],"award-info":[{"award-number":["2019B010155002; 2018B010108002"]}]},{"name":"National Natural Science Foundation of China (NSFC)","award":["61876208; 61836003"],"award-info":[{"award-number":["61876208; 61836003"]}]},{"name":"Guangdong","award":["2017ZT07X183"],"award-info":[{"award-number":["2017ZT07X183"]}]},{"name":"Pearl River S&T Nova Program of Guangzhou","award":["201806010081"],"award-info":[{"award-number":["201806010081"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2020,10,12]]},"DOI":"10.1145\/3394171.3413581","type":"proceedings-article","created":{"date-parts":[[2020,10,12]],"date-time":"2020-10-12T12:27:35Z","timestamp":1602505655000},"page":"3893-3901","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":91,"title":["Cross-Modal Relation-Aware Networks for Audio-Visual Event Localization"],"prefix":"10.1145","author":[{"given":"Haoming","family":"Xu","sequence":"first","affiliation":[{"name":"South China University of Technology, Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Runhao","family":"Zeng","sequence":"additional","affiliation":[{"name":"South China University of Technology, Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qingyao","family":"Wu","sequence":"additional","affiliation":[{"name":"South China University of Technology, Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mingkui","family":"Tan","sequence":"additional","affiliation":[{"name":"South China University of Technology, Guangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chuang","family":"Gan","sequence":"additional","affiliation":[{"name":"MIT-IBM Watson AI Lab, Cambridge, MA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2020,10,12]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.73"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.667"},{"key":"e_1_3_2_2_3_1","volume-title":"Relation Attention for Temporal Action Localization","author":"Chen Peihao","year":"2019","unstructured":"Peihao Chen , Chuang Gan , Guangyao Shen , Wenbing Huang , Runhao Zeng , and Mingkui Tan . 2019. Relation Attention for Temporal Action Localization . IEEE Transactions on Multimedia (TMM) ( 2019 ). Peihao Chen, Chuang Gan, Guangyao Shen, Wenbing Huang, Runhao Zeng, and Mingkui Tan. 2019. Relation Attention for Temporal Action Localization. IEEE Transactions on Multimedia (TMM) (2019)."},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01059"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1285"},{"key":"e_1_3_2_2_6_1","volume-title":"Proceedings of the Conference of the North American Chapter of the Association for Computational Linguistics (NAACL). 4171--4186","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin , Ming-Wei Chang , Kenton Lee , and Kristina Toutanova . 2019 . BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding . In Proceedings of the Conference of the North American Chapter of the Association for Computational Linguistics (NAACL). 4171--4186 . Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In Proceedings of the Conference of the North American Chapter of the Association for Computational Linguistics (NAACL). 4171--4186."},{"key":"e_1_3_2_2_7_1","volume-title":"Proceedings of the ACM International Conference on Multimedia (ACM MM). 25--31","author":"Fontanellaz Matthias","unstructured":"Matthias Fontanellaz , Stergios Christodoulidis , and Stavroula G. Mougiakakou . 2019. Self-Attention and Ingredient-Attention Based Model for Recipe Retrieval from Image Queries . In Proceedings of the ACM International Conference on Multimedia (ACM MM). 25--31 . Matthias Fontanellaz, Stergios Christodoulidis, and Stavroula G. Mougiakakou. 2019. Self-Attention and Ingredient-Attention Based Model for Recipe Retrieval from Image Queries. In Proceedings of the ACM International Conference on Multimedia (ACM MM). 25--31."},{"key":"e_1_3_2_2_8_1","volume-title":"Proceedings of the European Conference on Computer Vision (ECCV).","author":"Gan Chuang","year":"2020","unstructured":"Chuang Gan , Deng Huang , Peihao Chen , Joshua B. Tenenbaum , and Antonio Torralba . 2020 a. Foley Music: Learning to Generate Music from Videos . In Proceedings of the European Conference on Computer Vision (ECCV). Chuang Gan, Deng Huang, Peihao Chen, Joshua B. Tenenbaum, and Antonio Torralba. 2020 a. Foley Music: Learning to Generate Music from Videos. In Proceedings of the European Conference on Computer Vision (ECCV)."},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01049"},{"key":"e_1_3_2_2_10_1","volume-title":"Devnet: A deep event network for multimedia event detection and evidence recounting. In CVPR. 2568--2577.","author":"Gan Chuang","year":"2015","unstructured":"Chuang Gan , Naiyan Wang , Yi Yang , Dit-Yan Yeung , and Alex G Hauptmann . 2015 . Devnet: A deep event network for multimedia event detection and evidence recounting. In CVPR. 2568--2577. Chuang Gan, Naiyan Wang, Yi Yang, Dit-Yan Yeung, and Alex G Hauptmann. 2015. Devnet: A deep event network for multimedia event detection and evidence recounting. In CVPR. 2568--2577."},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00715"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01047"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952261"},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952132"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00378"},{"key":"e_1_3_2_2_17_1","volume-title":"Squeeze-and-Excitation Networks. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR). 7132--7141","author":"Hu Jie","year":"2018","unstructured":"Jie Hu , Li Shen , and Gang Sun . 2018 b. Squeeze-and-Excitation Networks. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR). 7132--7141 . Jie Hu, Li Shen, and Gang Sun. 2018b. Squeeze-and-Excitation Networks. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR). 7132--7141."},{"key":"e_1_3_2_2_18_1","volume-title":"Location-aware Graph Convolutional Networks for Video Question Answering. In The AAAI Conference on Artificial Intelligence (AAAI).","author":"Huang Deng","year":"2020","unstructured":"Deng Huang , Peihao Chen , Runhao Zeng , Qing Du , Mingkui Tan , and Chuang Gan . 2020 . Location-aware Graph Convolutional Networks for Video Question Answering. In The AAAI Conference on Artificial Intelligence (AAAI). Deng Huang, Peihao Chen, Runhao Zeng, Qing Du, Mingkui Tan, and Chuang Gan. 2020. Location-aware Graph Convolutional Networks for Video Question Answering. In The AAAI Conference on Artificial Intelligence (AAAI)."},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1209"},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00559"},{"key":"e_1_3_2_2_21_1","volume-title":"Proceedings of the International Conference on Learning Representations (ICLR).","author":"Diederik","unstructured":"Diederik P. Kingma and Jimmy Ba. 2015. Adam: A Method for Stochastic Optimization . In Proceedings of the International Conference on Learning Representations (ICLR). Diederik P. Kingma and Jimmy Ba. 2015. Adam: A Method for Stochastic Optimization. In Proceedings of the International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_2_22_1","unstructured":"Alex Krizhevsky Ilya Sutskever and Geoffrey E Hinton. 2012. Imagenet classification with deep convolutional neural networks. In Advances in neural information processing systems (NeurIPS). 1097--1105.  Alex Krizhevsky Ilya Sutskever and Geoffrey E Hinton. 2012. Imagenet classification with deep convolutional neural networks. In Advances in neural information processing systems (NeurIPS). 1097--1105."},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.113"},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683226"},{"key":"e_1_3_2_2_25_1","volume-title":"Proceedings of the International Conference on Machine Learning (ICML). Omnipress, 689--696","author":"Ngiam Jiquan","unstructured":"Jiquan Ngiam , Aditya Khosla , Mingyu Kim , Juhan Nam , Honglak Lee , and Andrew Y. Ng . 2011. Multimodal Deep Learning . In Proceedings of the International Conference on Machine Learning (ICML). Omnipress, 689--696 . Jiquan Ngiam, Aditya Khosla, Mingyu Kim, Juhan Nam, Honglak Lee, and Andrew Y. Ng. 2011. Multimodal Deep Learning. In Proceedings of the International Conference on Machine Learning (ICML). Omnipress, 689--696."},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00772"},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46448-0_48"},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682467"},{"key":"e_1_3_2_2_29_1","volume-title":"Proceedings of the International Conference on Learning Representations (ICLR).","author":"Simonyan Karen","year":"2015","unstructured":"Karen Simonyan and Andrew Zisserman . 2015 . Very deep convolutional networks for large-scale image recognition . In Proceedings of the International Conference on Learning Representations (ICLR). Karen Simonyan and Andrew Zisserman. 2015. Very deep convolutional networks for large-scale image recognition. In Proceedings of the International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01216-8_16"},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2013.137"},{"key":"e_1_3_2_2_32_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N Gomez \u0141ukasz Kaiser and Illia Polosukhin. 2017. Attention is all you need. In Advances in neural information processing systems (NeurIPS). 5998--6008.  Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N Gomez \u0141ukasz Kaiser and Illia Polosukhin. 2017. Attention is all you need. In Advances in neural information processing systems (NeurIPS). 5998--6008."},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00813"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298968"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/3343031.3350940"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00639"},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01075"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00644"},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2019.2922108"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00719"},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01030"},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3343031.3351044"},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00182"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01246-5_35"}],"event":{"name":"MM '20: The 28th ACM International Conference on Multimedia","location":"Seattle WA USA","acronym":"MM '20","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 28th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3394171.3413581","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3394171.3413581","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T20:47:14Z","timestamp":1750193234000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3394171.3413581"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,10,12]]},"references-count":44,"alternative-id":["10.1145\/3394171.3413581","10.1145\/3394171"],"URL":"https:\/\/doi.org\/10.1145\/3394171.3413581","relation":{},"subject":[],"published":{"date-parts":[[2020,10,12]]},"assertion":[{"value":"2020-10-12","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}