{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,1]],"date-time":"2025-12-01T11:22:07Z","timestamp":1764588127150,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":25,"publisher":"ACM","license":[{"start":{"date-parts":[[2020,10,12]],"date-time":"2020-10-12T00:00:00Z","timestamp":1602460800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"UESTC Funds for Outstanding Talents","award":["A1098531023601183"],"award-info":[{"award-number":["A1098531023601183"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2020,10,12]]},"DOI":"10.1145\/3394171.3416275","type":"proceedings-article","created":{"date-parts":[[2020,10,12]],"date-time":"2020-10-12T12:26:25Z","timestamp":1602505585000},"page":"4575-4579","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":11,"title":["Curriculum Learning for Wide Multimedia-Based Transformer with Graph Target Detection"],"prefix":"10.1145","author":[{"given":"Weilong","family":"Chen","sequence":"first","affiliation":[{"name":"University of Electronic Science and Technology of China &amp; Tencent, Chengdu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Feng","family":"Hong","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chenghao","family":"Huang","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shaoliang","family":"Zhang","sequence":"additional","affiliation":[{"name":"Tencent, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rui","family":"Wang","sequence":"additional","affiliation":[{"name":"Tencent, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ruobing","family":"Xie","sequence":"additional","affiliation":[{"name":"Tencent, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Feng","family":"Xia","sequence":"additional","affiliation":[{"name":"Tencent, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Leyu","family":"Lin","sequence":"additional","affiliation":[{"name":"Tencent, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yanru","family":"Zhang","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yan","family":"Wang","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Chengdu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2020,10,12]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"crossref","unstructured":"Peter Anderson Xiaodong He Chris Buehler Damien Teney Mark Johnson Stephen Gould and Lei Zhang. 2018. Bottom-Up and Top-Down Attention for Image Captioning and Visual Question Answering. In CVPR.  Peter Anderson Xiaodong He Chris Buehler Damien Teney Mark Johnson Stephen Gould and Lei Zhang. 2018. Bottom-Up and Top-Down Attention for Image Captioning and Visual Question Answering. In CVPR.","DOI":"10.1109\/CVPR.2018.00636"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/1553374.1553380"},{"key":"#cr-split#-e_1_3_2_2_3_1.1","doi-asserted-by":"crossref","unstructured":"Qi Cao Huawei Shen Cen Keting Wentao Ouyang and Xueqi Cheng. 2017. DeepHawkes: Bridging the Gap between Prediction and Understanding of Information Cascades. 1149--1158. https:\/\/doi.org\/10.1145\/3132847.3132973 10.1145\/3132847.3132973","DOI":"10.1145\/3132847.3132973"},{"key":"#cr-split#-e_1_3_2_2_3_1.2","doi-asserted-by":"crossref","unstructured":"Qi Cao Huawei Shen Cen Keting Wentao Ouyang and Xueqi Cheng. 2017. DeepHawkes: Bridging the Gap between Prediction and Understanding of Information Cascades. 1149--1158. https:\/\/doi.org\/10.1145\/3132847.3132973","DOI":"10.1145\/3132847.3132973"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2018.12.039"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3297662.3365812"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"crossref","unstructured":"Felix A Gers J\u00fcrgen Schmidhuber and Fred Cummins. 1999. Learning to forget: Continual prediction with LSTM. (1999).  Felix A Gers J\u00fcrgen Schmidhuber and Fred Cummins. 1999. Learning to forget: Continual prediction with LSTM. (1999).","DOI":"10.1049\/cp:19991218"},{"key":"e_1_3_2_2_7_1","volume-title":"Contextual lstm (clstm) models for large scale nlp tasks. arXiv preprint arXiv:1602.06291","author":"Ghosh Shalini","year":"2016","unstructured":"Shalini Ghosh , Oriol Vinyals , Brian Strope , Scott Roy , Tom Dean , and Larry Heck . 2016. Contextual lstm (clstm) models for large scale nlp tasks. arXiv preprint arXiv:1602.06291 ( 2016 ). Shalini Ghosh, Oriol Vinyals, Brian Strope, Scott Roy, Tom Dean, and Larry Heck. 2016. Contextual lstm (clstm) models for large scale nlp tasks. arXiv preprint arXiv:1602.06291 (2016)."},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.81"},{"key":"e_1_3_2_2_9_1","volume-title":"Densenet: Implementing efficient convnet descriptor pyramids. arXiv preprint arXiv:1404.1869","author":"Iandola Forrest","year":"2014","unstructured":"Forrest Iandola , Matt Moskewicz , Sergey Karayev , Ross Girshick , Trevor Darrell , and Kurt Keutzer . 2014 . Densenet: Implementing efficient convnet descriptor pyramids. arXiv preprint arXiv:1404.1869 (2014). Forrest Iandola, Matt Moskewicz, Sergey Karayev, Ross Girshick, Trevor Darrell, and Kurt Keutzer. 2014. Densenet: Implementing efficient convnet descriptor pyramids. arXiv preprint arXiv:1404.1869 (2014)."},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298927"},{"key":"e_1_3_2_2_11_1","unstructured":"Alex Krizhevsky Ilya Sutskever and Geoffrey E Hinton. 2012. Imagenet classification with deep convolutional neural networks. In Advances in neural information processing systems. 1097--1105.  Alex Krizhevsky Ilya Sutskever and Geoffrey E Hinton. 2012. Imagenet classification with deep convolutional neural networks. In Advances in neural information processing systems. 1097--1105."},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01225-0_13"},{"key":"e_1_3_2_2_13_1","volume-title":"Anna Veronika Dorogush, and Andrey Gulin","author":"Prokhorenkova Liudmila","year":"2018","unstructured":"Liudmila Prokhorenkova , Gleb Gusev , Aleksandr Vorobev , Anna Veronika Dorogush, and Andrey Gulin . 2018 . CatBoost: unbiased boosting with categorical features. In Advances in neural information processing systems. 6638--6648. Liudmila Prokhorenkova, Gleb Gusev, Aleksandr Vorobev, Anna Veronika Dorogush, and Andrey Gulin. 2018. CatBoost: unbiased boosting with categorical features. In Advances in neural information processing systems. 6638--6648."},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"crossref","unstructured":"Yao Qin Dongjin Song Haifeng Chen Wei Cheng Guofei Jiang and Garrison W Cottrell. 2017. A Dual-Stage Attention-Based Recurrent Neural Network for Time Series Prediction. In IJCAI.  Yao Qin Dongjin Song Haifeng Chen Wei Cheng Guofei Jiang and Garrison W Cottrell. 2017. A Dual-Stage Attention-Based Recurrent Neural Network for Time Series Prediction. In IJCAI.","DOI":"10.24963\/ijcai.2017\/366"},{"key":"e_1_3_2_2_15_1","volume-title":"VL-BERT: Pre-training of Generic Visual-Linguistic Representations. In International Conference on Learning Representations.","author":"Su Weijie","year":"2019","unstructured":"Weijie Su , Xizhou Zhu , Yue Cao , Bin Li , Lewei Lu , Furu Wei , and Jifeng Dai . 2019 . VL-BERT: Pre-training of Generic Visual-Linguistic Representations. In International Conference on Learning Representations. Weijie Su, Xizhou Zhu, Yue Cao, Bin Li, Lewei Lu, Furu Wei, and Jifeng Dai. 2019. VL-BERT: Pre-training of Generic Visual-Linguistic Representations. In International Conference on Learning Representations."},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"e_1_3_2_2_17_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N Gomez \u0141ukasz Kaiser and Illia Polosukhin. 2017. Attention is all you need. In Advances in neural information processing systems. 5998--6008.  Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan N Gomez \u0141ukasz Kaiser and Illia Polosukhin. 2017. Attention is all you need. In Advances in neural information processing systems. 5998--6008."},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/2964284.2964335"},{"key":"e_1_3_2_2_19_1","volume-title":"Sequential Prediction of Social Media Popularity with Deep Temporal Context Networks. In International Joint Conference on Artificial Intelligence (IJCAI)","author":"Wu Bo","year":"2017","unstructured":"Bo Wu , Wen-Huang Cheng , Yongdong Zhang , Huang Qiushi , Li Jintao , and Tao Mei . 2017 . Sequential Prediction of Social Media Popularity with Deep Temporal Context Networks. In International Joint Conference on Artificial Intelligence (IJCAI) ( Melbourne, Australia). Bo Wu, Wen-Huang Cheng, Yongdong Zhang, Huang Qiushi, Li Jintao, and Tao Mei. 2017. Sequential Prediction of Social Media Popularity with Deep Temporal Context Networks. In International Joint Conference on Artificial Intelligence (IJCAI) (Melbourne, Australia)."},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v30i1.9970"},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3343031.3356077"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11280-016-0405-1"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3015463"},{"key":"e_1_3_2_2_24_1","unstructured":"Sheng Yu and Subhash Kak. [n.d.]. A Survey of Prediction Using Social Media. ([n. d.]).  Sheng Yu and Subhash Kak. [n.d.]. A Survey of Prediction Using Social Media. ([n. d.])."}],"event":{"name":"MM '20: The 28th ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Seattle WA USA","acronym":"MM '20"},"container-title":["Proceedings of the 28th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3394171.3416275","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3394171.3416275","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T22:01:24Z","timestamp":1750197684000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3394171.3416275"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,10,12]]},"references-count":25,"alternative-id":["10.1145\/3394171.3416275","10.1145\/3394171"],"URL":"https:\/\/doi.org\/10.1145\/3394171.3416275","relation":{},"subject":[],"published":{"date-parts":[[2020,10,12]]},"assertion":[{"value":"2020-10-12","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}