{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,25]],"date-time":"2026-06-25T03:37:21Z","timestamp":1782358641617,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":25,"publisher":"ACM","license":[{"start":{"date-parts":[[2020,10,12]],"date-time":"2020-10-12T00:00:00Z","timestamp":1602460800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Beijing Natural Science Foundation","award":["4192028"],"award-info":[{"award-number":["4192028"]}]},{"name":"National Natural Science Foundation of China","award":["61772535"],"award-info":[{"award-number":["61772535"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2020,10,12]]},"DOI":"10.1145\/3394171.3413576","type":"proceedings-article","created":{"date-parts":[[2020,10,12]],"date-time":"2020-10-12T12:27:35Z","timestamp":1602505655000},"page":"4217-4225","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":14,"title":["ICECAP: Information Concentrated Entity-aware Image Captioning"],"prefix":"10.1145","author":[{"given":"Anwen","family":"Hu","sequence":"first","affiliation":[{"name":"Renming University of China, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shizhe","family":"Chen","sequence":"additional","affiliation":[{"name":"Renmin University of China, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qin","family":"Jin","sequence":"additional","affiliation":[{"name":"Renmin University of China, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2020,10,12]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"crossref","unstructured":"Peter Anderson Xiaodong He Chris Buehler Damien Teney Mark Johnson Stephen Gould and Lei Zhang. 2018. Bottom-Up and Top-Down Attention for Image Captioning and Visual Question Answering. In CVPR. IEEE Computer Society 6077--6086.  Peter Anderson Xiaodong He Chris Buehler Damien Teney Mark Johnson Stephen Gould and Lei Zhang. 2018. Bottom-Up and Top-Down Attention for Image Captioning and Visual Question Answering. In CVPR. IEEE Computer Society 6077--6086.","DOI":"10.1109\/CVPR.2018.00636"},{"key":"e_1_3_2_2_2_1","unstructured":"Sanjeev Arora Yingyu Liang and Tengyu Ma. 2017. A Simple but Tough-to-Beat Baseline for Sentence Embeddings. In ICLR. OpenReview.net.  Sanjeev Arora Yingyu Liang and Tengyu Ma. 2017. A Simple but Tough-to-Beat Baseline for Sentence Embeddings. In ICLR. OpenReview.net."},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"crossref","unstructured":"Ali Furkan Biten Llu'i s G\u00f3 mez Marcc al Rusi n ol and Dimosthenis Karatzas. 2019. acute; Good News Everyone! Context Driven Entity-Aware Captioning for News Images. In CVPR. Computer Vision Foundation \/ IEEE 12466--12475.  Ali Furkan Biten Llu'i s G\u00f3 mez Marcc al Rusi n ol and Dimosthenis Karatzas. 2019. acute; Good News Everyone! Context Driven Entity-Aware Captioning for News Images. In CVPR. Computer Vision Foundation \/ IEEE 12466--12475.","DOI":"10.1109\/CVPR.2019.01275"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"crossref","unstructured":"Michael J. Denkowski and Alon Lavie. 2014. Meteor Universal: Language Specific Translation Evaluation for Any Target Language. In WMT@ACL. The Association for Computer Linguistics 376--380.  Michael J. Denkowski and Alon Lavie. 2014. Meteor Universal: Language Specific Translation Evaluation for Any Target Language. In WMT@ACL. The Association for Computer Linguistics 376--380.","DOI":"10.3115\/v1\/W14-3348"},{"key":"e_1_3_2_2_5_1","unstructured":"Fartash Faghri David J. Fleet Jamie Ryan Kiros and Sanja Fidler. 2018. VSE+: Improving Visual-Semantic Embeddings with Hard Negatives. In BMVC. BMVA Press 12.  Fartash Faghri David J. Fleet Jamie Ryan Kiros and Sanja Fidler. 2018. VSE+: Improving Visual-Semantic Embeddings with Hard Negatives. In BMVC. BMVA Press 12."},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"crossref","unstructured":"Yansong Feng and Mirella Lapata. 2012. Automatic caption generation for news images. IEEE transactions on pattern analysis and machine intelligence Vol. 35 4 (2012) 797--812.  Yansong Feng and Mirella Lapata. 2012. Automatic caption generation for news images. IEEE transactions on pattern analysis and machine intelligence Vol. 35 4 (2012) 797--812.","DOI":"10.1109\/TPAMI.2012.118"},{"key":"e_1_3_2_2_7_1","unstructured":"Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Deep Residual Learning for Image Recognition. In CVPR. IEEE Computer Society 770--778.  Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Deep Residual Learning for Image Recognition. In CVPR. IEEE Computer Society 770--778."},{"key":"e_1_3_2_2_8_1","unstructured":"Matthew Honnibal and Ines Montani. 2017. spacy 2: Natural language understanding with bloom embeddings convolutional neural networks and incremental parsing. To appear Vol. 7 (2017).  Matthew Honnibal and Ines Montani. 2017. spacy 2: Natural language understanding with bloom embeddings convolutional neural networks and incremental parsing. To appear Vol. 7 (2017)."},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"crossref","unstructured":"Yan Huang Wei Wang and Liang Wang. 2017. Instance-Aware Image and Sentence Matching with Selective Multimodal LSTM. In CVPR. IEEE Computer Society 7254--7262.  Yan Huang Wei Wang and Liang Wang. 2017. Instance-Aware Image and Sentence Matching with Selective Multimodal LSTM. In CVPR. IEEE Computer Society 7254--7262.","DOI":"10.1109\/CVPR.2017.767"},{"key":"e_1_3_2_2_10_1","volume-title":"Rouge: A package for automatic evaluation of summaries. In Text summarization branches out. 74--81.","author":"Lin Chin-Yew","year":"2004"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"crossref","unstructured":"Tsung-Yi Lin Michael Maire Serge J. Belongie James Hays Pietro Perona Deva Ramanan Piotr Doll\u00e1 r and C. Lawrence Zitnick. 2014. Microsoft COCO: Common Objects in Context. In ECCV (5) (Lecture Notes in Computer Science) Vol. 8693. Springer 740--755.  Tsung-Yi Lin Michael Maire Serge J. Belongie James Hays Pietro Perona Deva Ramanan Piotr Doll\u00e1 r and C. Lawrence Zitnick. 2014. Microsoft COCO: Common Objects in Context. In ECCV (5) (Lecture Notes in Computer Science) Vol. 8693. Springer 740--755.","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"crossref","unstructured":"Di Lu Spencer Whitehead Lifu Huang Heng Ji and Shih-Fu Chang. 2018. Entity-aware Image Caption Generation. In EMNLP. Association for Computational Linguistics 4013--4023.  Di Lu Spencer Whitehead Lifu Huang Heng Ji and Shih-Fu Chang. 2018. Entity-aware Image Caption Generation. In EMNLP. Association for Computational Linguistics 4013--4023.","DOI":"10.18653\/v1\/D18-1435"},{"key":"e_1_3_2_2_13_1","unstructured":"Jiasen Lu Caiming Xiong Devi Parikh and Richard Socher. 2017. Knowing When to Look: Adaptive Attention via a Visual Sentinel for Image Captioning. In CVPR. IEEE Computer Society 3242--3250.  Jiasen Lu Caiming Xiong Devi Parikh and Richard Socher. 2017. Knowing When to Look: Adaptive Attention via a Visual Sentinel for Image Captioning. In CVPR. IEEE Computer Society 3242--3250."},{"key":"e_1_3_2_2_14_1","unstructured":"Hyeonseob Nam Jung-Woo Ha and Jeonghee Kim. 2017. Dual Attention Networks for Multimodal Reasoning and Matching. In CVPR. IEEE Computer Society 2156--2164.  Hyeonseob Nam Jung-Woo Ha and Jeonghee Kim. 2017. Dual Attention Networks for Multimodal Reasoning and Matching. In CVPR. IEEE Computer Society 2156--2164."},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"crossref","unstructured":"Kishore Papineni Salim Roukos Todd Ward and Wei-Jing Zhu. 2002. Bleu: a Method for Automatic Evaluation of Machine Translation. In ACL. ACL 311--318.  Kishore Papineni Salim Roukos Todd Ward and Wei-Jing Zhu. 2002. Bleu: a Method for Automatic Evaluation of Machine Translation. In ACL. ACL 311--318.","DOI":"10.3115\/1073083.1073135"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"crossref","unstructured":"Jeffrey Pennington Richard Socher and Christopher D. Manning. 2014. Glove: Global Vectors for Word Representation. In EMNLP. ACL 1532--1543.  Jeffrey Pennington Richard Socher and Christopher D. Manning. 2014. Glove: Global Vectors for Word Representation. In EMNLP. ACL 1532--1543.","DOI":"10.3115\/v1\/D14-1162"},{"key":"e_1_3_2_2_17_1","volume-title":"Breakingnews: Article annotation by image and text processing","author":"Ramisa Arnau","year":"2018"},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.131"},{"key":"e_1_3_2_2_19_1","volume-title":"ACL (1)","author":"Sennrich Rico"},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2016.2628585"},{"key":"e_1_3_2_2_21_1","volume-title":"Transform and Tell: Entity-Aware News Image Captioning. In The IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","author":"Tran Alasdair","year":"2020"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"crossref","unstructured":"Ramakrishna Vedantam C. Lawrence Zitnick and Devi Parikh. 2015. CIDEr: Consensus-based image description evaluation. In CVPR. IEEE Computer Society 4566--4575.  Ramakrishna Vedantam C. Lawrence Zitnick and Devi Parikh. 2015. CIDEr: Consensus-based image description evaluation. In CVPR. IEEE Computer Society 4566--4575.","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"crossref","unstructured":"Oriol Vinyals Alexander Toshev Samy Bengio and Dumitru Erhan. 2015. Show and tell: A neural image caption generator. In CVPR. IEEE Computer Society 3156--3164.  Oriol Vinyals Alexander Toshev Samy Bengio and Dumitru Erhan. 2015. Show and tell: A neural image caption generator. In CVPR. IEEE Computer Society 3156--3164.","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"e_1_3_2_2_24_1","volume-title":"Attend and Tell: Neural Image Caption Generation with Visual Attention. In ICML (JMLR Workshop and Conference Proceedings)","volume":"37","author":"Xu Kelvin","year":"2015"},{"key":"e_1_3_2_2_25_1","volume-title":"ACL (1)","author":"Zhao Sanqiang"}],"event":{"name":"MM '20: The 28th ACM International Conference on Multimedia","location":"Seattle WA USA","acronym":"MM '20","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 28th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3394171.3413576","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3394171.3413576","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T20:47:14Z","timestamp":1750193234000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3394171.3413576"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,10,12]]},"references-count":25,"alternative-id":["10.1145\/3394171.3413576","10.1145\/3394171"],"URL":"https:\/\/doi.org\/10.1145\/3394171.3413576","relation":{},"subject":[],"published":{"date-parts":[[2020,10,12]]},"assertion":[{"value":"2020-10-12","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}