{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,5]],"date-time":"2026-03-05T14:03:40Z","timestamp":1772719420320,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":40,"publisher":"ACM","license":[{"start":{"date-parts":[[2019,10,15]],"date-time":"2019-10-15T00:00:00Z","timestamp":1571097600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"National Natural Science Foundation of China","award":["61532018"],"award-info":[{"award-number":["61532018"]}]},{"name":"National Postdoctoral Program for Innovative Talents","award":["BX201700255"],"award-info":[{"award-number":["BX201700255"]}]},{"name":"Beijing Natural Science Foundation","award":["L182054"],"award-info":[{"award-number":["L182054"]}]},{"name":"China Postdoctoral Science Foundation","award":["2018M631583"],"award-info":[{"award-number":["2018M631583"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2019,10,15]]},"DOI":"10.1145\/3343031.3350913","type":"proceedings-article","created":{"date-parts":[[2019,10,21]],"date-time":"2019-10-21T16:32:26Z","timestamp":1571675546000},"page":"793-801","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["MUCH"],"prefix":"10.1145","author":[{"given":"Xinhang","family":"Song","sequence":"first","affiliation":[{"name":"Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bohan","family":"Wang","sequence":"additional","affiliation":[{"name":"Chinese Academy of Sciences &amp; University of Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Gongwei","family":"Chen","sequence":"additional","affiliation":[{"name":"Chinese Academy of Sciences &amp; University of Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shuqiang","family":"Jiang","sequence":"additional","affiliation":[{"name":"Chinese Academy of Sciences &amp; University of Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2019,10,15]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Neural machine translation by jointly learning to align and translate. ICLR","author":"Bahdanau Dzmitry","year":"2015"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2016.2555080"},{"key":"e_1_3_2_1_3_1","volume-title":"Roy-Chowdhury","author":"Bappy Jawadul H.","year":"2016"},{"key":"e_1_3_2_1_4_1","volume-title":"Classemes and Other Classifier-based Features for Efficient Object Categorization","author":"Bergamo Alessandro"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"crossref","unstructured":"Mandar Dixit Si Chen Dashan Gao Nikhil Rasiwasia and Nuno Vasconcelos. 2015. Scene Classification with Semantic Fisher Vectors. CVPR .  Mandar Dixit Si Chen Dashan Gao Nikhil Rasiwasia and Nuno Vasconcelos. 2015. Scene Classification with Semantic Fisher Vectors. CVPR .","DOI":"10.1109\/CVPR.2015.7298916"},{"key":"e_1_3_2_1_6_1","volume-title":"Object based Scene Representations using Fisher Scores of Local Subspace Projections. Advances in Neural Information Processing Systems 29","author":"Dixit Mandar D"},{"key":"e_1_3_2_1_7_1","unstructured":"Carl Doersch Abhinav Gupta and Alexei A Efros. 2013. Mid-level Visual Element Discovery as Discriminative Mode Seeking. NIPS. 494--502.  Carl Doersch Abhinav Gupta and Alexei A Efros. 2013. Mid-level Visual Element Discovery as Discriminative Mode Seeking. NIPS. 494--502."},{"key":"e_1_3_2_1_8_1","unstructured":"L. Fei-Fei and P. Perona. 2005. A bayesian hierarchical model for learning natural scene categories. In CVPR .  L. Fei-Fei and P. Perona. 2005. A bayesian hierarchical model for learning natural scene categories. In CVPR ."},{"key":"e_1_3_2_1_9_1","volume-title":"Semantic Clustering for Robust Fine-Grained Scene Recognition","author":"George Marian"},{"key":"e_1_3_2_1_10_1","volume-title":"Rich Feature Hierarchies for Accurate Object Detection and Semantic Segmentation. In The IEEE Conference on Computer Vision and Pattern Recognition (CVPR) .","author":"Girshick Ross","year":"2014"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"crossref","unstructured":"Y. Gong L. Wang R. Guo and S. Lazebnik. 2014. Multi-scale orderless pooling of deep convolutional activation features. In ECCV .  Y. Gong L. Wang R. Guo and S. Lazebnik. 2014. Multi-scale orderless pooling of deep convolutional activation features. In ECCV .","DOI":"10.1007\/978-3-319-10584-0_26"},{"key":"e_1_3_2_1_12_1","volume-title":"IEEE Conference on Computer Vision and Pattern Recognition (CVPR) .","author":"Johnson A. Karpathy J."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.visres.2007.09.013"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"crossref","unstructured":"Mayank Juneja Andrea Vedaldi C. V. Jawahar and Andrew Zisserman. 2013. Blocks that Shout: Distinctive Parts for Scene Classification. In CVPR .  Mayank Juneja Andrea Vedaldi C. V. Jawahar and Andrew Zisserman. 2013. Blocks that Shout: Distinctive Parts for Scene Classification. In CVPR .","DOI":"10.1109\/CVPR.2013.124"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-016-0981-7"},{"key":"e_1_3_2_1_16_1","unstructured":"S. Lazebnik C. Schmid and J. Ponce. 2006. Beyond bags of features: Spatial pyramid matching for recognizing natural scene categories. In CVPR .  S. Lazebnik C. Schmid and J. Ponce. 2006. Beyond bags of features: Spatial pyramid matching for recognizing natural scene categories. In CVPR ."},{"key":"e_1_3_2_1_17_1","volume-title":"Object Bank: A High-Level Image Representation for Scene Classification and Semantic Feature Sparsification. In NIPS .","author":"Li L.J.","year":"2010"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-013-0660-x"},{"key":"e_1_3_2_1_19_1","unstructured":"Xin Li and Yuhong Guo. 2014. Latent Semantic Representation Learning for Scene Classification. In ICML .  Xin Li and Yuhong Guo. 2014. Latent Semantic Representation Learning for Scene Classification. In ICML ."},{"key":"e_1_3_2_1_20_1","volume-title":"Proceedings of the Thirty-Second AAAI Conference on Artificial Intelligence","author":"Liu Yang","year":"2018"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1023\/B:VISI.0000029664.99615.94"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1167\/10.3.11"},{"key":"e_1_3_2_1_23_1","unstructured":"Junhua Mao Wei Xu Yi Yang Jiang Wang Zhiheng Huang and Alan Yuille. 2015. Deep captioning with multimodal recurrent neural networks (m-rnn). (2015).  Junhua Mao Wei Xu Yi Yang Jiang Wang Zhiheng Huang and Alan Yuille. 2015. Deep captioning with multimodal recurrent neural networks (m-rnn). (2015)."},{"key":"e_1_3_2_1_24_1","volume-title":"Efficient Estimation of Word Representations in Vector Space. CoRR","author":"Mikolov Tomas","year":"2013"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"crossref","unstructured":"A. Quattoni and A. Torralba. 2009. Recognizing indoor scenes. In CVPR .  A. Quattoni and A. Torralba. 2009. Recognizing indoor scenes. In CVPR .","DOI":"10.1109\/CVPR.2009.5206537"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2007.900138"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2011.175"},{"key":"e_1_3_2_1_28_1","volume-title":"Faster R-CNN: Towards Real-Time Object Detection with Region Proposal Networks. Advances in Neural Information Processing Systems 28","author":"Ren Shaoqing"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0816-y"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"crossref","unstructured":"J. Sanchez and F. Perronnin. 2011. High-Dimensional Signature Compression for Large-Scale Image Classification. In Neural Comput.  J. Sanchez and F. Perronnin. 2011. High-Dimensional Signature Compression for Large-Scale Image Classification. In Neural Comput.","DOI":"10.1109\/CVPR.2011.5995504"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"crossref","unstructured":"Lorenzo Torresani Martin Szummer and Andrew Fitzgibbon. 2010. Efficient Object Category Recognition Using Classemes. In ECCV .  Lorenzo Torresani Martin Szummer and Andrew Fitzgibbon. 2010. Efficient Object Category Recognition Using Classemes. In ECCV .","DOI":"10.1007\/978-3-642-15549-9_56"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"crossref","unstructured":"Oriol Vinyals Alexander Toshev Samy Bengio and Dumitru Erhan. 2015. Show and Tell: A Neural Image Caption Generator. CVPR. 3156--3164.  Oriol Vinyals Alexander Toshev Samy Bengio and Dumitru Erhan. 2015. Show and Tell: A Neural Image Caption Generator. CVPR. 3156--3164.","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-006-8614-1"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"crossref","unstructured":"Jinjun Wang Jianchao Yang Kai Yu Fengjun Lv T. Huang and Yihong Gong. 2010. Locality-constrained Linear Coding for image classification. In CVPR .  Jinjun Wang Jianchao Yang Kai Yu Fengjun Lv T. Huang and Yihong Gong. 2010. Locality-constrained Linear Coding for image classification. In CVPR .","DOI":"10.1109\/CVPR.2010.5540018"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"crossref","unstructured":"Qilong Wang Peihua Li Wangmeng Zuo and Lei Zhang. 2016. RAID-G: Robust Estimation of Approximate Infinite Dimensional Gaussian With Application to Material Recognition. In CVPR .  Qilong Wang Peihua Li Wangmeng Zuo and Lei Zhang. 2016. RAID-G: Robust Estimation of Approximate Infinite Dimensional Gaussian With Application to Material Recognition. In CVPR .","DOI":"10.1109\/CVPR.2016.480"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"crossref","unstructured":"J. Xiao J. Hayes K. Ehringer A. Olivia and A. Torralba. 2010. SUN database: Largescale scene recognition from Abbey to Zoo. In CVPR .  J. Xiao J. Hayes K. Ehringer A. Olivia and A. Torralba. 2010. SUN database: Largescale scene recognition from Abbey to Zoo. In CVPR .","DOI":"10.1109\/CVPR.2010.5539970"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.214"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2014.2330794"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2017.2723009"},{"key":"e_1_3_2_1_40_1","volume-title":"NIPS","author":"Zhou Bolei"}],"event":{"name":"MM '19: The 27th ACM International Conference on Multimedia","location":"Nice France","acronym":"MM '19","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 27th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3343031.3350913","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3343031.3350913","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T23:13:17Z","timestamp":1750201997000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3343031.3350913"}},"subtitle":["Mutual Coupling Enhancement of Scene Recognition and Dense Captioning"],"short-title":[],"issued":{"date-parts":[[2019,10,15]]},"references-count":40,"alternative-id":["10.1145\/3343031.3350913","10.1145\/3343031"],"URL":"https:\/\/doi.org\/10.1145\/3343031.3350913","relation":{},"subject":[],"published":{"date-parts":[[2019,10,15]]},"assertion":[{"value":"2019-10-15","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}