{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T04:27:00Z","timestamp":1750220820545,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":36,"publisher":"ACM","license":[{"start":{"date-parts":[[2019,10,15]],"date-time":"2019-10-15T00:00:00Z","timestamp":1571097600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"National Postdoctoral Program for Innovative Talents","award":["BX201700255"],"award-info":[{"award-number":["BX201700255"]}]},{"name":"National Natural Science Foundation of China","award":["61532018"],"award-info":[{"award-number":["61532018"]}]},{"name":"Beijing Natural Science Foundation","award":["L182054"],"award-info":[{"award-number":["L182054"]}]},{"DOI":"10.13039\/501100002858","name":"China Postdoctoral Science Foundation","doi-asserted-by":"publisher","award":["2018M631583"],"award-info":[{"award-number":["2018M631583"]}],"id":[{"id":"10.13039\/501100002858","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2019,10,15]]},"DOI":"10.1145\/3343031.3351051","type":"proceedings-article","created":{"date-parts":[[2019,10,21]],"date-time":"2019-10-21T16:32:26Z","timestamp":1571675546000},"page":"1286-1294","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":5,"title":["Aberrance-aware Gradient-sensitive Attentions for Scene Recognition with RGB-D Videos"],"prefix":"10.1145","author":[{"given":"Xinhang","family":"Song","sequence":"first","affiliation":[{"name":"The institute of the computing technology in Chinese science academy, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sixian","family":"Zhang","sequence":"additional","affiliation":[{"name":"The institute of the computing technology in Chinese science academy, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuyun","family":"Hua","sequence":"additional","affiliation":[{"name":"Key Lab of Intelligent Information Processing of Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shuqiang","family":"Jiang","sequence":"additional","affiliation":[{"name":"Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2019,10,15]]},"reference":[{"volume-title":"Second-Order Constrained Parametric Proposals and Sequential Search-Based Structured Prediction for Semantic Segmentation in RGB-D Images","author":"Banica Dan","key":"e_1_3_2_1_1_1","doi-asserted-by":"crossref","DOI":"10.1109\/CVPR.2015.7298974"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"volume-title":"The IEEE Conference on Computer Vision and Pattern Recognition (CVPR) .","author":"Derpanis Konstantinos G.","key":"e_1_3_2_1_3_1"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298878"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1167\/7.1.10"},{"volume-title":"Spatiotemporal Residual Networks for Video Action Recognition. Advances in Neural Information Processing Systems 29","author":"Feichtenhofer Christoph","key":"e_1_3_2_1_6_1"},{"volume-title":"Temporal Residual Networks for Dynamic Scene Recognition. In The IEEE Conference on Computer Vision and Pattern Recognition (CVPR) .","author":"Feichtenhofer Christoph","key":"e_1_3_2_1_7_1"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-014-0777-6"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-014-0777-6"},{"volume-title":"Cross Modal Distillation for Supervision Transfer. The IEEE Conference on Computer Vision and Pattern Recognition (CVPR) .","year":"2016","author":"Gupta Saurabh","key":"e_1_3_2_1_10_1"},{"volume-title":"Deep Residual Learning for Image Recognition. In The IEEE Conference on Computer Vision and Pattern Recognition (CVPR) .","year":"2016","author":"He Kaiming","key":"e_1_3_2_1_11_1"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2012.59"},{"volume-title":"Large-Scale Video Classification with Convolutional Neural Networks. In 2014 IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2014","year":"2014","author":"Karpathy Andrej","key":"e_1_3_2_1_13_1"},{"volume-title":"Hinton","year":"2012","author":"Krizhevsky Alex","key":"e_1_3_2_1_14_1"},{"key":"e_1_3_2_1_15_1","unstructured":"S. Lazebnik C. Schmid and J. Ponce. 2006. Beyond bags of features: Spatial pyramid matching for recognizing natural scene categories. In CVPR .  S. Lazebnik C. Schmid and J. Ponce. 2006. Beyond bags of features: Spatial pyramid matching for recognizing natural scene categories. In CVPR ."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.092277599"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"crossref","unstructured":"A. Quattoni and A. Torralba. 2009. Recognizing indoor scenes. In CVPR .  A. Quattoni and A. Torralba. 2009. Recognizing indoor scenes. In CVPR .","DOI":"10.1109\/CVPR.2009.5206537"},{"volume-title":"2010 IEEE Computer Society Conference on Computer Vision and Pattern Recognition. 1911--1918","year":"2010","author":"Shroff N.","key":"e_1_3_2_1_18_1"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-33715-4_54"},{"volume-title":"Two-Stream Convolutional Networks for Action Recognition in Videos. In Advances in Neural Information Processing Systems 27: Annual Conference on Neural Information Processing Systems 2014","year":"2014","author":"Simonyan Karen","key":"e_1_3_2_1_20_1"},{"volume-title":"Very Deep Convolutional Networks for Large-Scale Image Recognition. In International Conference on Learning Representations .","author":"Simonyan K.","key":"e_1_3_2_1_21_1"},{"key":"e_1_3_2_1_22_1","unstructured":"R. Socher B. Huval B. Bath C. D. Manning and A. Y. Ng. 2012. Convolutional-recursive deep learning for 3D object classification. In NIPS .  R. Socher B. Huval B. Bath C. D. Manning and A. Y. Ng. 2012. Convolutional-recursive deep learning for 3D object classification. In NIPS ."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298655"},{"volume-title":"Proceedings of the Thirty-First AAAI Conference on Artificial Intelligence, February 4--9","year":"2017","author":"Song Xinhang","key":"e_1_3_2_1_24_1"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2017\/631"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2018.2872629"},{"key":"e_1_3_2_1_27_1","volume-title":"Nature","volume":"381","author":"Thorpe S.","year":"1996"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.510"},{"volume-title":"Modality and Component Aware Feature Fusion For RGB-D Scene Classification. In The IEEE Conference on Computer Vision and Pattern Recognition (CVPR) .","year":"2016","author":"Wang Anran","key":"e_1_3_2_1_29_1"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33015232"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2017.2666742"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"crossref","unstructured":"J. Xiao J. Hayes K. Ehringer A. Olivia and A. Torralba. 2010. SUN database: Largescale scene recognition from Abbey to Zoo. In CVPR .  J. Xiao J. Hayes K. Ehringer A. Olivia and A. Torralba. 2010. SUN database: Largescale scene recognition from Abbey to Zoo. In CVPR .","DOI":"10.1109\/CVPR.2010.5539970"},{"volume-title":"Computer Vision (ICCV), 2013 IEEE International Conference on. 1625--1632","year":"2013","author":"Xiao Jianxiong","key":"e_1_3_2_1_33_1"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2013.2295756"},{"volume-title":"Places: An Image Database for Deep Scene Understanding","year":"2018","author":"Zhou Bolei","key":"e_1_3_2_1_35_1"},{"volume-title":"Discriminative Multi-Modal Feature Fusion for RGBD Indoor Scene Recognition. In The IEEE Conference on Computer Vision and Pattern Recognition (CVPR) .","year":"2016","author":"Zhu Hongyuan","key":"e_1_3_2_1_36_1"}],"event":{"name":"MM '19: The 27th ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Nice France","acronym":"MM '19"},"container-title":["Proceedings of the 27th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3343031.3351051","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3343031.3351051","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T23:13:12Z","timestamp":1750201992000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3343031.3351051"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,10,15]]},"references-count":36,"alternative-id":["10.1145\/3343031.3351051","10.1145\/3343031"],"URL":"https:\/\/doi.org\/10.1145\/3343031.3351051","relation":{},"subject":[],"published":{"date-parts":[[2019,10,15]]},"assertion":[{"value":"2019-10-15","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}