{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,20]],"date-time":"2025-12-20T22:18:11Z","timestamp":1766269091581,"version":"3.37.3"},"reference-count":44,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2022,9,15]],"date-time":"2022-09-15T00:00:00Z","timestamp":1663200000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2022,9,15]],"date-time":"2022-09-15T00:00:00Z","timestamp":1663200000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61873274"],"award-info":[{"award-number":["61873274"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Multimed Info Retr"],"published-print":{"date-parts":[[2022,12]]},"DOI":"10.1007\/s13735-022-00252-7","type":"journal-article","created":{"date-parts":[[2022,9,15]],"date-time":"2022-09-15T13:07:51Z","timestamp":1663247271000},"page":"611-618","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":12,"title":["FCT: fusing CNN and transformer for scene classification"],"prefix":"10.1007","volume":"11","author":[{"given":"Yuxiang","family":"Xie","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jie","family":"Yan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lai","family":"Kang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9184-5313","authenticated-orcid":false,"given":"Yanming","family":"Guo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiahui","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xidao","family":"Luan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,9,15]]},"reference":[{"key":"252_CR1","doi-asserted-by":"crossref","unstructured":"Peng Z, Huang W, Gu S et al (2021) Conformer: local features coupling global representations for visual recognition. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 367\u2013376","DOI":"10.1109\/ICCV48922.2021.00042"},{"key":"252_CR2","first-page":"5998","volume":"30","author":"A Vaswani","year":"2017","unstructured":"Vaswani A et al (2017) Attention is all you need. Adv Neural Inf Process Syst 30:5998\u20136008","journal-title":"Adv Neural Inf Process Syst"},{"key":"252_CR3","unstructured":"Chen M, Radford A, Child R, et al (2020) Generative pretraining from pixels. In: International conference on machine learning. PMLR, pp 1691\u20131703"},{"key":"252_CR4","unstructured":"Dosovitskiy A, Beyer L, Kolesnikov A et al (2020) An image is worth 16x16 words: transformers for image recognition at scalet. arXiv preprint arXiv:2010.11929"},{"key":"252_CR5","doi-asserted-by":"crossref","unstructured":"Carion N, Massa F, Synnaeve G et al (2020) End-to-end object detection with transformers. In: European conference on computer vision. Springer, pp 213\u2013229","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"252_CR6","unstructured":"Zhu X, Su W, Lu L, et al (2020) Deformable detr: deformable transformers for end-to-end object detection. arXiv preprint arXiv:2010.04159"},{"key":"252_CR7","doi-asserted-by":"crossref","unstructured":"Zheng S, Lu J, Zhao H et al (2021) Rethinking semantic segmentation from a sequence-to-sequence perspective with transformers. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 6881\u20136890","DOI":"10.1109\/CVPR46437.2021.00681"},{"key":"252_CR8","doi-asserted-by":"crossref","unstructured":"Chen H, Wang Y, Guo T et al (2021) Pre-trained image processing transformer. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 12299\u201312310","DOI":"10.1109\/CVPR46437.2021.01212"},{"issue":"6","key":"252_CR9","doi-asserted-by":"publisher","first-page":"1452","DOI":"10.1109\/TPAMI.2017.2723009","volume":"40","author":"B Zhou","year":"2017","unstructured":"Zhou B, Lapedriza A, Khosla A et al (2017) Places: a 10 million image database for scene recognition. IEEE Trans Pattern Anal Mach Intell 40(6):1452\u20131464","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"252_CR10","doi-asserted-by":"crossref","unstructured":"Deng J, Dong W, Socher R et al (2009) Imagenet: a large-scale hierarchical image database. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 248\u2013255","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"252_CR11","unstructured":"Krizhevsky A, Sutskever I, Hinton GE (2012) Imagenet classification with deep convolutional neural networks. In: Advances in neural information processing systems 25"},{"key":"252_CR12","unstructured":"Simonyan K, Zisserman A (2014) Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556"},{"key":"252_CR13","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S et al (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"252_CR14","doi-asserted-by":"crossref","unstructured":"Huang G, Liu Z, Van Der Maaten L, et al (2017) Densely connected convolutional networks. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 4700\u20134708","DOI":"10.1109\/CVPR.2017.243"},{"key":"252_CR15","doi-asserted-by":"crossref","unstructured":"Zoph B, Vasudevan V, Shlens J et al (2018) Learning transferable architectures for scalable image recognition. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 8697\u20138710","DOI":"10.1109\/CVPR.2018.00907"},{"key":"252_CR16","unstructured":"Zhou B, Lapedriza A, Xiao J et al (2014) Learning deep features for scene recognition using places database. In: Advances in neural information processing systems 27"},{"issue":"6","key":"252_CR17","doi-asserted-by":"publisher","first-page":"1519","DOI":"10.1109\/TMM.2019.2944241","volume":"22","author":"H Zeng","year":"2019","unstructured":"Zeng H, Song X, Chen G et al (2019) Learning scene attribute for scene recognition. IEEE Trans Multimed 22(6):1519\u20131530","journal-title":"IEEE Trans Multimed"},{"issue":"1","key":"252_CR18","doi-asserted-by":"publisher","first-page":"59","DOI":"10.1007\/s11263-013-0695-z","volume":"108","author":"G Patterson","year":"2014","unstructured":"Patterson G, Xu C, Su H et al (2014) The sun attribute database: beyond categories for deeper scene understanding. Int J Comput Vis 108(1):59\u201381","journal-title":"Int J Comput Vis"},{"issue":"10","key":"252_CR19","doi-asserted-by":"publisher","first-page":"1735","DOI":"10.1109\/LGRS.2017.2731997","volume":"14","author":"G Cheng","year":"2017","unstructured":"Cheng G, Li Z, Yao X et al (2017) Remote sensing image scene classification using bag of convolutional features. IEEE Geosci Remote Sens Lett 14(10):1735\u20131739","journal-title":"IEEE Geosci Remote Sens Lett"},{"issue":"10","key":"252_CR20","doi-asserted-by":"publisher","first-page":"5653","DOI":"10.1109\/TGRS.2017.2711275","volume":"55","author":"E Li","year":"2017","unstructured":"Li E, Xia J, Du P et al (2017) Integrating multilayer features of convolutional neural networks for remote sensing scene classification. IEEE Trans Geosci Remote Sens 55(10):5653\u20135665","journal-title":"IEEE Trans Geosci Remote Sens"},{"key":"252_CR21","doi-asserted-by":"crossref","unstructured":"Liu Y, Chen Q, Chen W, et al (2018) Dictionary learning inspired deep network for scene recognition. In: Proceedings of the AAAI conference on artificial intelligence, vol 32, No 1","DOI":"10.1609\/aaai.v32i1.12312"},{"key":"252_CR22","doi-asserted-by":"crossref","unstructured":"Chen Y, Dai X, Chen D et al (2021) Mobile-former: bridging mobilenet and transformer. arXiv preprint arXiv:2108.05895","DOI":"10.1109\/CVPR52688.2022.00520"},{"key":"252_CR23","doi-asserted-by":"crossref","unstructured":"Herranz L, Jiang S, Li X (2016) Scene recognition with cnns: objects, scales and dataset bias. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 571\u2013579","DOI":"10.1109\/CVPR.2016.68"},{"key":"252_CR24","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2020.107205","volume":"102","author":"L Xie","year":"2020","unstructured":"Xie L, Lee F, Liu L et al (2020) Scene recognition: a comprehensive survey. Pattern Recognit 102:107205","journal-title":"Pattern Recognit"},{"key":"252_CR25","doi-asserted-by":"crossref","unstructured":"Sharif Razavian A, Azizpour H, Sullivan J et al (2014) CNN features off-the-shelf: an astounding baseline for recognition. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 806\u2013813","DOI":"10.1109\/CVPRW.2014.131"},{"key":"252_CR26","doi-asserted-by":"crossref","unstructured":"Quattoni A, Torralba A (2009) Recognizing indoor scenes. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 413\u2013420","DOI":"10.1109\/CVPR.2009.5206537"},{"key":"252_CR27","doi-asserted-by":"crossref","unstructured":"Xiao J, Hays J, Ehinger KA, et al (2010) Sun database: large-scale scene recognition from abbey to zoo. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 3485\u20133492","DOI":"10.1109\/CVPR.2010.5539970"},{"issue":"4","key":"252_CR28","doi-asserted-by":"publisher","first-page":"2028","DOI":"10.1109\/TIP.2017.2666739","volume":"26","author":"Z Wang","year":"2017","unstructured":"Wang Z, Wang L, Wang Y et al (2017) Weakly supervised patchnets: describing and aggregating local patches for scene recognition. IEEE Trans Image Process 26(4):2028\u20132041","journal-title":"IEEE Trans Image Process"},{"key":"252_CR29","unstructured":"Dixit MD, Vasconcelos N (2016) Object based scene representations using fisher scores of local subspace projections. In: Advances in neural information processing systems 29"},{"issue":"12","key":"252_CR30","doi-asserted-by":"publisher","first-page":"2335","DOI":"10.1109\/TPAMI.2017.2651061","volume":"39","author":"L Liu","year":"2017","unstructured":"Liu L, Wang P, Shen C et al (2017) Compositional model based fisher vector coding for image classification. IEEE Trans Pattern Anal Mach Intell 39(12):2335\u20132348","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"252_CR31","doi-asserted-by":"crossref","unstructured":"Li Y, Dixit M, Vasconcelos N (2017) Deep scene image classification with the MFAFVNet. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 5746\u20135754","DOI":"10.1109\/ICCV.2017.613"},{"key":"252_CR32","doi-asserted-by":"publisher","first-page":"339","DOI":"10.1016\/j.patcog.2017.10.039","volume":"76","author":"B Chen","year":"2018","unstructured":"Chen B, Li J, Wei G et al (2018) A novel localized and second order feature coding network for image recognition. Pattern Recognit 76:339\u2013348","journal-title":"Pattern Recognit"},{"key":"252_CR33","doi-asserted-by":"crossref","unstructured":"Sicre R, Avrithis Y, Kijak E et al (2017) Unsupervised part learning for visual recognition. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 6271\u20136279","DOI":"10.1109\/CVPR.2017.332"},{"key":"252_CR34","doi-asserted-by":"crossref","unstructured":"Khan SH, Hayat M, Porikli F (2017) Scene categorization with spectral features. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 5638\u20135648","DOI":"10.1109\/ICCV.2017.601"},{"key":"252_CR35","doi-asserted-by":"publisher","first-page":"5877","DOI":"10.1109\/TIP.2020.2986599","volume":"29","author":"G Chen","year":"2020","unstructured":"Chen G, Song X, Zeng H et al (2020) Scene recognition with prototype-agnostic scene layout. IEEE Trans Image Process 29:5877\u20135888","journal-title":"IEEE Trans Image Process"},{"key":"252_CR36","doi-asserted-by":"crossref","unstructured":"Qiu J, Yang Y, Wang X et al (2021) Scene essence. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 8322\u20138333","DOI":"10.1109\/CVPR46437.2021.00822"},{"key":"252_CR37","unstructured":"Touvron H, Cord M, Douze M, et al (2021) Training data-efficient image transformers & distillation through attention. In: International conference on machine learning. PMLR, pp 10347\u201310357"},{"key":"252_CR38","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2020.107256","volume":"102","author":"A L\u00f3pez-Cifuentes","year":"2020","unstructured":"L\u00f3pez-Cifuentes A, Escudero-Vi\u00f1lo M, Besc\u00f3s J et al (2020) Semantic-aware scene recognition. Pattern Recognit 102:107256","journal-title":"Pattern Recognit"},{"key":"252_CR39","doi-asserted-by":"crossref","unstructured":"Laranjeira C, Lacerda A, Nascimento ER (2019) On modeling context from objects with a long short-term memory for indoor scene recognition. In: 32nd SIBGRAPI conference on graphics, patterns and images, pp 249\u2013256","DOI":"10.1109\/SIBGRAPI.2019.00041"},{"key":"252_CR40","doi-asserted-by":"publisher","first-page":"141","DOI":"10.1109\/TMM.2020.3046877","volume":"24","author":"H Zeng","year":"2022","unstructured":"Zeng H, Song X, Chen G et al (2022) Amorphous region context modeling for scene recognition. IEEE Trans Multimed 24:141\u2013151","journal-title":"IEEE Trans Multimed"},{"issue":"20","key":"252_CR41","doi-asserted-by":"publisher","first-page":"4143","DOI":"10.3390\/rs13204143","volume":"13","author":"J Zhang","year":"2021","unstructured":"Zhang J, Zhao H, Li J (2021) TRS: transformers for remote sensing scene classification. Remote Sens 13(20):4143","journal-title":"Remote Sens"},{"issue":"6","key":"252_CR42","doi-asserted-by":"publisher","first-page":"1507","DOI":"10.3390\/rs14061507","volume":"14","author":"S Hao","year":"2022","unstructured":"Hao S, Wu B, Zhao K et al (2022) Two-stream swin transformer with differentiable sobel operator for remote sensing image classification. Remote Sens 14(6):1507","journal-title":"Remote Sens"},{"key":"252_CR43","first-page":"1","volume":"60","author":"P Lv","year":"2022","unstructured":"Lv P, Wu W, Zhong Y et al (2022) SCViT: a spatial-channel feature preserving vision transformer for remote sensing image scene classification. IEEE Trans Geosci Remote Sens 60:1\u201312","journal-title":"IEEE Trans Geosci Remote Sens"},{"key":"252_CR44","doi-asserted-by":"crossref","unstructured":"Liu Z, Lin Y, Cao Y, et al (2021) Swin transformer: hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 10012\u201310022","DOI":"10.1109\/ICCV48922.2021.00986"}],"container-title":["International Journal of Multimedia Information Retrieval"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13735-022-00252-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s13735-022-00252-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13735-022-00252-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,12,17]],"date-time":"2022-12-17T14:22:27Z","timestamp":1671286947000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s13735-022-00252-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,9,15]]},"references-count":44,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2022,12]]}},"alternative-id":["252"],"URL":"https:\/\/doi.org\/10.1007\/s13735-022-00252-7","relation":{},"ISSN":["2192-6611","2192-662X"],"issn-type":[{"type":"print","value":"2192-6611"},{"type":"electronic","value":"2192-662X"}],"subject":[],"published":{"date-parts":[[2022,9,15]]},"assertion":[{"value":"25 April 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"20 June 2022","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 August 2022","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 September 2022","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no known competing financial interests or personal relationships that could have appeared to influence the work reported in this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Interest Statement"}}]}}