{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T15:37:04Z","timestamp":1782833824516,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":34,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,26]],"date-time":"2023-10-26T00:00:00Z","timestamp":1698278400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,26]]},"DOI":"10.1145\/3581783.3613815","type":"proceedings-article","created":{"date-parts":[[2023,10,27]],"date-time":"2023-10-27T07:27:12Z","timestamp":1698391632000},"page":"3003-3011","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":7,"title":["CTCP: Cross Transformer and CNN for Pansharpening"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-5847-1714","authenticated-orcid":false,"given":"Zhao","family":"Su","sequence":"first","affiliation":[{"name":"Jiangxi University of Finance and Economics, Nanchang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9467-0942","authenticated-orcid":false,"given":"Yong","family":"Yang","sequence":"additional","affiliation":[{"name":"School of Computer Science and Technology, Tiangong University, Tianjin, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2771-8461","authenticated-orcid":false,"given":"Shuying","family":"Huang","sequence":"additional","affiliation":[{"name":"School of Software, Tiangong University, Tianjin, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3537-979X","authenticated-orcid":false,"given":"Weiguo","family":"Wan","sequence":"additional","affiliation":[{"name":"School of Software and Internet of Things Engineering, Jiangxi University of Finance and Economics, Nanchang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2673-2192","authenticated-orcid":false,"given":"Wei","family":"Tu","sequence":"additional","affiliation":[{"name":"School of Big Data Science, Jiangxi Science and Technology Normal University, Nanchang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2829-8553","authenticated-orcid":false,"given":"Hangyuan","family":"Lu","sequence":"additional","affiliation":[{"name":"College of Information Engineering, Jinhua Polytechnic, Jinhua, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7216-427X","authenticated-orcid":false,"given":"Changjie","family":"Chen","sequence":"additional","affiliation":[{"name":"School of Information Technology, Jiangxi University of Finance and Economics, Nanchang, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2023,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/MGRS.2021.3088865"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2019.2930982"},{"key":"e_1_3_2_1_3_1","volume-title":"Jon Atli Benediktsson, and Yudong Lu","author":"Cui Guoqing","year":"2018","unstructured":"Guoqing Cui, Zhiyong Lv, Guangfei Li, Jon Atli Benediktsson, and Yudong Lu. 2018. Refining Land Cover Classification Maps Based on Dual-Adaptive Majority Voting Strategy for Very High Resolution Remote Sensing Images. Remote Sensing 10, 8, (2018)."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2021.3137020"},{"key":"e_1_3_2_1_5_1","volume-title":"Remote Sensing 8, 7","author":"Masi Giuseppe","year":"2016","unstructured":"Giuseppe Masi, Davide Cozzolino, Luisa Verdoliva, and Giuseppe Scarpa. 2016.Pansharpening by Convolutional Neural Networks. Remote Sensing 8, 7 (2016)."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/LGRS.2017.2736020"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/JSTARS.2018.2794888"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2020.3031366"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2020.3010441"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2022.3179449"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2021.3098752"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/LGRS.2022.3179473"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i3.20267"},{"key":"e_1_3_2_1_14_1","volume-title":"Proceedings of the Conference on Neural Information Processing Systems (NIPS). 5998--6008","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N. Gomez, Lukasz Kaiser, and Illia. Polosukhin. 2017. Attention is All You Need. In Proceedings of the Conference on Neural Information Processing Systems (NIPS). 5998--6008."},{"key":"e_1_3_2_1_15_1","volume-title":"Proceedings of the Annual Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (NAACL-HLT). 4171--4186","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Mingwei Chang, Kenton Lee, and Kristina Toutanova. 2019. Bert: Pre-Training of Deep Bidirectional Transformers for Language Understanding. In Proceedings of the Annual Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (NAACL-HLT). 4171--4186."},{"key":"e_1_3_2_1_16_1","volume-title":"Proceedings of the Conference on Neural Information Processing Systems (NIPS). 1877--1901","author":"Tom","unstructured":"Tom B. Brown et al. 2020. Language Models Are Few-Shot Learners. In Proceedings of the Conference on Neural Information Processing Systems (NIPS). 1877--1901."},{"key":"e_1_3_2_1_17_1","volume-title":"Proceedings of the International Conference on Learning Representations (ICLR). 1--21","author":"Dosovitskiy Alexey","year":"2021","unstructured":"Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, Xiaohua Zhai, Thomas Unterthiner, Mostafa Dehghani, Matthias Minderer, Georg Heigold, Sylvain Gelly, Jakob Uszkoreit, and Neil Houlsby, 2021. An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. In Proceedings of the International Conference on Learning Representations (ICLR). 1--21."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2022.3168465"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2021.3137967"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2022.3152425"},{"key":"e_1_3_2_1_21_1","volume-title":"Pan-Sharpening Based on CNN+ Pyramid Transformer by Using No-Reference Loss. Remote Sensing 14","author":"Li Sijia","year":"2022","unstructured":"Sijia Li, Qing Guo, and An Li. 2022. Pan-Sharpening Based on CNN+ Pyramid Transformer by Using No-Reference Loss. Remote Sensing 14 (2022)."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2023.3239013"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00358"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00181"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2018.2857824"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00082"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00042"},{"key":"e_1_3_2_1_28_1","volume-title":"Xception: Deep Learning with Depthwise Separable Convolutions. In Proceedings of the IEEE\/CVF Conferenceon Computer Vision and Pattern Recognition (CVPR). 1800--1807","author":"Chollet Fran\u00e7ois","unstructured":"Fran\u00e7ois Chollet. .2017. Xception: Deep Learning with Depthwise Separable Convolutions. In Proceedings of the IEEE\/CVF Conferenceon Computer Vision and Pattern Recognition (CVPR). 1800--1807."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/TGRS.2007.901007"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2018.2819501"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01003"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547774"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547924"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1177\/001316446002000104"}],"event":{"name":"MM '23: The 31st ACM International Conference on Multimedia","location":"Ottawa ON Canada","acronym":"MM '23","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 31st ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3613815","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3581783.3613815","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T00:01:07Z","timestamp":1755820867000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3613815"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,26]]},"references-count":34,"alternative-id":["10.1145\/3581783.3613815","10.1145\/3581783"],"URL":"https:\/\/doi.org\/10.1145\/3581783.3613815","relation":{},"subject":[],"published":{"date-parts":[[2023,10,26]]},"assertion":[{"value":"2023-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}