{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T17:03:26Z","timestamp":1785603806749,"version":"3.56.0"},"publisher-location":"Cham","reference-count":46,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031733369","type":"print"},{"value":"9783031733376","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T00:00:00Z","timestamp":1730332800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T00:00:00Z","timestamp":1730332800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-73337-6_26","type":"book-chapter","created":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T23:02:27Z","timestamp":1730329347000},"page":"460-477","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["SCPNet: Unsupervised Cross-Modal Homography Estimation via\u00a0Intra-modal Self-supervised Learning"],"prefix":"10.1007","author":[{"given":"Runmin","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jun","family":"Ma","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Si-Yuan","family":"Cao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lun","family":"Luo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Beinan","family":"Yu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shu-Jie","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Junwei","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hui-Liang","family":"Shen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,10,31]]},"reference":[{"key":"26_CR1","doi-asserted-by":"crossref","unstructured":"Aguilera, C.A., Sappa, A.D., Toledo, R.: LGHD: a feature descriptor for matching across non-linear intensity variations. In: Proceedings of the IEEE International Conference on Image Processing, pp. 178\u2013181. IEEE (2015)","DOI":"10.1109\/ICIP.2015.7350783"},{"key":"26_CR2","doi-asserted-by":"crossref","unstructured":"Arar, M., Ginger, Y., Danon, D., Bermano, A.H., Cohen-Or, D.: Unsupervised multi-modal image registration via geometry preserving image-to-image translation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13410\u201313419 (2020)","DOI":"10.1109\/CVPR42600.2020.01342"},{"key":"26_CR3","doi-asserted-by":"crossref","unstructured":"Barath, D., Matas, J., Noskova, J.: MAGSAC: marginalizing sample consensus. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10197\u201310205 (2019)","DOI":"10.1109\/CVPR.2019.01044"},{"key":"26_CR4","doi-asserted-by":"publisher","unstructured":"Bay, H., Tuytelaars, T., Van\u00a0Gool, L.: SURF: speeded up robust features. In: Proceedings of the European Conference on Computer Vision, pp. 404\u2013417. Springer (2006). https:\/\/doi.org\/10.1007\/11744023_32","DOI":"10.1007\/11744023_32"},{"key":"26_CR5","doi-asserted-by":"crossref","unstructured":"Brown, M., S\u00fcsstrunk, S.: Multi-spectral SIFT for scene category recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 177\u2013184 (2011)","DOI":"10.1109\/CVPR.2011.5995637"},{"key":"26_CR6","doi-asserted-by":"crossref","unstructured":"Cao, S.Y., Hu, J., Sheng, Z., Shen, H.L.: Iterative deep homography estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1879\u20131888 (2022)","DOI":"10.1109\/CVPR52688.2022.00192"},{"key":"26_CR7","doi-asserted-by":"publisher","first-page":"5147","DOI":"10.1109\/TIP.2020.2980972","volume":"29","author":"SY Cao","year":"2020","unstructured":"Cao, S.Y., Shen, H.L., Chen, S.J., Li, C.: Boosting structure consistency for multispectral and multimodal image registration. IEEE Trans. Image Process. 29, 5147\u20135162 (2020)","journal-title":"IEEE Trans. Image Process."},{"key":"26_CR8","doi-asserted-by":"crossref","unstructured":"Cao, S.Y., Zhang, R., Luo, L., Yu, B., Sheng, Z., Li, J., Shen, H.L.: Recurrent homography estimation using homography-guided image warping and focus transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9833\u20139842 (2023)","DOI":"10.1109\/CVPR52729.2023.00948"},{"key":"26_CR9","doi-asserted-by":"crossref","unstructured":"Caruana, R.: Multitask learning. Mach. learn. 28, 41\u201375 (1997)","DOI":"10.1023\/A:1007379606734"},{"key":"26_CR10","doi-asserted-by":"crossref","unstructured":"Chakrabarti, A., Zickler, T.: Statistics of real-world hyperspectral images. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 193\u2013200 (2011)","DOI":"10.1109\/CVPR.2011.5995660"},{"issue":"3","key":"26_CR11","doi-asserted-by":"publisher","first-page":"1297","DOI":"10.1109\/TIP.2017.2776753","volume":"27","author":"SJ Chen","year":"2017","unstructured":"Chen, S.J., Shen, H.L., Li, C., Xin, J.H.: Normalized total gradient: a new measure for multispectral image registration. IEEE Trans. Image Process. 27(3), 1297\u20131310 (2017)","journal-title":"IEEE Trans. Image Process."},{"key":"26_CR12","unstructured":"DeTone, D., Malisiewicz, T., Rabinovich, A.: Deep image homography estimation. arXiv preprint arXiv:1606.03798 (2016)"},{"key":"26_CR13","doi-asserted-by":"crossref","unstructured":"Dharejo, F.A., et al.: Multimodal-boost: multimodal medical image super-resolution using multi-attention network with wavelet transform. IEEE\/ACM Trans. Comput. Biol. Bioinform. (2022)","DOI":"10.1109\/TCBB.2022.3191387"},{"key":"26_CR14","unstructured":"Di, W., Jinyuan, L., Xin, F., Liu, R.: Unsupervised misaligned infrared and visible image fusion via cross-modality image generation and registration. In: International Joint Conference on Artificial Intelligence (2022)"},{"key":"26_CR15","doi-asserted-by":"crossref","unstructured":"Doersch, C., Zisserman, A.: Multi-task self-supervised visual learning. In: Proceedings of the IEEE international conference on computer vision, pp. 2051\u20132060 (2017)","DOI":"10.1109\/ICCV.2017.226"},{"key":"26_CR16","unstructured":"Dubrofsky, E.: Homography estimation. Diplomov\u00e1 pr\u00e1ce. Vancouver: Univerzita Britsk\u00e9 Kolumbie 5 (2009)"},{"issue":"6","key":"26_CR17","doi-asserted-by":"publisher","first-page":"381","DOI":"10.1145\/358669.358692","volume":"24","author":"MA Fischler","year":"1981","unstructured":"Fischler, M.A., Bolles, R.C.: Random sample consensus: a paradigm for model fitting with applications to image analysis and automated cartography. Commun. ACM 24(6), 381\u2013395 (1981)","journal-title":"Commun. ACM"},{"key":"26_CR18","doi-asserted-by":"crossref","unstructured":"Girdhar, R., Singh, M., Ravi, N., Van Der\u00a0Maaten, L., Joulin, A., Misra, I.: Omnivore: a single model for many visual modalities. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16102\u201316112 (2022)","DOI":"10.1109\/CVPR52688.2022.01563"},{"key":"26_CR19","doi-asserted-by":"crossref","unstructured":"Goforth, H., Lucey, S.: GPS-denied UAV localization using pre-existing satellite imagery. In: 2019 International Conference on Robotics and Automation, pp. 2974\u20132980. IEEE (2019)","DOI":"10.1109\/ICRA.2019.8793558"},{"key":"26_CR20","unstructured":"Goodfellow, I., et al.: Generative adversarial nets. Adv. Neural Inf. Process. Syst. 27 (2014)"},{"key":"26_CR21","doi-asserted-by":"publisher","unstructured":"He, S., Lau, R.W.: Saliency detection with flash and no-flash image pairs. In: Proceedings of the European Conference on Computer Vision, pp. 110\u2013124. Springer (2014). https:\/\/doi.org\/10.1007\/978-3-319-10578-9_8","DOI":"10.1007\/978-3-319-10578-9_8"},{"issue":"9","key":"26_CR22","doi-asserted-by":"publisher","first-page":"813","DOI":"10.1080\/03610927708827533","volume":"6","author":"PW Holland","year":"1977","unstructured":"Holland, P.W., Welsch, R.E.: Robust regression using iteratively reweighted least-squares. Commun. Stat. theory Methods 6(9), 813\u2013827 (1977)","journal-title":"Commun. Stat. theory Methods"},{"key":"26_CR23","doi-asserted-by":"crossref","unstructured":"Hong, M., Lu, Y., Ye, N., Lin, C., Zhao, Q., Liu, S.: Unsupervised homography estimation with coplanarity-aware GAN. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 17663\u201317672 (2022)","DOI":"10.1109\/CVPR52688.2022.01714"},{"key":"26_CR24","doi-asserted-by":"crossref","unstructured":"Hu, R., Singh, A.: Unit: multimodal multitask learning with a unified transformer. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp. 1439\u20131449 (2021)","DOI":"10.1109\/ICCV48922.2021.00147"},{"key":"26_CR25","doi-asserted-by":"publisher","unstructured":"Johnson, J., Alahi, A., Fei-Fei, L.: Perceptual losses for real-time style transfer and super-resolution. In: Proceedings of the European Conference on Computer Vision, pp. 694\u2013711. Springer (2016). https:\/\/doi.org\/10.1007\/978-3-319-46475-6_43","DOI":"10.1007\/978-3-319-46475-6_43"},{"key":"26_CR26","doi-asserted-by":"crossref","unstructured":"Kim, S., Min, D., Ham, B., Ryu, S., Do, M.N., Sohn, K.: DASC: dense adaptive self-correlation descriptor for multi-modal and multi-spectral correspondence. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2103\u20132112 (2015)","DOI":"10.1109\/CVPR.2015.7298822"},{"key":"26_CR27","doi-asserted-by":"crossref","unstructured":"Koguciuk, D., Arani, E., Zonooz, B.: Perceptual loss for robust unsupervised homography estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4274\u20134283 (2021)","DOI":"10.1109\/CVPRW53098.2021.00483"},{"key":"26_CR28","doi-asserted-by":"crossref","unstructured":"Le, H., Liu, F., Zhang, S., Agarwala, A.: Deep homography estimation for dynamic scenes. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7652\u20137661 (2020)","DOI":"10.1109\/CVPR42600.2020.00767"},{"key":"26_CR29","doi-asserted-by":"publisher","first-page":"3296","DOI":"10.1109\/TIP.2019.2959244","volume":"29","author":"J Li","year":"2019","unstructured":"Li, J., Hu, Q., Ai, M.: RIFT: multi-modal image matching based on radiation-variation insensitive feature transform. IEEE Trans. Image Process. 29, 3296\u20133310 (2019)","journal-title":"IEEE Trans. Image Process."},{"key":"26_CR30","doi-asserted-by":"publisher","unstructured":"Lin, T.Y., et al.: Microsoft COCO: common objects in context. In: Proceedings of the European Conference on Computer Vision, pp. 740\u2013755. Springer (2014). https:\/\/doi.org\/10.1007\/978-3-319-10602-1_48","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"26_CR31","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101 (2017)"},{"issue":"2","key":"26_CR32","doi-asserted-by":"publisher","first-page":"91","DOI":"10.1023\/B:VISI.0000029664.99615.94","volume":"60","author":"DG Lowe","year":"2004","unstructured":"Lowe, D.G.: Distinctive image features from scale-invariant keypoints. Int. J. Comput. Vision 60(2), 91\u2013110 (2004)","journal-title":"Int. J. Comput. Vision"},{"issue":"9","key":"26_CR33","doi-asserted-by":"publisher","first-page":"5830","DOI":"10.1109\/TCSVT.2022.3163649","volume":"32","author":"I Marivani","year":"2022","unstructured":"Marivani, I., Tsiligianni, E., Cornelis, B., Deligiannis, N.: Designing CNNs for multimodal image restoration and fusion via unfolding the method of multipliers. IEEE Trans. Circuits Syst. Video Technol. 32(9), 5830\u20135845 (2022)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"issue":"5","key":"26_CR34","doi-asserted-by":"publisher","first-page":"1147","DOI":"10.1109\/TRO.2015.2463671","volume":"31","author":"R Mur-Artal","year":"2015","unstructured":"Mur-Artal, R., Montiel, J.M.M., Tardos, J.D.: ORB-SLAM: a versatile and accurate monocular SLAM system. IEEE Trans. Rob. 31(5), 1147\u20131163 (2015)","journal-title":"IEEE Trans. Rob."},{"issue":"3","key":"26_CR35","doi-asserted-by":"publisher","first-page":"2346","DOI":"10.1109\/LRA.2018.2809549","volume":"3","author":"T Nguyen","year":"2018","unstructured":"Nguyen, T., Chen, S.W., Shivakumar, S.S., Taylor, C.J., Kumar, V.: Unsupervised deep homography: a fast and robust homography estimation model. IEEE Robot. Autom. Lett. 3(3), 2346\u20132353 (2018)","journal-title":"IEEE Robot. Autom. Lett."},{"key":"26_CR36","doi-asserted-by":"crossref","unstructured":"Rublee, E., Rabaud, V., Konolige, K., Bradski, G.: ORB: an efficient alternative to sift or surf. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2564\u20132571 (2011)","DOI":"10.1109\/ICCV.2011.6126544"},{"key":"26_CR37","doi-asserted-by":"crossref","unstructured":"Shao, R., Wu, G., Zhou, Y., Fu, Y., Fang, L., Liu, Y.: LocalTrans: a multiscale local transformer network for cross-resolution homography estimation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 14890\u201314899 (2021)","DOI":"10.1109\/ICCV48922.2021.01462"},{"key":"26_CR38","doi-asserted-by":"publisher","first-page":"355","DOI":"10.1016\/j.patrec.2019.09.021","volume":"128","author":"C Wang","year":"2019","unstructured":"Wang, C., Wang, X., Bai, X., Liu, Y., Zhou, J.: Self-supervised deep homography estimation with invertibility constraints. Pattern Recogn. Lett. 128, 355\u2013360 (2019)","journal-title":"Pattern Recogn. Lett."},{"key":"26_CR39","doi-asserted-by":"crossref","unstructured":"Xu, H., Ma, J., Yuan, J., Le, Z., Liu, W.: RFNet: unsupervised network for mutually reinforcing multi-modal image registration and fusion. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 19679\u201319688 (2022)","DOI":"10.1109\/CVPR52688.2022.01906"},{"issue":"9","key":"26_CR40","doi-asserted-by":"publisher","first-page":"2241","DOI":"10.1109\/TIP.2010.2046811","volume":"19","author":"F Yasuma","year":"2010","unstructured":"Yasuma, F., Mitsunaga, T., Iso, D., Nayar, S.K.: Generalized assorted pixel camera: postcapture control of resolution, dynamic range, and spectrum. IEEE Trans. Image Process. 19(9), 2241\u20132253 (2010)","journal-title":"IEEE Trans. Image Process."},{"key":"26_CR41","doi-asserted-by":"crossref","unstructured":"Ye, N., Wang, C., Fan, H., Liu, S.: Motion basis learning for unsupervised deep homography estimation with subspace projection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 13117\u201313125 (2021)","DOI":"10.1109\/ICCV48922.2021.01287"},{"key":"26_CR42","first-page":"1","volume":"60","author":"Y Ye","year":"2022","unstructured":"Ye, Y., Tang, T., Zhu, B., Yang, C., Li, B., Hao, S.: A multiscale framework with unsupervised learning for remote sensing image registration. IEEE Trans. Geosci. Remote Sens. 60, 1\u201315 (2022)","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"26_CR43","doi-asserted-by":"crossref","unstructured":"Ying, J., Shen, H.L., Cao, S.Y.: Unaligned hyperspectral image fusion via registration and interpolation modeling. IEEE Trans. Geosci. Remote Sens. (2021)","DOI":"10.1109\/TGRS.2021.3081136"},{"key":"26_CR44","doi-asserted-by":"publisher","unstructured":"Zhang, J., et al.: Content-aware unsupervised deep homography estimation. In: Proceedings of the European Conference on Computer Vision, pp. 653\u2013669. Springer (2020). https:\/\/doi.org\/10.48550\/arXiv.1909.05983","DOI":"10.48550\/arXiv.1909.05983"},{"key":"26_CR45","doi-asserted-by":"crossref","unstructured":"Zhao, Y., Huang, X., Zhang, Z.: Deep Lucas-Kanade homography for multimodal image alignment. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15950\u201315959 (2021)","DOI":"10.1109\/CVPR46437.2021.01569"},{"issue":"5","key":"26_CR46","doi-asserted-by":"publisher","first-page":"3020","DOI":"10.1109\/TGRS.2019.2946803","volume":"58","author":"Y Zhou","year":"2019","unstructured":"Zhou, Y., Rangarajan, A., Gader, P.D.: An integrated approach to registration and fusion of hyperspectral and multispectral images. IEEE Trans. Geosci. Remote Sens. 58(5), 3020\u20133033 (2019)","journal-title":"IEEE Trans. Geosci. Remote Sens."}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-73337-6_26","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T23:07:43Z","timestamp":1730329663000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-73337-6_26"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,31]]},"ISBN":["9783031733369","9783031733376"],"references-count":46,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-73337-6_26","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,31]]},"assertion":[{"value":"31 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}