{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,12]],"date-time":"2026-08-12T21:27:07Z","timestamp":1786570027752,"version":"build-2736575974"},"reference-count":30,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/100020595","name":"National Science and Technology Council","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100020595","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100006477","name":"National Taiwan University","doi-asserted-by":"publisher","award":["113-2634-F-002-002-"],"award-info":[{"award-number":["113-2634-F-002-002-"]}],"id":[{"id":"10.13039\/501100006477","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100006477","name":"National Taiwan University","doi-asserted-by":"publisher","award":["113-2223-E-002-006-"],"award-info":[{"award-number":["113-2223-E-002-006-"]}],"id":[{"id":"10.13039\/501100006477","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100006477","name":"National Taiwan University","doi-asserted-by":"publisher","award":["113-2221-E-002-127-MY3"],"award-info":[{"award-number":["113-2221-E-002-127-MY3"]}],"id":[{"id":"10.13039\/501100006477","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Journal of Visual Communication and Image Representation"],"published-print":{"date-parts":[[2026,9]]},"DOI":"10.1016\/j.jvcir.2026.104918","type":"journal-article","created":{"date-parts":[[2026,8,5]],"date-time":"2026-08-05T06:43:42Z","timestamp":1785912222000},"page":"104918","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Semantic guided image-to-point localization via monocular depth estimation in indoor environments"],"prefix":"10.1016","volume":"120","author":[{"given":"Jun-Jie","family":"Hu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kuan-Ting","family":"Lin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Li-Chen","family":"Fu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.jvcir.2026.104918_b1","doi-asserted-by":"crossref","unstructured":"R.B. Rusu, N. Blodow, M. Beetz, Fast Point Feature Histograms (FPFH) for 3D registration, in: 2009 IEEE International Conference on Robotics and Automation, 2009, pp. 3212\u20133217.","DOI":"10.1109\/ROBOT.2009.5152473"},{"key":"10.1016\/j.jvcir.2026.104918_b2","doi-asserted-by":"crossref","unstructured":"C. Choy, J. Park, V. Koltun, Fully Convolutional Geometric Features, in: 2019 IEEE\/CVF International Conference on Computer Vision, ICCV, 2019, pp. 8957\u20138965.","DOI":"10.1109\/ICCV.2019.00905"},{"key":"10.1016\/j.jvcir.2026.104918_b3","series-title":"Proceedings of the 30th ACM International Conference on Multimedia","first-page":"1630","article-title":"You only hypothesize once: Point cloud registration with rotation-equivariant descriptors","author":"Wang","year":"2022"},{"key":"10.1016\/j.jvcir.2026.104918_b4","doi-asserted-by":"crossref","unstructured":"D. DeTone, T. Malisiewicz, A. Rabinovich, SuperPoint: Self-Supervised Interest Point Detection and Description, in: 2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshops, CVPRW, 2018, pp. 337\u201333712.","DOI":"10.1109\/CVPRW.2018.00060"},{"key":"10.1016\/j.jvcir.2026.104918_b5","doi-asserted-by":"crossref","unstructured":"P.-E. Sarlin, D. DeTone, T. Malisiewicz, A. Rabinovich, SuperGlue: Learning Feature Matching With Graph Neural Networks, in: 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, 2020, pp. 4937\u20134946.","DOI":"10.1109\/CVPR42600.2020.00499"},{"key":"10.1016\/j.jvcir.2026.104918_b6","series-title":"The Thirteenth International Conference on Learning Representations","article-title":"Depth pro: Sharp monocular metric depth in less than a second","author":"Bochkovskiy","year":"2025"},{"key":"10.1016\/j.jvcir.2026.104918_b7","doi-asserted-by":"crossref","unstructured":"T. L\u00fcddecke, A. Ecker, Image Segmentation Using Text and Image Prompts, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, 2022, pp. 7086\u20137096.","DOI":"10.1109\/CVPR52688.2022.00695"},{"key":"10.1016\/j.jvcir.2026.104918_b8","series-title":"2019 International Conference on Robotics and Automation","first-page":"4790","article-title":"2D3d-matchnet: Learning to match keypoints across 2d image and 3d point cloud","author":"Feng","year":"2019"},{"key":"10.1016\/j.jvcir.2026.104918_b9","series-title":"Proceedings of the Seventh IEEE International Conference on Computer Vision","first-page":"1150","article-title":"Object recognition from local scale-invariant features","volume":"Vol. 2","author":"Lowe","year":"1999"},{"key":"10.1016\/j.jvcir.2026.104918_b10","series-title":"2009 IEEE 12th International Conference on Computer Vision Workshops, ICCV Workshops","first-page":"689","article-title":"Intrinsic shape signatures: A shape descriptor for 3D object recognition","author":"Zhong","year":"2009"},{"key":"10.1016\/j.jvcir.2026.104918_b11","doi-asserted-by":"crossref","unstructured":"J. Li, G.H. Lee, DeepI2P: Image-to-point cloud registration via deep classification, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2021, pp. 15960\u201315969.","DOI":"10.1109\/CVPR46437.2021.01570"},{"issue":"3","key":"10.1016\/j.jvcir.2026.104918_b12","doi-asserted-by":"crossref","first-page":"1198","DOI":"10.1109\/TCSVT.2022.3208859","article-title":"CorrI2P: Deep image-to-point cloud registration via dense correspondence","volume":"33","author":"Ren","year":"2022","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.jvcir.2026.104918_b13","doi-asserted-by":"crossref","DOI":"10.1109\/LRA.2024.3466068","article-title":"CoFiI2P: Coarse-to-fine correspondences-based image to point cloud registration","author":"Kang","year":"2024","journal-title":"IEEE Robot. Autom. Lett."},{"key":"10.1016\/j.jvcir.2026.104918_b14","series-title":"ICLR","article-title":"FreeReg: Image-to-point cloud registration leveraging pretrained diffusion models and monocular depth estimators","author":"Wang","year":"2024"},{"key":"10.1016\/j.jvcir.2026.104918_b15","series-title":"European Conference on Computer Vision","first-page":"407","article-title":"Is geometry enough for matching in visual localization?","author":"Zhou","year":"2022"},{"key":"10.1016\/j.jvcir.2026.104918_b16","unstructured":"H. Ye, Y. Liu, Y. Liu, S. Shen, PlanaReLoc: Camera Relocalization in 3D Planar Primitives via Region-Based Structure Matching, in: IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR, 2026."},{"key":"10.1016\/j.jvcir.2026.104918_b17","doi-asserted-by":"crossref","first-page":"123","DOI":"10.1016\/j.isprsjprs.2019.10.009","article-title":"NRLI-UAV: Non-rigid registration of sequential raw laser scans and images for low-cost UAV LiDAR point cloud quality improvement","volume":"158","author":"Li","year":"2019","journal-title":"ISPRS J. Photogramm. Remote Sens."},{"key":"10.1016\/j.jvcir.2026.104918_b18","doi-asserted-by":"crossref","first-page":"22","DOI":"10.1016\/j.isprsjprs.2026.03.036","article-title":"AGI2P: Benchmarking aerial\u2013ground image-to-point cloud localization with a large-scale dataset","volume":"236","author":"Yang","year":"2026","journal-title":"ISPRS J. Photogramm. Remote Sens."},{"key":"10.1016\/j.jvcir.2026.104918_b19","doi-asserted-by":"crossref","first-page":"476","DOI":"10.1016\/j.isprsjprs.2026.01.006","article-title":"AEOS: Active environment-aware optimal scanning control for UAV LiDAR-inertial odometry in complex scenes","volume":"232","author":"Li","year":"2026","journal-title":"ISPRS J. Photogramm. Remote Sens."},{"issue":"2","key":"10.1016\/j.jvcir.2026.104918_b20","doi-asserted-by":"crossref","first-page":"1883","DOI":"10.1109\/LRA.2024.3349915","article-title":"IG-LIO: An incremental GICP-based tightly-coupled LiDAR-Inertial odometry","volume":"9","author":"Chen","year":"2024","journal-title":"IEEE Robot. Autom. Lett."},{"key":"10.1016\/j.jvcir.2026.104918_b21","doi-asserted-by":"crossref","DOI":"10.1007\/s10514-012-9321-0","article-title":"OctoMap: An efficient probabilistic 3D mapping framework based on octrees","author":"Hornung","year":"2013","journal-title":"Auton. Robots"},{"key":"10.1016\/j.jvcir.2026.104918_b22","doi-asserted-by":"crossref","DOI":"10.1016\/j.jvcir.2024.104252","article-title":"Versatile depth estimator based on common relative depth estimation and camera-specific relative-to-metric depth conversion","volume":"103","author":"Jun","year":"2024","journal-title":"J. Vis. Commun. Image Represent."},{"key":"10.1016\/j.jvcir.2026.104918_b23","series-title":"Digital Image Processing","author":"Pratt","year":"1978"},{"issue":"2","key":"10.1016\/j.jvcir.2026.104918_b24","doi-asserted-by":"crossref","first-page":"129","DOI":"10.1109\/TIT.1982.1056489","article-title":"Least squares quantization in PCM","volume":"28","author":"Lloyd","year":"1982","journal-title":"IEEE Trans. Inform. Theory"},{"key":"10.1016\/j.jvcir.2026.104918_b25","series-title":"Computer Vision \u2013 ECCV 2016","first-page":"766","article-title":"Fast global registration","author":"Zhou","year":"2016"},{"key":"10.1016\/j.jvcir.2026.104918_b26","series-title":"Computer Vision: A Reference Guide","first-page":"433","article-title":"Iterative closest point (ICP)","author":"Zhang","year":"2014"},{"issue":"2","key":"10.1016\/j.jvcir.2026.104918_b27","doi-asserted-by":"crossref","first-page":"155","DOI":"10.1007\/s11263-008-0152-6","article-title":"EPnP: An accurate O(n) solution to the PnP problem","volume":"81","author":"Lepetit","year":"2009","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.jvcir.2026.104918_b28","series-title":"Ceres solver","author":"Agarwal","year":"2023"},{"key":"10.1016\/j.jvcir.2026.104918_b29","series-title":"FreeReg: Feature matching-free image-to-point cloud registration","author":"WHU-USI3DV","year":"2024"},{"key":"10.1016\/j.jvcir.2026.104918_b30","series-title":"Proc. Computer Vision and Pattern Recognition","article-title":"ScanNet: Richly-annotated 3D reconstructions of indoor scenes","author":"Dai","year":"2017"}],"container-title":["Journal of Visual Communication and Image Representation"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1047320326002130?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1047320326002130?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,8,12]],"date-time":"2026-08-12T20:28:55Z","timestamp":1786566535000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S1047320326002130"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,9]]},"references-count":30,"alternative-id":["S1047320326002130"],"URL":"https:\/\/doi.org\/10.1016\/j.jvcir.2026.104918","relation":{},"ISSN":["1047-3203"],"issn-type":[{"value":"1047-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,9]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Semantic guided image-to-point localization via monocular depth estimation in indoor environments","name":"articletitle","label":"Article Title"},{"value":"Journal of Visual Communication and Image Representation","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.jvcir.2026.104918","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Inc. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"104918"}}