{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,21]],"date-time":"2026-05-21T17:17:07Z","timestamp":1779383827004,"version":"3.53.1"},"reference-count":54,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2022YFC3803700"],"award-info":[{"award-number":["2022YFC3803700"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62432002"],"award-info":[{"award-number":["62432002"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.patcog.2026.113422","type":"journal-article","created":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T23:57:35Z","timestamp":1772841455000},"page":"113422","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Physics-informed visual-inertial mamba for robust train localization in harsh conditions"],"prefix":"10.1016","volume":"178","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-1851-6551","authenticated-orcid":false,"given":"Xiaoyu","family":"Xian","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1532-6551","authenticated-orcid":false,"given":"Qiuyang","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5206-3711","authenticated-orcid":false,"given":"Yin","family":"Tian","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7796-5650","authenticated-orcid":false,"given":"Daxin","family":"Tian","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5331-6162","authenticated-orcid":false,"given":"Jianshan","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"issue":"3","key":"10.1016\/j.patcog.2026.113422_bib0001","doi-asserted-by":"crossref","first-page":"2182","DOI":"10.1109\/TITS.2023.3319135","article-title":"Machine learning in urban rail transit systems: a survey","volume":"25","author":"Zhu","year":"2023","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"key":"10.1016\/j.patcog.2026.113422_bib0002","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2023.109442","article-title":"Efficient large-scale oblique image matching based on cascade hashing and match data scheduling","volume":"138","author":"Zhang","year":"2023","journal-title":"Pattern Recognit."},{"issue":"1","key":"10.1016\/j.patcog.2026.113422_bib0003","doi-asserted-by":"crossref","first-page":"20889","DOI":"10.1109\/TITS.2024.3449892","article-title":"Road semantic-enhanced land vehicle integrated navigation in GNSS denied environments","volume":"25","author":"Zhang","year":"2024","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"issue":"6","key":"10.1016\/j.patcog.2026.113422_bib0004","doi-asserted-by":"crossref","first-page":"7908","DOI":"10.1109\/TVT.2024.3360076","article-title":"Enhancing positioning in GNSS denied environments based on an extended Kalman filter using past GNSS measurements and IMU","volume":"73","author":"Iyer","year":"2024","journal-title":"IEEE Trans. Veh. Technol."},{"key":"10.1016\/j.patcog.2026.113422_bib0005","article-title":"Lane-level map-aided SLAM approach for persistent positioning in GNSS-Denied areas","volume":"74","author":"Liu","year":"2025","journal-title":"IEEE Trans. Instrum. Meas."},{"key":"10.1016\/j.patcog.2026.113422_bib0006","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2020.107334","article-title":"Robust one-stage object detection with location-aware classifiers","volume":"105","author":"Chen","year":"2020","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113422_bib0007","series-title":"2024 IEEE 27th International Conference on Intelligent Transportation Systems (ITSC)","first-page":"3197","article-title":"A seamless train positioning method based on visual place recognition","author":"Liu","year":"2024"},{"key":"10.1016\/j.patcog.2026.113422_bib0008","article-title":"Dynamic clustering transformer for LiDAR-based 3D object detection","volume":"172","author":"Cui","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113422_bib0009","series-title":"2025 11th International Conference on Automation, Robotics, and Applications (ICARA)","first-page":"290","article-title":"Autonomous navigation systems in GPS-denied environments: a review of techniques and applications","author":"Alghamdi","year":"2025"},{"key":"10.1016\/j.patcog.2026.113422_bib0010","series-title":"Proceedings 2007 IEEE International Conference on Robotics and Automation","first-page":"3565","article-title":"A multi-state constraint Kalman filter for vision-aided inertial navigation","author":"Mourikis","year":"2007"},{"issue":"4","key":"10.1016\/j.patcog.2026.113422_bib0011","doi-asserted-by":"crossref","first-page":"1004","DOI":"10.1109\/TRO.2018.2853729","article-title":"VINS-Mono: a robust and versatile monocular visual-inertial state estimator","volume":"34","author":"Qin","year":"2018","journal-title":"IEEE Trans. Rob."},{"key":"10.1016\/j.patcog.2026.113422_bib0012","doi-asserted-by":"crossref","first-page":"610","DOI":"10.1109\/TIP.2023.3348293","article-title":"High-similarity-pass attention for single image super-resolution","volume":"33","author":"Su","year":"2024","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.patcog.2026.113422_bib0013","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"821","article-title":"Libra R-CNN: towards balanced learning for object detection","author":"Pang","year":"2019"},{"key":"10.1016\/j.patcog.2026.113422_bib0014","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"14454","article-title":"Sparse R-CNN: end-to-end object detection with learnable proposals","author":"Sun","year":"2021"},{"key":"10.1016\/j.patcog.2026.113422_bib0015","article-title":"YOLO-FCE: a feature and clustering enhanced object detection model for species classification","volume":"171","author":"Zhang","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113422_bib0016","unstructured":"C. Lyu, W. Zhang, H. Huang, Y. Zhou, Y. Wang, Y. Liu, S. Zhang, K. Chen, RTMDet: an empirical study of designing real-time object detectors, 2022. arXiv: 2212.07784."},{"key":"10.1016\/j.patcog.2026.113422_bib0017","series-title":"European Conference on Computer Vision","first-page":"213","article-title":"End-to-end object detection with transformers","author":"Carion","year":"2020"},{"key":"10.1016\/j.patcog.2026.113422_bib0018","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110648","article-title":"Prompt-guided DETR with RoI-pruned masked attention for open-vocabulary object detection","volume":"155","author":"Song","year":"2024","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113422_bib0019","unstructured":"H. Zhang, F. Li, S. Liu, L. Zhang, H. Su, J. Zhu, L.M. Ni, H.-Y. Shum, DINO: DETR with improved DeNoising anchor boxes for end-to-end object detection, 2022. arXiv: 2203.03605."},{"key":"10.1016\/j.patcog.2026.113422_bib0020","doi-asserted-by":"crossref","unstructured":"S. Liu, Z. Zeng, T. Ren, F. Li, H. Zhang, J. Yang, Q. Jiang, C. Li, J. Yang, H. Su, J. Zhu, L. Zhang, Grounding DINO: marrying DINO with grounded pre-training for open-set object detection, 2023. arXiv: 2303.05499.","DOI":"10.1007\/978-3-031-72970-6_3"},{"key":"10.1016\/j.patcog.2026.113422_bib0021","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","article-title":"DETRs beat YOLOs on real-time object detection","author":"Zhao","year":"2024"},{"key":"10.1016\/j.patcog.2026.113422_bib0022","unstructured":"I. Robinson, P. Robicheaux, M. Popov, D. Ramanan, N. Peri, RF-DETR: neural architecture search for real-time detection transformers, 2025. arXiv: 2511.09554."},{"key":"10.1016\/j.patcog.2026.113422_bib0023","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.111201","article-title":"Automatic cervical cancer classification using adaptive vision transformer encoder with CNN for medical application","volume":"160","author":"Nirmala","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113422_bib0024","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110491","article-title":"UCTNet: uncertainty-guided CNN-transformer hybrid networks for medical image segmentation","volume":"152","author":"Guo","year":"2024","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113422_bib0025","first-page":"581","article-title":"Guiding monocular depth estimation using depth-attention volume","author":"Huynh","year":"2020","journal-title":"European Conference on Computer Vision"},{"key":"10.1016\/j.patcog.2026.113422_bib0026","first-page":"770","article-title":"Deep residual learning for image recognition","author":"He","year":"2016","journal-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition"},{"key":"10.1016\/j.patcog.2026.113422_bib0027","first-page":"2002","article-title":"Deep ordinal regression network for monocular depth estimation","author":"Fu","year":"2018","journal-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition"},{"key":"10.1016\/j.patcog.2026.113422_bib0028","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"1851","article-title":"Unsupervised learning of depth and ego-motion from video","author":"Zhou","year":"2017"},{"key":"10.1016\/j.patcog.2026.113422_bib0029","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV)","article-title":"Vision transformers for dense prediction","author":"Ranftl","year":"2021"},{"key":"10.1016\/j.patcog.2026.113422_bib0030","series-title":"Advances in Neural Information Processing Systems (NeurIPS)","article-title":"The surprising effectiveness of diffusion models for optical flow and monocular depth estimation","author":"Saxena","year":"2023"},{"key":"10.1016\/j.patcog.2026.113422_bib0031","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2026.113093","article-title":"Freestyle: free lunch for text-guided style transfer using diffusion models","volume":"175","author":"He","year":"2026","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113422_bib0032","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","article-title":"Depth anything: unleashing the power of large-scale unlabeled data","author":"Yang","year":"2024"},{"key":"10.1016\/j.patcog.2026.113422_bib0033","series-title":"Advances in Neural Information Processing Systems (NeurIPS)","article-title":"Depth anything V2","author":"Yang","year":"2024"},{"key":"10.1016\/j.patcog.2026.113422_bib0034","unstructured":"S.F. Bhat, R. Birkl, D. Wofk, P. Wonka, M. M\u00fcller, Zoedepth: zero-shot transfer by combining relative and metric depth, (2023). arXiv preprint arXiv: 2302.12288."},{"key":"10.1016\/j.patcog.2026.113422_bib0035","doi-asserted-by":"crossref","unstructured":"M. Hu, W. Yin, C. Zhang, Z. Cai, X. Long, K. Wang, H. Chen, G. Yu, C. Shen, S. Shen, Metric3Dv2: a versatile monocular geometric foundation model for zero-shot metric depth and surface normal estimation, 2024. arXiv: 2404.15506.","DOI":"10.1109\/TPAMI.2024.3444912"},{"issue":"12","key":"10.1016\/j.patcog.2026.113422_bib0036","doi-asserted-by":"crossref","first-page":"11476","DOI":"10.1109\/TPAMI.2024.3457790","article-title":"Revealing the dark side of non-local attention in single image super-resolution","volume":"46","author":"Su","year":"2024","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.patcog.2026.113422_bib0037","doi-asserted-by":"crossref","DOI":"10.1016\/j.jvcir.2024.104051","article-title":"Texture-aware and color-consistent learning for underwater image enhancement","volume":"98","author":"Hu","year":"2024","journal-title":"J. Vis. Commun. Image Represent."},{"key":"10.1016\/j.patcog.2026.113422_bib0038","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"2834","article-title":"IniRetinex: rethinking retinex-type low-light image enhancer via initialization perspective","volume":"vol. 39","author":"Fan","year":"2025"},{"issue":"4","key":"10.1016\/j.patcog.2026.113422_bib0039","first-page":"585","article-title":"Visual-inertial navigation: a concise review","volume":"4","author":"Huang","year":"2019","journal-title":"IEEE Trans. Intell. Veh."},{"key":"10.1016\/j.patcog.2026.113422_bib0040","unstructured":"O. Elmaghraby, E. Mounier, P.R.M. de Araujo, A. Noureldin, Look to locate: vision-based multisensory navigation with 3-D digital maps for GNSS-challenged environments, (2025). arXiv preprint arXiv: 2506.19827."},{"key":"10.1016\/j.patcog.2026.113422_bib0041","unstructured":"A. Gu, T. Dao, Mamba: linear-time sequence modeling with selective state spaces, (2023). arXiv preprint arXiv: 2312.00752."},{"key":"10.1016\/j.patcog.2026.113422_bib0042","unstructured":"L. Zhu, B. Liao, Q. Zhang, X. Wang, W. Liu, X. Wang, Vision mamba: efficient visual representation learning with bidirectional state space model, (2024). arXiv preprint arXiv: 2401.09417."},{"key":"10.1016\/j.patcog.2026.113422_bib0043","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"12179","article-title":"Vision transformers for dense prediction","author":"Ranftl","year":"2021"},{"key":"10.1016\/j.patcog.2026.113422_bib0044","first-page":"1305","article-title":"CondConv: conditionally parameterized convolutions for efficient inference","volume":"32","author":"Yang","year":"2019","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.113422_bib0045","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"658","article-title":"Generalized intersection over union: a metric and a loss for bounding box regression","author":"Rezatofighi","year":"2019"},{"key":"10.1016\/j.patcog.2026.113422_bib0046","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"270","article-title":"Unsupervised monocular depth estimation with left-right consistency","author":"Godard","year":"2017"},{"key":"10.1016\/j.patcog.2026.113422_bib0047","unstructured":"V. Guizilini, R. Hou, J. Li, R. Ambrus, A. Gaidon, Semantically-guided representation learning for self-supervised monocular depth, (2020). arXiv preprint arXiv: 2002.12319."},{"key":"10.1016\/j.patcog.2026.113422_bib0048","doi-asserted-by":"crossref","DOI":"10.1016\/j.cma.2022.114909","article-title":"CAN-PINN: a fast physics-informed neural network based on coupled-automatic\u2013numerical differentiation method","volume":"395","author":"Chiu","year":"2022","journal-title":"Comput. Methods Appl. Mech. Eng."},{"key":"10.1016\/j.patcog.2026.113422_bib0049","series-title":"International Conference on Medical Image Computing and Computer-Assisted Intervention","first-page":"208","article-title":"EndoDAC: efficient adapting foundation model for self-supervised depth estimation from any endoscopic camera","author":"Cui","year":"2024"},{"key":"10.1016\/j.patcog.2026.113422_bib0050","unstructured":"G. Jocher, A. Chaurasia, J. Qiu, Ultralytics YOLOv8, 2023. https:\/\/github.com\/ultralytics\/ultralytics."},{"key":"10.1016\/j.patcog.2026.113422_bib0051","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"12345","article-title":"RF-DETR: real-time detection transformer with decoupled refinement","author":"Liu","year":"2025"},{"key":"10.1016\/j.patcog.2026.113422_bib0052","doi-asserted-by":"crossref","unstructured":"L. Yang, B. Kang, Z. Huang, X. Xu, J. Feng, H. Zhao, Depth anything: unleashing the power of large-scale unlabeled data, (2024). arXiv: 2401.10891.","DOI":"10.1109\/CVPR52733.2024.00987"},{"key":"10.1016\/j.patcog.2026.113422_sbref0053","series-title":"International Conference on Learning Representations","article-title":"Depth pro: sharp monocular metric depth in less than a second","year":"2024"},{"key":"10.1016\/j.patcog.2026.113422_bib0054","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"9233","article-title":"Towards zero-shot scale-aware monocular depth estimation","author":"Guizilini","year":"2023"}],"container-title":["Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326003870?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326003870?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,5,21]],"date-time":"2026-05-21T16:54:49Z","timestamp":1779382489000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0031320326003870"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":54,"alternative-id":["S0031320326003870"],"URL":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113422","relation":{},"ISSN":["0031-3203"],"issn-type":[{"value":"0031-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Physics-informed visual-inertial mamba for robust train localization in harsh conditions","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113422","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"113422"}}