{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,7]],"date-time":"2026-08-07T21:23:02Z","timestamp":1786137782649,"version":"3.56.0"},"reference-count":49,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100014188","name":"Ministry of Science and ICT, South Korea","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100014188","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100010418","name":"Institute of Information & Communications Technology Planning & Evaluation","doi-asserted-by":"publisher","award":["RS-2022-00156287"],"award-info":[{"award-number":["RS-2022-00156287"]}],"id":[{"id":"10.13039\/501100010418","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100010418","name":"Institute of Information & Communications Technology Planning & Evaluation","doi-asserted-by":"publisher","award":["RS-2023-00256629"],"award-info":[{"award-number":["RS-2023-00256629"]}],"id":[{"id":"10.13039\/501100010418","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100010418","name":"Institute of Information & Communications Technology Planning & Evaluation","doi-asserted-by":"publisher","award":["RS-2024-00437718"],"award-info":[{"award-number":["RS-2024-00437718"]}],"id":[{"id":"10.13039\/501100010418","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1016\/j.eswa.2026.132274","type":"journal-article","created":{"date-parts":[[2026,4,3]],"date-time":"2026-04-03T01:57:02Z","timestamp":1775181422000},"page":"132274","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":1,"special_numbering":"C","title":["MonoSTR: Structure-aware monocular 3D object detection via keypoint-based geometric representation"],"prefix":"10.1016","volume":"321","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-7094-1153","authenticated-orcid":false,"given":"Yeon","family":"Woo Cho","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-0738-6572","authenticated-orcid":false,"given":"Jung","family":"Woo Cheon","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6528-701X","authenticated-orcid":false,"given":"Seok","family":"Bong Yoo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.132274_bib0001","series-title":"Cvpr","first-page":"11682","article-title":"Seeing through fog without seeing fog: Deep multimodal sensor fusion in unseen adverse weather","author":"Bijelic","year":"2020"},{"key":"10.1016\/j.eswa.2026.132274_bib0002","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"9287","article-title":"M3d-RPN: Monocular 3D region proposal network for object detection","author":"Brazil","year":"2019"},{"key":"10.1016\/j.eswa.2026.132274_bib0003","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"7291","article-title":"Realtime multi-person 2D pose estimation using part affinity fields","author":"Cao","year":"2017"},{"key":"10.1016\/j.eswa.2026.132274_bib0004","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"10379","article-title":"Monorun: Monocular 3D object detection by reconstruction and uncertainty propagation","author":"Chen","year":"2021"},{"issue":"11","key":"10.1016\/j.eswa.2026.132274_bib0005","doi-asserted-by":"crossref","first-page":"11232","DOI":"10.1109\/JSEN.2022.3189174","article-title":"M3DGAF: Monocular 3D object detection with geometric appearance awareness and feature fusion","volume":"23","author":"Chen","year":"2023","journal-title":"IEEE Sensors Journal"},{"key":"10.1016\/j.eswa.2026.132274_bib0006","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"12093","article-title":"Monopair: Monocular 3D object detection using pairwise spatial relationships","author":"Chen","year":"2020"},{"key":"10.1016\/j.eswa.2026.132274_bib0007","first-page":"4479","article-title":"Fast fourier convolution","volume":"33","author":"Chi","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132274_bib0008","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2021.114877","article-title":"Deep monocular depth estimation leveraging a large-scale outdoor stereo dataset","volume":"178","author":"Cho","year":"2021","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132274_bib0009","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110939","article-title":"Occlusion-guided multi-modal fusion for vehicle-infrastructure cooperative 3D object detection","volume":"157","author":"Chu","year":"2025","journal-title":"Pattern Recognition"},{"key":"10.1016\/j.eswa.2026.132274_bib0010","article-title":"Dynamic clustering transformer for liDAR-based 3D object detection","author":"Cui","year":"2025","journal-title":"Pattern Recognition"},{"key":"10.1016\/j.eswa.2026.132274_bib0011","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"6569","article-title":"CenterNet: Keypoint triplets for object detection","author":"Duan","year":"2019"},{"issue":"11","key":"10.1016\/j.eswa.2026.132274_bib0012","doi-asserted-by":"crossref","first-page":"1231","DOI":"10.1177\/0278364913491297","article-title":"Vision meets robotics: The kitti dataset","volume":"32","author":"Geiger","year":"2013","journal-title":"The International Journal of Robotics Research"},{"key":"10.1016\/j.eswa.2026.132274_bib0013","first-page":"513","article-title":"A kernel method for the two-sample-problem","volume":"19","author":"Gretton","year":"2006","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.132274_bib0014","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"22962","article-title":"Unified keypoint-based action recognition framework via structured keypoint pooling","author":"Hachiuma","year":"2023"},{"key":"10.1016\/j.eswa.2026.132274_bib0015","series-title":"2025 IEEE\/RSJ international conference on intelligent robots and systems (IROS)","first-page":"9870","article-title":"Mitigating hallucinations in YOLO-based object detection models: A revisit to out-of-distribution detection","author":"He","year":"2025"},{"issue":"5786","key":"10.1016\/j.eswa.2026.132274_bib0016","doi-asserted-by":"crossref","first-page":"504","DOI":"10.1126\/science.1127647","article-title":"Reducing the dimensionality of data with neural networks","volume":"313","author":"Hinton","year":"2006","journal-title":"Science"},{"key":"10.1016\/j.eswa.2026.132274_bib0017","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"4012","article-title":"MonoDTR: Monocular 3D object detection with depth-aware transformer","author":"Huang","year":"2022"},{"key":"10.1016\/j.eswa.2026.132274_bib0018","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"12783","article-title":"Keypointdeformer: Unsupervised 3D keypoint discovery for shape control","author":"Jakab","year":"2021"},{"key":"10.1016\/j.eswa.2026.132274_bib0019","series-title":"Advances in neural information processing systems","first-page":"11392","article-title":"MonoMAE: Enhancing monocular 3D detection through depth-aware masked autoencoders","volume":"vol. 37","author":"Jiang","year":"2024"},{"key":"10.1016\/j.eswa.2026.132274_bib0020","series-title":"European conference on computer vision","first-page":"664","article-title":"Deviant: Depth equivariant network for monocular 3D object detection","author":"Kumar","year":"2022"},{"key":"10.1016\/j.eswa.2026.132274_bib0021","series-title":"ICASSP 2024-2024 IEEE international conference on acoustics, speech and signal processing (ICASSP)","first-page":"4135","article-title":"Aeam3d: Adverse environment-adaptive monocular 3D object detection via feature extraction regularization","author":"Lei","year":"2024"},{"key":"10.1016\/j.eswa.2026.132274_bib0022","series-title":"European conference on computer vision","first-page":"644","article-title":"RTM3D: Real-time monocular 3D detection from object keypoints for autonomous driving","author":"Li","year":"2020"},{"key":"10.1016\/j.eswa.2026.132274_bib0023","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2023.123042","article-title":"Automated measurement of beef cattle body size via key point detection and monocular depth estimation","volume":"244","author":"Li","year":"2024","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132274_bib0024","unstructured":"Li, X., Liu, J., Lei, Y., Ma, L., Fan, X., & Liu, R. (2023). Monotdp: Twin depth perception for monocular 3D object detection in adverse scenes.arXiv: 2305.10974."},{"key":"10.1016\/j.eswa.2026.132274_bib0025","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2024.123805","article-title":"WS-SSD: Achieving faster 3D object detection for autonomous driving via weighted point cloud sampling","volume":"249","author":"Li","year":"2024","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132274_bib0026","article-title":"PMM3D: A transformer-based monocular 3D detector with parallel multi-time inquiry and mixup enhancement","author":"Lin","year":"2025","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132274_bib0027","series-title":"European conference on computer vision","first-page":"740","article-title":"Microsoft coco: Common objects in context","author":"Lin","year":"2014"},{"key":"10.1016\/j.eswa.2026.132274_bib0028","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"1810","article-title":"Learning auxiliary monocular contexts helps monocular 3D object detection","volume":"vol. 36","author":"Liu","year":"2022"},{"key":"10.1016\/j.eswa.2026.132274_bib0029","unstructured":"Lugaresi, C., Tang, J., Nash, H., McClanahan, C., Uboweja, E., Hays, M., Zhang, F., Chang, C.-L., Yong, M. G., Lee, J. et al. (2019). Mediapipe: A framework for building perception pipelines. arXiv: 1906.08172."},{"issue":"21","key":"10.1016\/j.eswa.2026.132274_bib0030","doi-asserted-by":"crossref","first-page":"40763","DOI":"10.1109\/JSEN.2025.3578608","article-title":"MonoICT: A monocular 3-d object detection model integrating CNN and transformer","volume":"25","author":"Na","year":"2025","journal-title":"IEEE Sensors Journal"},{"key":"10.1016\/j.eswa.2026.132274_bib0031","series-title":"2023 IEEE 26th international conference on intelligent transportation systems (ITSC)","first-page":"4367","article-title":"Skope3d: A synthetic dataset for vehicle keypoint perception in 3D from traffic monitoring cameras","author":"Pahadia","year":"2023"},{"issue":"4-5","key":"10.1016\/j.eswa.2026.132274_bib0032","doi-asserted-by":"crossref","first-page":"681","DOI":"10.1177\/0278364920979368","article-title":"Canadian adverse driving conditions dataset","volume":"40","author":"Pitropov","year":"2021","journal-title":"The International Journal of Robotics Research"},{"key":"10.1016\/j.eswa.2026.132274_bib0033","series-title":"Proceedings of the computer vision and pattern recognition conference","first-page":"6520","article-title":"Monodgp: Monocular 3D object detection with decoupled-query and geometry-error priors","author":"Pu","year":"2025"},{"key":"10.1016\/j.eswa.2026.132274_bib0034","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2026.131131","article-title":"Bevformer++: Enhancing bev fusion with normalized embedding and range attention for 3D object detection","author":"Qayyum","year":"2026","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132274_bib0035","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"8555","article-title":"Categorical depth distribution network for monocular 3D object detection","author":"Reading","year":"2021"},{"key":"10.1016\/j.eswa.2026.132274_bib0036","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"1906","article-title":"Carfusion: Combining point tracking and part detection for dynamic 3D reconstruction of vehicles","author":"Reddy","year":"2018"},{"key":"10.1016\/j.eswa.2026.132274_bib0037","unstructured":"\u0160ebek, P., Pokorn\u1ef3, \u0160., Vacek, P., & Svoboda, T. (2022). Real3d-aug: Point cloud augmentation by placing real objects with occlusion handling for 3D detection and segmentation.arXiv: 2206.07634."},{"key":"10.1016\/j.eswa.2026.132274_bib0038","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"1145","article-title":"Hand keypoint detection in single images using multiview bootstrapping","author":"Simon","year":"2017"},{"key":"10.1016\/j.eswa.2026.132274_bib0039","series-title":"European conference on computer vision","first-page":"216","article-title":"Keypoint promptable re-identification","author":"Somers","year":"2024"},{"key":"10.1016\/j.eswa.2026.132274_bib0040","series-title":"Proceedings of the IEEE\/CVF winter conference on applications of computer vision","first-page":"2149","article-title":"Resolution-robust large mask inpainting with fourier convolutions","author":"Suvorov","year":"2022"},{"key":"10.1016\/j.eswa.2026.132274_bib0041","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"454","article-title":"Depth-conditioned dynamic message propagation for monocular 3d object detection","author":"Wang","year":"2021"},{"key":"10.1016\/j.eswa.2026.132274_bib0042","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"913","article-title":"FCOS3D: Fully convolutional one-stage monocular 3D object detection","author":"Wang","year":"2021"},{"key":"10.1016\/j.eswa.2026.132274_bib0043","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"8445","article-title":"Pseudo-lidar from visual depth estimation: Bridging the gap in 3D object detection for autonomous driving","author":"Wang","year":"2019"},{"key":"10.1016\/j.eswa.2026.132274_bib0044","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"10248","article-title":"Monocd: Monocular 3D object detection with complementary depths","author":"Yan","year":"2024"},{"issue":"5","key":"10.1016\/j.eswa.2026.132274_bib0045","doi-asserted-by":"crossref","first-page":"4593","DOI":"10.1109\/TITS.2023.3323036","article-title":"Occlusion-aware plane-constraints for monocular 3D object detection","volume":"25","author":"Yao","year":"2024","journal-title":"IEEE Transactions on Intelligent Transportation Systems"},{"issue":"10","key":"10.1016\/j.eswa.2026.132274_bib0046","doi-asserted-by":"crossref","first-page":"19068","DOI":"10.1109\/TNNLS.2025.3577618","article-title":"Monori: Orientation-guided pnp for monocular 3-d object detection","volume":"36","author":"Yao","year":"2025","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"10.1016\/j.eswa.2026.132274_bib0047","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"2403","article-title":"Deep layer aggregation","author":"Yu","year":"2018"},{"key":"10.1016\/j.eswa.2026.132274_bib0048","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"9155","article-title":"MonoDETR: Depth-guided transformer for monocular 3D object detection","author":"Zhang","year":"2023"},{"issue":"12","key":"10.1016\/j.eswa.2026.132274_bib0049","doi-asserted-by":"crossref","first-page":"10114","DOI":"10.1109\/TPAMI.2021.3136899","article-title":"MonoEF: Extrinsic parameter free monocular 3D object detection","volume":"44","author":"Zhou","year":"2021","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426011875?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426011875?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,8]],"date-time":"2026-06-08T15:56:01Z","timestamp":1780934161000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426011875"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":49,"alternative-id":["S0957417426011875"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132274","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"MonoSTR: Structure-aware monocular 3D object detection via keypoint-based geometric representation","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132274","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"132274"}}