{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,27]],"date-time":"2026-01-27T18:46:23Z","timestamp":1769539583825,"version":"3.49.0"},"reference-count":50,"publisher":"IEEE","license":[{"start":{"date-parts":[[2022,6,5]],"date-time":"2022-06-05T00:00:00Z","timestamp":1654387200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2022,6,5]],"date-time":"2022-06-05T00:00:00Z","timestamp":1654387200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100004204","name":"Tongji University","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100004204","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022,6,5]]},"DOI":"10.1109\/iv51971.2022.9827462","type":"proceedings-article","created":{"date-parts":[[2022,7,19]],"date-time":"2022-07-19T15:33:28Z","timestamp":1658244808000},"page":"411-418","source":"Crossref","is-referenced-by-count":8,"title":["DST3D: DLA-Swin Transformer for Single-Stage Monocular 3D Object Detection"],"prefix":"10.1109","author":[{"given":"Zhihong","family":"Wu","sequence":"first","affiliation":[{"name":"Tongji University,School of Automotive Studies,Shanghai,PR China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xin","family":"Jiang","sequence":"additional","affiliation":[{"name":"Tongji University,School of Automotive Studies,Shanghai,PR China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ruidong","family":"Xu","sequence":"additional","affiliation":[{"name":"Tongji University,School of Automotive Studies,Shanghai,PR China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ke","family":"Lu","sequence":"additional","affiliation":[{"name":"Tongji University,School of Automotive Studies,Shanghai,PR China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuan","family":"Zhu","sequence":"additional","affiliation":[{"name":"Tongji University,School of Automotive Studies,Shanghai,PR China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mingzhi","family":"Wu","sequence":"additional","affiliation":[{"name":"Nanchang Automotive Institute of Intelligence &#x0026; New Energy, Tongji University,Nanchang,Jiangxi,PR China,330052"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58592-1_9"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00208"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58601-0_19"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00864"},{"key":"ref31","article-title":"Orthographic feature transform for monocular 3d object detection","author":"roddick","year":"2018","journal-title":"arXiv Computer Vision and Pattern Recognition"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.324"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.236"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/IROS40897.2019.8967624"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00695"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2019.00114"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.106"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.81"},{"key":"ref29","article-title":"Swin transformer v2: Scaling up capacity and resolution","author":"liu","year":"2021"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref1","article-title":"Very deep convolutional networks for large-scale image recognition","author":"simonyan","year":"2014","journal-title":"Computer Vision and Pattern Recognition"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00255"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"ref21","article-title":"Attention is all you need","author":"vaswani","year":"2017","journal-title":"Neural Information Processing Systems"},{"key":"ref24","article-title":"Training data-efficient image transformers & distillation through attention","author":"touvron","year":"2020","journal-title":"arXiv Computer Vision and Pattern Recognition"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"ref26","article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","author":"dosovitskiy","year":"2020","journal-title":"arXiv Computer Vision and Pattern Recognition"},{"key":"ref25","article-title":"Objects as points","author":"zhou","year":"2019","journal-title":"arXiv Computer Vision and Pattern Recognition"},{"key":"ref50","article-title":"R-drop: Regularized dropout for neural networks","author":"liang","year":"2021","journal-title":"arXiv Learning"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01298"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00315"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/IV47402.2020.9304847"},{"key":"ref12","article-title":"Voxel r-cnn: Towards high performance voxel-based 3d object detection","author":"deng","year":"2020","journal-title":"arXiv Computer Vision and Pattern Recognition"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW50498.2020.00506"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2012.6248074"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW54120.2021.00107"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01489"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00845"},{"key":"ref18","article-title":"Probabilistic and geometric depth: Detecting objects in perspective","author":"wang","year":"2021","journal-title":"arXiv Computer Vision and Pattern Recognition"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.243"},{"key":"ref4","article-title":"Faster r-cnn: towards real-time object detection with region proposal networks","author":"ren","year":"2015","journal-title":"Neural Information Processing Systems"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.169"},{"key":"ref6","article-title":"Ssd: Single shot multibox detector","author":"liu","year":"2016","journal-title":"European Conference on Computer Vision"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.322"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00086"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.91"},{"key":"ref49","article-title":"Shift rcnn: Deep monocular 3d object detection with closed-form geometric constraints","author":"naiden","year":"2019","journal-title":"International Conference on Image Processing"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00472"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00217"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01264-9_45"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00115"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00111"},{"key":"ref42","article-title":"Monocular 3d object detection and box fitting trained end-to-end using intersection-over-union loss","author":"j\u00f6rgensen","year":"2019","journal-title":"arXiv Computer Vision and Pattern Recognition"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00938"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.597"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58580-8_38"}],"event":{"name":"2022 IEEE Intelligent Vehicles Symposium (IV)","location":"Aachen, Germany","start":{"date-parts":[[2022,6,4]]},"end":{"date-parts":[[2022,6,9]]}},"container-title":["2022 IEEE Intelligent Vehicles Symposium (IV)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9826996\/9826997\/09827462.pdf?arnumber=9827462","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,27]],"date-time":"2026-01-27T05:39:27Z","timestamp":1769492367000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9827462\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,6,5]]},"references-count":50,"URL":"https:\/\/doi.org\/10.1109\/iv51971.2022.9827462","relation":{},"subject":[],"published":{"date-parts":[[2022,6,5]]}}}