{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,17]],"date-time":"2026-06-17T23:03:23Z","timestamp":1781737403860,"version":"3.54.5"},"reference-count":54,"publisher":"IEEE","license":[{"start":{"date-parts":[[2023,5,29]],"date-time":"2023-05-29T00:00:00Z","timestamp":1685318400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,5,29]],"date-time":"2023-05-29T00:00:00Z","timestamp":1685318400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100008122","name":"Institute of Information & communications Technology Planning & Evaluation (IITP)","doi-asserted-by":"publisher","award":["2021-0-01343"],"award-info":[{"award-number":["2021-0-01343"]}],"id":[{"id":"10.13039\/501100008122","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023,5,29]]},"DOI":"10.1109\/icra48891.2023.10160657","type":"proceedings-article","created":{"date-parts":[[2023,7,4]],"date-time":"2023-07-04T17:20:56Z","timestamp":1688491256000},"page":"4983-4990","source":"Crossref","is-referenced-by-count":42,"title":["SwinDepth: Unsupervised Depth Estimation using Monocular Sequences via Swin Transformer and Densely Cascaded Network"],"prefix":"10.1109","author":[{"given":"Dongseok","family":"Shim","sequence":"first","affiliation":[{"name":"Seoul National University,Interdisciplinary Program in Artificial Intelligence (IPAI)"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"H. Jin","family":"Kim","sequence":"additional","affiliation":[{"name":"Seoul National University,Interdisciplinary Program in Artificial Intelligence (IPAI)"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"ref15","first-page":"213","article-title":"End-to-end object detection with transformers","author":"carion","year":"2020","journal-title":"European Conference on Computer Vision"},{"key":"ref14","article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","author":"dosovitskiy","year":"2020","journal-title":"International Conference on Learning Representations"},{"key":"ref53","author":"ridnik","year":"2021","journal-title":"ImageNet-21K Pretraining for the Masses"},{"key":"ref52","first-page":"716","article-title":"Discrete-continuous depth estimation from a single image","author":"miaomiao","year":"2014","journal-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition"},{"key":"ref11","author":"simonyan","year":"2015","journal-title":"Very Deep Convolutional Networks for Large-scale Image Recognition"},{"key":"ref10","first-page":"234","article-title":"U-net: Con-volutional networks for biomedical image segmentation","author":"olaf","year":"2015","journal-title":"International Conference on Medical Image Computing and Computer-Assisted Intervention"},{"key":"ref54","author":"chen","year":"2017","journal-title":"Rethinking atrous convolution for semantic image segmentation"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00681"},{"key":"ref16","author":"beal","year":"2020","journal-title":"Toward transformer-based object detection"},{"key":"ref19","author":"li","year":"2021","journal-title":"Revisiting stereo depth estimation from a sequence-to-sequence perspective with transformers"},{"key":"ref18","first-page":"4009","article-title":"Ad-abins: Depth estimation using adaptive bins","author":"bhat","year":"2021","journal-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition"},{"key":"ref51","doi-asserted-by":"crossref","first-page":"2144","DOI":"10.1109\/TPAMI.2014.2316835","article-title":"Depth transfer: Depth extraction from video using non-parametric sampling","volume":"36","author":"kevin","year":"2014","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"ref50","article-title":"Hrformer: High-resolution vision transformer for dense predict","volume":"34","author":"yuan","year":"2021","journal-title":"Advances in neural information processing systems"},{"key":"ref46","first-page":"2881","article-title":"Pyramid scene parsing network","author":"hengshuang","year":"2017","journal-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i3.16329"},{"key":"ref48","author":"kingma","year":"2017","journal-title":"Adam A method for stochastic optimization"},{"key":"ref47","article-title":"Automatic differentiation in PyTorch","author":"adam","year":"2017","journal-title":"NeurIPS-W"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2019.2930258"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01252"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00256"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33018001"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0816-y"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/TRO.2016.2624754"},{"key":"ref7","first-page":"3828","article-title":"Digging into self-supervised monocular depth estimation","author":"cl\u00e9ment","year":"2019","journal-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2018.8593691"},{"key":"ref4","first-page":"740","article-title":"Un-supervised cnn for single view depth estimation: Geometry to the rescue","author":"garg","year":"2016","journal-title":"European Conference on Computer Vision"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00214"},{"key":"ref6","first-page":"1851","article-title":"Unsupervised learning of depth and ego-motion from video","author":"tinghui","year":"2017","journal-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition"},{"key":"ref5","first-page":"270","article-title":"Unsuper-vised monocular depth estimation with left-right consistency","author":"cl\u00e9ment","year":"2017","journal-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00031"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.12257"},{"key":"ref34","first-page":"484","article-title":"Learning monocular depth by distilling cross-domain stereo networks","author":"guo","year":"2018","journal-title":"Proceedings of the European Conference on Computer Vision (ECCV)"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00212"},{"key":"ref36","first-page":"5667","article-title":"Unsuper-vised learning of depth and ego-motion from monocular video using 3d geometric constraints","author":"reza","year":"2018","journal-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition"},{"key":"ref31","first-page":"6647","article-title":"Semi-supervised deep learning for monocular depth map prediction","author":"yevhen","year":"2017","journal-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00281"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00024"},{"key":"ref32","first-page":"817","article-title":"Deep virtual stereo odometry: Leveraging deep depth prediction for monocular direct sparse odometry","author":"yang","year":"2018","journal-title":"Proceedings of the European Conference on Computer Vision (ECCV)"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/3DV.2016.32"},{"key":"ref1","first-page":"2366","article-title":"Depth map prediction from a single image using a multi-scale deep network","volume":"27","author":"david","year":"2014","journal-title":"Advances in neural information processing systems"},{"key":"ref39","first-page":"36","article-title":"Df-net: Unsupervised joint learning of depth and flow using cross-task consistency","author":"yuliang","year":"2018","journal-title":"Proceedings of the European Conference on Computer Vision (ECCV)"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00216"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1177\/0278364913491297"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.106"},{"key":"ref26","first-page":"5998","article-title":"Attention is all you need","author":"vaswani","year":"2017","journal-title":"Advances in neural information processing systems"},{"key":"ref25","first-page":"824","article-title":"Make3d: Learning 3d scene structure from a single still image","volume":"31","author":"ashutosh","year":"2008","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"ref20","author":"ranftl","year":"2021","journal-title":"Vision Transformers for Dense Prediction"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.243"},{"key":"ref21","author":"liu","year":"2021","journal-title":"Swin transformer Hierarchical vision transformer using shifted windows"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2015.2505283"},{"key":"ref27","first-page":"10347","article-title":"Training data-efficient image transformers & distillation through attention","author":"touvron","year":"2021","journal-title":"Int Conference on Machine Learning"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01249-6_43"}],"event":{"name":"2023 IEEE International Conference on Robotics and Automation (ICRA)","location":"London, United Kingdom","start":{"date-parts":[[2023,5,29]]},"end":{"date-parts":[[2023,6,2]]}},"container-title":["2023 IEEE International Conference on Robotics and Automation (ICRA)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/10160211\/10160212\/10160657.pdf?arnumber=10160657","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,7,24]],"date-time":"2023-07-24T17:35:37Z","timestamp":1690220137000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10160657\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,5,29]]},"references-count":54,"URL":"https:\/\/doi.org\/10.1109\/icra48891.2023.10160657","relation":{},"subject":[],"published":{"date-parts":[[2023,5,29]]}}}