{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,14]],"date-time":"2026-07-14T20:16:33Z","timestamp":1784060193736,"version":"3.55.0"},"publisher-location":"Cham","reference-count":65,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031200793","type":"print"},{"value":"9783031200809","type":"electronic"}],"license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022]]},"DOI":"10.1007\/978-3-031-20080-9_13","type":"book-chapter","created":{"date-parts":[[2022,11,2]],"date-time":"2022-11-02T19:59:12Z","timestamp":1667419152000},"page":"211-228","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":13,"title":["RayTran: 3D Pose Estimation and\u00a0Shape Reconstruction of\u00a0Multiple Objects from\u00a0Videos with\u00a0Ray-Traced Transformers"],"prefix":"10.1007","author":[{"given":"Micha\u0142 J.","family":"Tyszkiewicz","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kevis-Kokitsi","family":"Maninis","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Stefan","family":"Popov","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Vittorio","family":"Ferrari","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2022,11,3]]},"reference":[{"key":"13_CR1","doi-asserted-by":"crossref","unstructured":"Andriluka, M., Roth, S., Schiele, B.: People-tracking-by-detection and people-detection-by-tracking. In: CVPR (2008)","DOI":"10.1109\/CVPR.2008.4587583"},{"key":"13_CR2","doi-asserted-by":"crossref","unstructured":"Arnab, A., Dehghani, M., Heigold, G., Sun, C., Lu\u010di\u0107, M., Schmid, C.: ViViT: a video vision transformer. In: CVPR (2021)","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"13_CR3","doi-asserted-by":"crossref","unstructured":"Avetisyan, A., Dahnert, M., Dai, A., Savva, M., Chang, A.X., Nie\u00dfner, M.: Scan2CAD: learning CAD model alignment in RGB-D scans. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00272"},{"key":"13_CR4","doi-asserted-by":"crossref","unstructured":"Avetisyan, A., Dai, A., Nie\u00dfner, M.: End-to-end CAD model retrieval and 9DoF alignment in 3D scans. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00264"},{"key":"13_CR5","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"596","DOI":"10.1007\/978-3-030-58542-6_36","volume-title":"Computer Vision \u2013 ECCV 2020","author":"A Avetisyan","year":"2020","unstructured":"Avetisyan, A., Khanova, T., Choy, C., Dash, D., Dai, A., Nie\u00dfner, M.: SceneCAD: predicting object alignments and layouts in RGB-D scans. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12367, pp. 596\u2013612. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58542-6_36"},{"key":"13_CR6","doi-asserted-by":"crossref","unstructured":"Bergmann, P., Meinhardt, T., Leal-Taixe, L.: Tracking without bells and whistles. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00103"},{"key":"13_CR7","unstructured":"Bertasius, G., Wang, H., Torresani, L.: Is space-time attention all you need for video understanding? In: ICML (2021)"},{"key":"13_CR8","doi-asserted-by":"crossref","unstructured":"Breitenstein, M.D., Reichlin, F., Leibe, B., Koller-Meier, E., Van Gool, L.: Robust tracking-by-detection using a detector confidence particle filter. In: ICCV (2009)","DOI":"10.1109\/ICCV.2009.5459278"},{"key":"13_CR9","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"213","DOI":"10.1007\/978-3-030-58452-8_13","volume-title":"Computer Vision \u2013 ECCV 2020","author":"N Carion","year":"2020","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., Zagoruyko, S.: End-to-end object detection with transformers. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12346, pp. 213\u2013229. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58452-8_13"},{"key":"13_CR10","unstructured":"Chang, A.X., et al.: ShapeNet: an information-rich 3D model repository. arXiv:1512.03012 (2015)"},{"key":"13_CR11","doi-asserted-by":"crossref","unstructured":"Chen, Z., Tagliasacchi, A., Zhang, H.: BSP-Net: generating compact meshes via binary space partitioning. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00012"},{"key":"13_CR12","unstructured":"Cheng, B., Schwing, A.G., Kirillov, A.: Per-pixel classification is not all you need for semantic segmentation. In: NeurIPS (2021)"},{"key":"13_CR13","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"628","DOI":"10.1007\/978-3-319-46484-8_38","volume-title":"Computer Vision \u2013 ECCV 2016","author":"CB Choy","year":"2016","unstructured":"Choy, C.B., Xu, D., Gwak, J.Y., Chen, K., Savarese, S.: 3D-R2N2: a unified approach for single and multi-view 3D object reconstruction. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9912, pp. 628\u2013644. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46484-8_38"},{"key":"13_CR14","doi-asserted-by":"crossref","unstructured":"Dai, A., Chang, A.X., Savva, M., Halber, M., Funkhouser, T., Nie\u00dfner, M.: ScanNet: richly-annotated 3D reconstructions of indoor scenes. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.261"},{"key":"13_CR15","doi-asserted-by":"crossref","unstructured":"Rukhovich, D., Vorontsova, A., Konushin, A.: ImVoxelNet: image to voxels projection for monocular and multi-view general-purpose 3D object detection. In: WACV (2022)","DOI":"10.1109\/WACV51458.2022.00133"},{"key":"13_CR16","unstructured":"Dosovitskiy, A., et al.: An image is worth $$16\\times 16$$ words: transformers for image recognition at scale. In: ICLR (2020)"},{"key":"13_CR17","doi-asserted-by":"crossref","unstructured":"Duzceker, A., Galliani, S., Vogel, C., Speciale, P., Dusmanu, M., Pollefeys, M.: DeepVideoMVS: multi-view stereo on video with recurrent spatio-temporal fusion. In: CVPR (2021)","DOI":"10.1109\/CVPR46437.2021.01507"},{"issue":"3","key":"13_CR18","doi-asserted-by":"publisher","first-page":"611","DOI":"10.1109\/TPAMI.2017.2658577","volume":"40","author":"J Engel","year":"2017","unstructured":"Engel, J., Koltun, V., Cremers, D.: Direct sparse odometry. TPAMI 40(3), 611\u2013625 (2017)","journal-title":"TPAMI"},{"key":"13_CR19","doi-asserted-by":"crossref","unstructured":"Engelmann, F., Rematas, K., Leibe, B., Ferrari, V.: From points to multi-object 3D reconstruction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4588\u20134597 (2021)","DOI":"10.1109\/CVPR46437.2021.00456"},{"key":"13_CR20","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"318","DOI":"10.1007\/978-3-030-01252-6_19","volume-title":"Computer Vision \u2013 ECCV 2018","author":"X Fei","year":"2018","unstructured":"Fei, X., Soatto, S.: Visual-inertial object detection and mapping. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11215, pp. 318\u2013334. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01252-6_19"},{"key":"13_CR21","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"224","DOI":"10.1007\/978-3-540-24672-5_18","volume-title":"Computer Vision - ECCV 2004","author":"A Frome","year":"2004","unstructured":"Frome, A., Huber, D., Kolluri, R., B\u00fclow, T., Malik, J.: Recognizing objects in range data using regional point descriptors. In: Pajdla, T., Matas, J. (eds.) ECCV 2004. LNCS, vol. 3023, pp. 224\u2013237. Springer, Heidelberg (2004). https:\/\/doi.org\/10.1007\/978-3-540-24672-5_18"},{"key":"13_CR22","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"484","DOI":"10.1007\/978-3-319-46466-4_29","volume-title":"Computer Vision \u2013 ECCV 2016","author":"R Girdhar","year":"2016","unstructured":"Girdhar, R., Fouhey, D.F., Rodriguez, M., Gupta, A.: Learning a predictable and generative vector representation for objects. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9910, pp. 484\u2013499. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46466-4_29"},{"key":"13_CR23","doi-asserted-by":"crossref","unstructured":"Gkioxari, G., Malik, J., Johnson, J.: Mesh R-CNN. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00988"},{"key":"13_CR24","doi-asserted-by":"crossref","unstructured":"G\u00fcmeli, C., Dai, A., Nie\u00dfner, M.: ROCA: robust CAD model retrieval and alignment from a single image. arXiv:2112.01988 (2021)","DOI":"10.1109\/CVPR52688.2022.00399"},{"key":"13_CR25","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"13_CR26","doi-asserted-by":"crossref","unstructured":"Hu, H.N., et al.: Joint monocular 3D vehicle detection and tracking. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00549"},{"key":"13_CR27","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"194","DOI":"10.1007\/978-3-030-01234-2_12","volume-title":"Computer Vision \u2013 ECCV 2018","author":"S Huang","year":"2018","unstructured":"Huang, S., Qi, S., Zhu, Y., Xiao, Y., Xu, Y., Zhu, S.-C.: Holistic 3D scene parsing and reconstruction from a single RGB image. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11211, pp. 194\u2013211. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01234-2_12"},{"key":"13_CR28","doi-asserted-by":"crossref","unstructured":"Izadinia, H., Seitz, S.M.: Scene recomposition by learning-based ICP. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00101"},{"key":"13_CR29","doi-asserted-by":"crossref","unstructured":"Izadinia, H., Shan, Q., Seitz, S.M.: Im2CAD. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.260"},{"key":"13_CR30","doi-asserted-by":"crossref","unstructured":"Kundu, A., Li, Y., Rehg, J.M.: 3D-RCNN: instance-level 3D object reconstruction via render-and-compare. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00375"},{"key":"13_CR31","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"260","DOI":"10.1007\/978-3-030-58580-8_16","volume-title":"Computer Vision \u2013 ECCV 2020","author":"W Kuo","year":"2020","unstructured":"Kuo, W., Angelova, A., Lin, T.-Y., Dai, A.: Mask2CAD: 3D shape prediction by learning to segment and retrieve. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12348, pp. 260\u2013277. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58580-8_16"},{"key":"13_CR32","doi-asserted-by":"crossref","unstructured":"Kuo, W., Angelova, A., Lin, T.Y., Dai, A.: Patch2CAD: patchwise embedding learning for in-the-wild shape retrieval from a single image. In: ICCV (2021)","DOI":"10.1109\/ICCV48922.2021.01236"},{"issue":"2","key":"13_CR33","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1080\/10867651.2003.10487582","volume":"8","author":"T Lewiner","year":"2003","unstructured":"Lewiner, T., Lopes, H., Vieira, A.W., Tavares, G.: Efficient implementation of marching cubes\u2019 cases with topological guarantees. J. Graph. Tools 8(2), 1\u201315 (2003)","journal-title":"J. Graph. Tools"},{"key":"13_CR34","doi-asserted-by":"crossref","unstructured":"Li, K., et al.: ODAM: object detection, association, and mapping using posed RGB video. In: ICCV (2021)","DOI":"10.1109\/ICCV48922.2021.00594"},{"issue":"2","key":"13_CR35","doi-asserted-by":"publisher","first-page":"3341","DOI":"10.1109\/LRA.2021.3061080","volume":"6","author":"K Li","year":"2021","unstructured":"Li, K., Rezatofighi, H., Reid, I.: MOLTR: multiple object localization, tracking and reconstruction from monocular RGB videos. IEEE Robot. Autom. Lett. 6(2), 3341\u20133348 (2021)","journal-title":"IEEE Robot. Autom. Lett."},{"key":"13_CR36","doi-asserted-by":"crossref","unstructured":"Li, Y., Dai, A., Guibas, L., Nie\u00dfner, M.: Database-assisted object retrieval for real-time 3D reconstruction. In: Computer Graphics Forum, vol. 34. Wiley Online Library (2015)","DOI":"10.1111\/cgf.12573"},{"key":"13_CR37","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101 (2017)"},{"key":"13_CR38","unstructured":"Mahendran, S., Ali, H., Vidal, R.: A mixed classification-regression framework for 3D pose estimation from 2D images. arXiv:1805.03225 (2018)"},{"key":"13_CR39","doi-asserted-by":"crossref","unstructured":"Maninis, K.K., Popov, S., Niesser, M., Ferrari, V.: Vid2CAD: CAD model alignment using multi-view constraints from videos. IEEE Trans. Pattern Anal. Mach. Intell. (2022)","DOI":"10.1109\/TPAMI.2022.3146082"},{"key":"13_CR40","doi-asserted-by":"crossref","unstructured":"Meinhardt, T., Kirillov, A., Leal-Taixe, L., Feichtenhofer, C.: TrackFormer: multi-object tracking with transformers. arXiv (2021)","DOI":"10.1109\/CVPR52688.2022.00864"},{"key":"13_CR41","doi-asserted-by":"crossref","unstructured":"Mescheder, L., Oechsle, M., Niemeyer, M., Nowozin, S., Geiger, A.: Occupancy networks: learning 3D reconstruction in function space. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00459"},{"key":"13_CR42","doi-asserted-by":"crossref","unstructured":"Mousavian, A., Anguelov, D., Flynn, J., Kosecka, J.: 3D bounding box estimation using deep learning and geometry. In: CVPR (2017)","DOI":"10.1109\/CVPR.2017.597"},{"issue":"5","key":"13_CR43","doi-asserted-by":"publisher","first-page":"1147","DOI":"10.1109\/TRO.2015.2463671","volume":"31","author":"R Mur-Artal","year":"2015","unstructured":"Mur-Artal, R., Montiel, J.M.M., Tardos, J.D.: ORB-SLAM: a versatile and accurate monocular SLAM system. IEEE Trans. Robot. 31(5), 1147\u20131163 (2015)","journal-title":"IEEE Trans. Robot."},{"issue":"6","key":"13_CR44","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/2366145.2366156","volume":"31","author":"L Nan","year":"2012","unstructured":"Nan, L., Xie, K., Sharf, A.: A search-classify approach for cluttered indoor scene understanding. ACM Trans. Graph. (TOG) 31(6), 1\u201310 (2012)","journal-title":"ACM Trans. Graph. (TOG)"},{"issue":"1","key":"13_CR45","first-page":"1","volume":"4","author":"L Nicholson","year":"2018","unstructured":"Nicholson, L., Milford, M., S\u00fcnderhauf, N.: QuadricSLAM: dual quadrics from object detections as landmarks in object-oriented SLAM. RA-L 4(1), 1\u20138 (2018)","journal-title":"RA-L"},{"key":"13_CR46","doi-asserted-by":"crossref","unstructured":"Nie, Y., Han, X., Guo, S., Zheng, Y., Chang, J., Zhang, J.J.: Total3DUnderstanding: joint layout, object pose and mesh reconstruction for indoor scenes from a single image. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00013"},{"key":"13_CR47","doi-asserted-by":"crossref","unstructured":"Park, J.J., Florence, P., Straub, J., Newcombe, R., Lovegrove, S.: DeepSDF: learning continuous signed distance functions for shape representation. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00025"},{"key":"13_CR48","unstructured":"Paszke, A., et al.: PyTorch: an imperative style, high-performance deep learning library. In: Wallach, H., Larochelle, H., Beygelzimer, A., d\u2019Alch\u00e9-Buc, F., Fox, E., Garnett, R. (eds.) Advances in Neural Information Processing Systems, vol. 32, pp. 8024\u20138035. Curran Associates, Inc. (2019). http:\/\/papers.neurips.cc\/paper\/9015-pytorch-an-imperative-style-high-performance-deep-learning-library.pdf"},{"issue":"1","key":"13_CR49","doi-asserted-by":"publisher","first-page":"7","DOI":"10.1023\/A:1008109111715","volume":"32","author":"M Pollefeys","year":"1999","unstructured":"Pollefeys, M., Koch, R., Van Gool, L.: Self-calibration and metric reconstruction inspite of varying and unknown intrinsic camera parameters. IJCV 32(1), 7\u201325 (1999). https:\/\/doi.org\/10.1023\/A:1008109111715","journal-title":"IJCV"},{"key":"13_CR50","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"366","DOI":"10.1007\/978-3-030-58536-5_22","volume-title":"Computer Vision \u2013 ECCV 2020","author":"S Popov","year":"2020","unstructured":"Popov, S., Bauszat, P., Ferrari, V.: CoReNet: coherent 3D scene reconstruction from a single RGB image. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12347, pp. 366\u2013383. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58536-5_22"},{"key":"13_CR51","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"140","DOI":"10.1007\/978-3-030-58555-6_9","volume-title":"Computer Vision \u2013 ECCV 2020","author":"S Qian","year":"2020","unstructured":"Qian, S., Jin, L., Fouhey, D.F.: Associative3D: volumetric reconstruction from sparse views. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12360, pp. 140\u2013157. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58555-6_9"},{"key":"13_CR52","doi-asserted-by":"crossref","unstructured":"Runz, M., et al.: FroDO: from detections to 3D objects. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.01473"},{"key":"13_CR53","doi-asserted-by":"crossref","unstructured":"Salas-Moreno, R.F., Newcombe, R.A., Strasdat, H., Kelly, P.H., Davison, A.J.: SLAM++: simultaneous localisation and mapping at the level of objects. In: CVPR (2013)","DOI":"10.1109\/CVPR.2013.178"},{"key":"13_CR54","doi-asserted-by":"crossref","unstructured":"Sch\u00f6nberger, J.L., Frahm, J.M.: Structure-from-motion revisited. In: CVPR (2016)","DOI":"10.1109\/CVPR.2016.445"},{"key":"13_CR55","doi-asserted-by":"crossref","unstructured":"Shan, M., Feng, Q., Jau, Y.Y., Atanasov, N.: ELLIPSDF: joint object pose and shape optimization with a bi-level ellipsoid and signed distance function description. In: ICCV (2021)","DOI":"10.1109\/ICCV48922.2021.00589"},{"issue":"6","key":"13_CR56","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/2366145.2366155","volume":"31","author":"T Shao","year":"2012","unstructured":"Shao, T., Xu, W., Zhou, K., Wang, J., Li, D., Guo, B.: An interactive approach to semantic modeling of indoor scenes with an RGBD camera. ACM Trans. Graph. (TOG) 31(6), 1\u201311 (2012)","journal-title":"ACM Trans. Graph. (TOG)"},{"key":"13_CR57","doi-asserted-by":"crossref","unstructured":"Strudel, R., Garcia, R., Laptev, I., Schmid, C.: Segmenter: transformer for semantic segmentation. In: ICCV (2021)","DOI":"10.1109\/ICCV48922.2021.00717"},{"key":"13_CR58","doi-asserted-by":"crossref","unstructured":"Tulsiani, S., Gupta, S., Fouhey, D., Efros, A.A., Malik, J.: Factoring shape, pose, and layout from the 2D image of a 3D scene. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00039"},{"key":"13_CR59","unstructured":"Vaswani, A., et al.: Attention is all you need. In: NeurIPS (2017)"},{"key":"13_CR60","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"55","DOI":"10.1007\/978-3-030-01252-6_4","volume-title":"Computer Vision \u2013 ECCV 2018","author":"N Wang","year":"2018","unstructured":"Wang, N., Zhang, Y., Li, Z., Fu, Y., Liu, W., Jiang, Y.-G.: Pixel2Mesh: generating 3D mesh models from single RGB images. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11215, pp. 55\u201371. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01252-6_4"},{"key":"13_CR61","doi-asserted-by":"crossref","unstructured":"Wu, C.: Towards linear-time incremental structure from motion. In: 3DV (2013)","DOI":"10.1109\/3DV.2013.25"},{"key":"13_CR62","unstructured":"Wu, J., Zhang, C., Xue, T., Freeman, W.T., Tenenbaum, J.B.: Learning a probabilistic latent space of object shapes via 3D generative-adversarial modeling. In: NIPS (2016)"},{"issue":"12","key":"13_CR63","doi-asserted-by":"publisher","first-page":"2919","DOI":"10.1007\/s11263-020-01347-6","volume":"128","author":"H Xie","year":"2020","unstructured":"Xie, H., Yao, H., Zhang, S., Zhou, S., Sun, W.: Pix2Vox++: multi-scale context-aware 3D object reconstruction from single and multiple images. IJCV 128(12), 2919\u20132935 (2020). https:\/\/doi.org\/10.1007\/s11263-020-01347-6","journal-title":"IJCV"},{"issue":"4","key":"13_CR64","doi-asserted-by":"publisher","first-page":"925","DOI":"10.1109\/TRO.2019.2909168","volume":"35","author":"S Yang","year":"2019","unstructured":"Yang, S., Scherer, S.: CubeSLAM: monocular 3-D object SLAM. IEEE Trans. Robot. 35(4), 925\u2013938 (2019)","journal-title":"IEEE Trans. Robot."},{"key":"13_CR65","doi-asserted-by":"crossref","unstructured":"Zheng, S., et al.: Rethinking semantic segmentation from a sequence-to-sequence perspective with transformers. In: CVPR (2021)","DOI":"10.1109\/CVPR46437.2021.00681"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2022"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-20080-9_13","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,11,7]],"date-time":"2022-11-07T00:24:12Z","timestamp":1667780652000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-20080-9_13"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"ISBN":["9783031200793","9783031200809"],"references-count":65,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-20080-9_13","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022]]},"assertion":[{"value":"3 November 2022","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Tel Aviv","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Israel","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2022","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 October 2022","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 October 2022","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2022","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2022.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"CMT","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"5804","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"1645","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"28% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.21","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.91","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}