{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T15:34:00Z","timestamp":1785512040632,"version":"3.56.0"},"reference-count":34,"publisher":"Springer Science and Business Media LLC","issue":"9","license":[{"start":{"date-parts":[[2018,3,7]],"date-time":"2018-03-07T00:00:00Z","timestamp":1520380800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100000781","name":"European Research Council","doi-asserted-by":"publisher","award":["647769"],"award-info":[{"award-number":["647769"]}],"id":[{"id":"10.13039\/501100000781","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100000781","name":"European Research Council","doi-asserted-by":"publisher","award":["647769"],"award-info":[{"award-number":["647769"]}],"id":[{"id":"10.13039\/501100000781","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2018,9]]},"DOI":"10.1007\/s11263-018-1070-x","type":"journal-article","created":{"date-parts":[[2018,3,7]],"date-time":"2018-03-07T01:12:06Z","timestamp":1520385126000},"page":"961-972","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":349,"title":["Augmented Reality Meets Computer Vision: Efficient Data Generation for Urban Driving Scenes"],"prefix":"10.1007","volume":"126","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-6669-7072","authenticated-orcid":false,"given":"Hassan","family":"Abu Alhaija","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Siva Karthik","family":"Mustikovela","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lars","family":"Mescheder","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Andreas","family":"Geiger","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Carsten","family":"Rother","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2018,3,7]]},"reference":[{"key":"1070_CR1","unstructured":"Blender Online Community. (2006). Blender: a 3D modelling and rendering package. Amsterdam: Blender Foundation, Blender Institute. http:\/\/www.blender.org . Accessed 01 May 2017."},{"issue":"2","key":"1070_CR2","doi-asserted-by":"publisher","first-page":"88","DOI":"10.1016\/j.patrec.2008.04.005","volume":"30","author":"GJ Brostow","year":"2009","unstructured":"Brostow, G. J., Fauqueur, J., & Cipolla, R. (2009). Semantic object classes in video: A high-definition ground truth database. Pattern Recognition Letters, 30(2), 88\u201397.","journal-title":"Pattern Recognition Letters"},{"key":"1070_CR3","doi-asserted-by":"crossref","unstructured":"Chen, W., Wang, H., Li, Y., Su, H., Wang, Z., Tu, C., et al. (2016). Synthesizing training images for boosting human 3D pose estimation. In Proceedings of the international conference on 3D vision (3DV).","DOI":"10.1109\/3DV.2016.58"},{"key":"1070_CR4","doi-asserted-by":"crossref","unstructured":"Cordts, M., Omran, M., Ramos, S., Rehfeld, T., Enzweiler, M., Benenson, R., et al. (2016). The cityscapes dataset for semantic urban scene understanding. In Proceedings of IEEE conference on computer vision and pattern recognition (CVPR).","DOI":"10.1109\/CVPR.2016.350"},{"key":"1070_CR5","doi-asserted-by":"crossref","unstructured":"Dai, J., He, K., & Sun, J. (2016). Instance-aware semantic segmentation via multi-task network cascades. In Proceedings of IEEE conference on computer vision and pattern recognition (CVPR).","DOI":"10.1109\/CVPR.2016.343"},{"key":"1070_CR6","unstructured":"de\u00a0Souza, C. R., Gaidon, A., Cabon, Y., & Pe\u00f1a, A. M. L. (2016). Procedural generation of videos to train deep action recognition networks. arXiv:1612.00881 ."},{"key":"1070_CR7","doi-asserted-by":"crossref","unstructured":"Dosovitskiy, A., Fischer, P., Ilg, E., Haeusser, P., Hazirbas, C., Golkov, V., et al. (2015). Flownet: Learning optical flow with convolutional networks. In: Proceedings of the IEEE international conference on computer vision (ICCV).","DOI":"10.1109\/ICCV.2015.316"},{"key":"1070_CR8","unstructured":"Gaidon, A., Wang, Q., Cabon, Y., & Vig, E. (2016). Virtual worlds as proxy for multi-object tracking analysis. In Proceedings of IEEE conference on computer vision and pattern recognition (CVPR)."},{"issue":"11","key":"1070_CR9","doi-asserted-by":"publisher","first-page":"1231","DOI":"10.1177\/0278364913491297","volume":"32","author":"A Geiger","year":"2013","unstructured":"Geiger, A., Lenz, P., Stiller, C., & Urtasun, R. (2013). Vision meets robotics: The KITTI dataset. International Journal of Robotics Research (IJRR), 32(11), 1231\u20131237.","journal-title":"International Journal of Robotics Research (IJRR)"},{"key":"1070_CR10","doi-asserted-by":"crossref","unstructured":"Gupta, A., Vedaldi, A., & Zisserman, A. (2016). Synthetic data for text localisation in natural images. In Proceedings of IEEE conference on computer vision and pattern recognition (CVPR).","DOI":"10.1109\/CVPR.2016.254"},{"key":"1070_CR11","unstructured":"Handa, A., Patraucean, V., Badrinarayanan, V., Stent, S., & Cipolla, R. (2016). Understanding real world indoor scenes with synthetic data. In Proceedings of IEEE conference on computer vision and pattern recognition (CVPR)."},{"key":"1070_CR12","doi-asserted-by":"crossref","unstructured":"Hattori, H., Boddeti, V. N., Kitani, K. M., & Kanade, T. (2015). Learning scene-specific pedestrian detectors without real data. In Proceedigs of IEEE conference on computer vision and pattern recognition (CVPR).","DOI":"10.1109\/CVPR.2015.7299006"},{"key":"1070_CR13","unstructured":"Jakob, W. (2010). Mitsuba renderer. http:\/\/www.mitsuba-renderer.org . Accessed 01 May 2017."},{"issue":"2","key":"1070_CR14","doi-asserted-by":"publisher","first-page":"643","DOI":"10.1111\/cgf.12591","volume":"34","author":"J Kronander","year":"2015","unstructured":"Kronander, J., Banterle, F., Gardner, A., Miandji, E., & Unger, J. (2015). Photorealistic rendering of mixed reality scenes. Computer Graphics Forum, 34(2), 643\u2013665. https:\/\/doi.org\/10.1111\/cgf.12591 .","journal-title":"Computer Graphics Forum"},{"key":"1070_CR15","doi-asserted-by":"crossref","unstructured":"Menze, M., & Geiger, A. (2015). Object scene flow for autonomous vehicles. In Proceedings of IEEE conference on computer vision and pattern recognition (CVPR).","DOI":"10.1109\/CVPR.2015.7298925"},{"key":"1070_CR16","doi-asserted-by":"crossref","unstructured":"Movshovitz-Attias, Y., Kanade, T., & Sheikh, Y. (2016). How useful is photo-realistic rendering for visual learning? In Proceedings of the European conference on computer vision (ECCV) workshops.","DOI":"10.1007\/978-3-319-49409-8_18"},{"key":"1070_CR17","doi-asserted-by":"crossref","unstructured":"Peng, X., Sun, B., Ali, K., & Saenko, K. (2015). Learning deep object detectors from 3D models. In Proceedings of the IEEE international conference on computer vision (ICCV).","DOI":"10.1109\/ICCV.2015.151"},{"key":"1070_CR18","doi-asserted-by":"crossref","unstructured":"Philbin, J., Chum, O., Isard, M., Sivic, J., & Zisserman, A. (2007). Object retrieval with large vocabularies and fast spatial matching. In: Proceedings of IEEE conference on computer vision and pattern recognition (CVPR).","DOI":"10.1109\/CVPR.2007.383172"},{"key":"1070_CR19","doi-asserted-by":"crossref","unstructured":"Pishchulin, L., Jain, A., Wojek, C., Andriluka, M., Thorm\u00e4hlen, T., & Schiele, B. (2011). Learning people detection models from few training samples. In Proceedings of IEEE conference on computer vision and pattern recognition (CVPR).","DOI":"10.1109\/CVPR.2011.5995574"},{"key":"1070_CR20","first-page":"91","volume-title":"Advances in Neural Information Processing Systems","author":"S Ren","year":"2015","unstructured":"Ren, S., He, K., Girshick, R., & Sun, J. (2015). Faster R-CNN: Towards real-time object detection with region proposal networks. In C. Cortes, N. D. Lawrence, D. D. Lee, M. Sugiyama, & R. Garnett (Eds.), Advances in Neural Information Processing Systems (Vol. 28, pp. 91\u201399). Red Hook, NY: Curran Associates Inc."},{"key":"1070_CR21","doi-asserted-by":"crossref","unstructured":"Richter, S. R., Vineet, V., Roth, S., & Koltun, V. (2016). Playing for data: Ground truth from computer games. In Proceedings of the European conference on computer vision (ECCV).","DOI":"10.1007\/978-3-319-46475-6_7"},{"key":"1070_CR22","doi-asserted-by":"crossref","unstructured":"Ros, G., Sellart, L., Materzynska, J., Vazquez, D., & Lopez, A. (2016). The synthia dataset: A large collection of synthetic images for semantic segmentation of urban scenes. In Proceedings of IEEE conference on computer vision and pattern recognition (CVPR).","DOI":"10.1109\/CVPR.2016.352"},{"key":"1070_CR23","doi-asserted-by":"publisher","first-page":"24","DOI":"10.1016\/j.cviu.2014.12.006","volume":"137","author":"A Rozantsev","year":"2015","unstructured":"Rozantsev, A., Lepetit, V., & Fua, P. (2015). On rendering synthetic images for training an object detector. Computer Vision and Image Understanding (CVIU), 137, 24\u201337.","journal-title":"Computer Vision and Image Understanding (CVIU)"},{"key":"1070_CR24","doi-asserted-by":"crossref","unstructured":"Shafaei, A., Little, J. J., & Schmidt, M. (2016). Play and learn: Using video games to train computer vision models. arXiv:1608.01745","DOI":"10.5244\/C.30.26"},{"key":"1070_CR25","unstructured":"Simonyan, K., & Zisserman, A. (2015). Very deep convolutional networks for large-scale image recognition. In Proceedings of the international conference on learning representations (ICLR)."},{"key":"1070_CR26","doi-asserted-by":"crossref","unstructured":"Stark, M., Goesele, M., & Schiele, B. (2010). Back to the future: Learning shape models from 3D CAD data. In Proceedings of the British machine vision conference (BMVC).","DOI":"10.5244\/C.24.106"},{"key":"1070_CR27","doi-asserted-by":"crossref","unstructured":"Su, H., Qi, C. R., Li, Y., & Guibas, L. J. (2015). Render for CNN: viewpoint estimation in images using CNNS trained with rendered 3D model views. In Proceedings of the IEEE international conference on computer vision (ICCV).","DOI":"10.1109\/ICCV.2015.308"},{"key":"1070_CR28","doi-asserted-by":"crossref","unstructured":"Taigman, Y., Yang, M., Ranzato, M., & Wolf, L. (2014). Deepface: Closing the gap to human-level performance in face verification. In Proceedings of IEEE conference on computer vision and pattern recognition (CVPR).","DOI":"10.1109\/CVPR.2014.220"},{"key":"1070_CR29","unstructured":"Teichmann, M., Weber, M., Z\u00f6llner, J. M., Cipolla, R., & Urtasun, R. (2016). Multinet: Real-time joint semantic reasoning for autonomous driving. arXiv:1612.07695 ."},{"key":"1070_CR30","unstructured":"Varol, G., Romero, J., Martin, X., Mahmood, N., Black, M. J., Laptev, I. et al. (2017). Learning from synthetic humans. arXiv:1701.01370 ."},{"key":"1070_CR31","doi-asserted-by":"crossref","unstructured":"Xie, J., Kiefel, M., Sun, M. T., & Geiger, A. (2016). Semantic instance annotation of street scenes by 3D to 2D label transfer. In Proceedings of IEEE conference on computer vision and pattern recognition (CVPR).","DOI":"10.1109\/CVPR.2016.401"},{"key":"1070_CR32","unstructured":"Zhang, Y., Qiu, W., Chen, Q., Hu, X., & Yuille, A. L. (2016a). Unrealstereo: A synthetic dataset for analyzing stereo vision. arXiv:1612.04647 ."},{"key":"1070_CR33","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Song, S., Yumer, E., Savva, M., Lee, J., Jin, H., et al. (2016b). Physically-based rendering for indoor scene understanding using convolutional neural networks. arXiv:1612.07429 .","DOI":"10.1109\/CVPR.2017.537"},{"key":"1070_CR34","unstructured":"Zhu, Y., Mottaghi, R., Kolve, E., Lim, J. J., Gupta, A., Fei-Fei, L., et al. (2016). Target-driven visual navigation in indoor scenes using deep reinforcement learning. arXiv:1609.05143 ."}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11263-018-1070-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-018-1070-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-018-1070-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,10,12]],"date-time":"2019-10-12T05:35:05Z","timestamp":1570858505000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11263-018-1070-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,3,7]]},"references-count":34,"journal-issue":{"issue":"9","published-print":{"date-parts":[[2018,9]]}},"alternative-id":["1070"],"URL":"https:\/\/doi.org\/10.1007\/s11263-018-1070-x","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2018,3,7]]},"assertion":[{"value":"29 July 2017","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 February 2018","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 March 2018","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}