{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T14:50:53Z","timestamp":1784299853863,"version":"3.55.0"},"reference-count":200,"publisher":"Springer Science and Business Media LLC","issue":"8","license":[{"start":{"date-parts":[[2023,5,17]],"date-time":"2023-05-17T00:00:00Z","timestamp":1684281600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,5,17]],"date-time":"2023-05-17T00:00:00Z","timestamp":1684281600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"National Key Research and Development Program of China","award":["2018AAA0100500"],"award-info":[{"award-number":["2018AAA0100500"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2023,8]]},"DOI":"10.1007\/s11263-023-01784-z","type":"journal-article","created":{"date-parts":[[2023,5,17]],"date-time":"2023-05-17T17:01:55Z","timestamp":1684342915000},"page":"2122-2152","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":169,"title":["Multi-Modal 3D Object Detection in Autonomous Driving: A Survey"],"prefix":"10.1007","volume":"131","author":[{"given":"Yingjie","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qiuyu","family":"Mao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hanqi","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiajun","family":"Deng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yu","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jianmin","family":"Ji","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Houqiang","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6520-255X","authenticated-orcid":false,"given":"Yanyong","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,5,17]]},"reference":[{"key":"1784_CR1","doi-asserted-by":"crossref","unstructured":"Ahmad, W. A., Wessel, J., Ng, H. J., & Kissinger, D. (2020). IoT-ready millimeter-wave radar sensors. In IEEE global conference on artificial intelligence and Internet of Things (GCAIoT) (pp. 1\u20135).","DOI":"10.1109\/GCAIoT51063.2020.9345836"},{"key":"1784_CR2","doi-asserted-by":"crossref","unstructured":"Andriluka, M., Roth, S., & Schiele, B. (2010). Monocular 3d pose estimation and tracking by detection. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 623\u2013630).","DOI":"10.1109\/CVPR.2010.5540156"},{"issue":"10","key":"1784_CR3","doi-asserted-by":"publisher","first-page":"3782","DOI":"10.1109\/TITS.2019.2892405","volume":"20","author":"E Arnold","year":"2019","unstructured":"Arnold, E., Al-Jarrah, O. Y., Dianati, M., Fallah, S., Oxtoby, D., & Mouzakitis, A. (2019). A survey on 3d object detection methods for autonomous driving applications. IEEE Transactions on Intelligent Transportation Systems (TITS), 20(10), 3782\u20133795.","journal-title":"IEEE Transactions on Intelligent Transportation Systems (TITS)"},{"key":"1784_CR4","doi-asserted-by":"crossref","unstructured":"Asvadi, A., Garrote, L., Premebida, C., Peixoto, P., & Nunes, U. (2017). Multimodal vehicle detection: Fusing 3d-lidar and color camera data. Pattern Recognition Letters,115, 20\u201329.","DOI":"10.1016\/j.patrec.2017.09.038"},{"key":"1784_CR5","doi-asserted-by":"publisher","first-page":"20","DOI":"10.1016\/j.patrec.2017.09.038","volume":"115","author":"A Asvadi","year":"2018","unstructured":"Asvadi, A., Garrote, L., Premebida, C., Peixoto, P., & Nunes, U. J. (2018). Multimodal vehicle detection: Fusing 3d-lidar and color camera data. Pattern Recognition Letters, 115, 20\u201329.","journal-title":"Pattern Recognition Letters"},{"key":"1784_CR6","doi-asserted-by":"crossref","unstructured":"Bai, X., Hu, Z., Zhu, X., Huang, Q., Chen, Y., Fu, H., & Tai, C. L. (2022). Transfusion: Robust lidar-camera fusion for 3d object detection with transformers. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 1090\u20131099).","DOI":"10.1109\/CVPR52688.2022.00116"},{"key":"1784_CR7","doi-asserted-by":"crossref","unstructured":"Beltr\u00e1n, J., Guindel, C., Moreno, F. M., Cruzado, D., Garc\u00eda, F., & De\u00a0La\u00a0Escalera, A. (2018). Birdnet: A 3d object detection framework from lidar information. In 2018 21st international conference on intelligent transportation systems (ITSC) (pp. 3517\u20133523).","DOI":"10.1109\/ITSC.2018.8569311"},{"key":"1784_CR8","doi-asserted-by":"crossref","unstructured":"Caesar, H., Bankiti, V., Lang, A. H., Vora, S., Liong, V. E., Xu, Q., Krishnan, A., Pan, Y., Baldan, G., & Beijbom, O. (2020). nuscenes: A multimodal dataset for autonomous driving. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 11618\u201311628).","DOI":"10.1109\/CVPR42600.2020.01164"},{"key":"1784_CR9","unstructured":"Caine, B., Roelofs, R., Vasudevan, V., Ngiam, J., Chai, Y., Chen, Z., & Shlens, J. (2021). Pseudo-labeling for scalable 3d object detection. CoRR abs arXiv:2103.02093"},{"key":"1784_CR10","doi-asserted-by":"publisher","first-page":"125","DOI":"10.1016\/j.robot.2018.11.002","volume":"111","author":"L Caltagirone","year":"2019","unstructured":"Caltagirone, L., Bellone, M., Svensson, L., & Wahde, M. (2019). Lidar-camera fusion for road detection using fully convolutional neural networks. Robotics and Autonomous Systems, 111, 125\u2013131.","journal-title":"Robotics and Autonomous Systems"},{"key":"1784_CR11","doi-asserted-by":"crossref","unstructured":"Carr, P., Sheikh, Y., & Matthews, I. (2012). Monocular object detection using 3d geometric primitives. In A. Fitzgibbon, S. Lazebnik, P. Perona, Y. Sato, & C. Schmid (Eds.), European conference on computer vision (ECCV) (pp. 864\u2013878).","DOI":"10.1007\/978-3-642-33718-5_62"},{"key":"1784_CR12","doi-asserted-by":"crossref","unstructured":"Chadwick, S., Maddern, W., & Newman, P. (2019). Distant vehicle detection using radar and vision. In IEEE international conference on robotics and automation (ICRA) (pp. 8311\u20138317).","DOI":"10.1109\/ICRA.2019.8794312"},{"key":"1784_CR13","doi-asserted-by":"crossref","unstructured":"Chang, M. F., Lambert, J., Sangkloy, P., Singh, J., Bak, S., Hartnett, A., Wang, D., Carr, P., Lucey, S., Ramanan, D., & Hays, J. (2019). Argoverse: 3d tracking and forecasting with rich maps. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 8740\u20138749).","DOI":"10.1109\/CVPR.2019.00895"},{"key":"1784_CR14","doi-asserted-by":"crossref","unstructured":"Charles, R. Q., Su, H., Kaichun, M., & Guibas, L. J. (2017). Pointnet: Deep learning on point sets for 3d classification and segmentation. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 77\u201385).","DOI":"10.1109\/CVPR.2017.16"},{"key":"1784_CR15","doi-asserted-by":"publisher","unstructured":"Chen, X., Kundu, K., Zhang, Z., Ma, H., Fidler, S., & Urtasun, R. (2016). Monocular 3d object detection for autonomous driving. In 2016 IEEE conference on computer vision and pattern recognition (CVPR), Las Vegas, NV, USA, June 27\u201330, 2016 (pp. 2147\u20132156). IEEE Computer Society. https:\/\/doi.org\/10.1109\/CVPR.2016.236","DOI":"10.1109\/CVPR.2016.236"},{"key":"1784_CR16","doi-asserted-by":"crossref","unstructured":"Chen, Z., Li, Z., Zhang, S., Fang, L., Jiang, Q., & Zhao, F. (2022b). Autoalignv2: Deformable feature aggregation for dynamic multi-modal 3d object detection. CoRR. arXiv:2207.10316","DOI":"10.1007\/978-3-031-20074-8_36"},{"key":"1784_CR17","doi-asserted-by":"crossref","unstructured":"Chen, Z., Li, Z., Zhang, S., Fang, L., Jiang, Q., Zhao, F., Zhou, B., & Zhao, H. (2022c). AutoAlign: Pixel-instance feature aggregation for multi-modal 3d object detection. In IJCAI.","DOI":"10.24963\/ijcai.2022\/116"},{"key":"1784_CR18","unstructured":"Chen, Y., Liu, J., Qi, X., Zhang, X., Sun, J., & Jia, J. (2022a). Scaling up kernels in 3d CNNs. arXiv preprint arXiv:2206.10555"},{"key":"1784_CR19","doi-asserted-by":"crossref","unstructured":"Chen, X., Ma, H., Wan, J., Li, B., & Xia, T. (2017). Multi-view 3d object detection network for autonomous driving. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 1907\u20131915).","DOI":"10.1109\/CVPR.2017.691"},{"issue":"5","key":"1784_CR20","doi-asserted-by":"publisher","first-page":"1259","DOI":"10.1109\/TPAMI.2017.2706685","volume":"40","author":"X Chen","year":"2018","unstructured":"Chen, X., Kundu, K., Zhu, Y., Ma, H., Fidler, S., & Urtasun, R. (2018). 3d object proposals using stereo imagery for accurate object class detection. IEEE Transactions on Pattern Recognition and Machine Intelligence (TPAMI), 40(5), 1259\u20131272.","journal-title":"IEEE Transactions on Pattern Recognition and Machine Intelligence (TPAMI)"},{"key":"1784_CR21","doi-asserted-by":"crossref","unstructured":"Chen, L., Lin, S., Lu, X., Cao, D., Wu, H., Guo, C., Liu, C., & Wang, F. Y. (2021). Deep neural network based vehicle and pedestrian detection for autonomous driving: A survey. IEEE Transactions on Intelligent Transportation Systems (TITS), 22(6), 3234\u20133246.","DOI":"10.1109\/TITS.2020.2993926"},{"issue":"4","key":"1784_CR22","doi-asserted-by":"publisher","first-page":"834","DOI":"10.1109\/TPAMI.2017.2699184","volume":"40","author":"LC Chen","year":"2018","unstructured":"Chen, L. C., Papandreou, G., Kokkinos, I., Murphy, K., & Yuille, A. L. (2018). DeepLab: Semantic image segmentation with deep convolutional nets, atrous convolution, and fully connected CRFs. IEEE Transactions on Pattern Recognition and Machine Intelligence (TPAMI), 40(4), 834\u2013848.","journal-title":"IEEE Transactions on Pattern Recognition and Machine Intelligence (TPAMI)"},{"issue":"12","key":"1784_CR23","doi-asserted-by":"publisher","first-page":"5110","DOI":"10.1109\/TITS.2019.2949005","volume":"21","author":"L Chen","year":"2019","unstructured":"Chen, L., Zou, Q., Pan, Z., Lai, D., & Cao, D. (2019). Surrounding vehicle detection using an FPGA panoramic camera and deep CNNs. IEEE Transactions on Intelligent Transportation Systems, 21(12), 5110\u20135122.","journal-title":"IEEE Transactions on Intelligent Transportation Systems"},{"key":"1784_CR24","doi-asserted-by":"crossref","unstructured":"Chu, X., Deng, J., Li, Y., Yuan, Z., Zhang, Y., Ji, J., & Zhang, Y. (2021). Neighbor-vote: Improving monocular 3d object detection through neighbor distance voting. In ACM international conference on multimedia (ACM MM), ACM (pp. 5239\u20135247).","DOI":"10.1145\/3474085.3475641"},{"key":"1784_CR25","doi-asserted-by":"crossref","unstructured":"Cordts, M., Omran, M., Ramos, S., Rehfeld, T., Enzweiler, M., Benenson, R., Franke, U., Roth, S., & Schiele, B. (2016). The cityscapes dataset for semantic urban scene understanding. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 3213\u20133223).","DOI":"10.1109\/CVPR.2016.350"},{"key":"1784_CR26","doi-asserted-by":"crossref","unstructured":"Cui, Y., Chen, R., Chu, W., Chen, L., Tian, D., Li, Y., & Cao, D. (2021). Deep learning for image and point cloud fusion in autonomous driving: A review. IEEE Transactions on Intelligent Transportation Systems (TITS), 23, 1\u201318.","DOI":"10.1109\/TITS.2020.3023541"},{"key":"1784_CR27","doi-asserted-by":"crossref","unstructured":"de Paula Veronese, L., Auat-Cheein, F., Mutz, F., Oliveira-Santos, T., Guivant, J. E., de Aguiar, E., Badue, C. & De Souza, A. F. (2020). Evaluating the limits of a lidar for an autonomous driving localization. IEEE Transactions on Intelligent Transportation Systems (TITS), 22(3), 1449\u20131458.","DOI":"10.1109\/TITS.2020.2971054"},{"key":"1784_CR28","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.J., Li, K., & Fei-Fei, L. (2009). Imagenet: A large-scale hierarchical image database. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 248\u2013255).","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"1784_CR29","doi-asserted-by":"crossref","unstructured":"Deng, J., Shi, S., Li, P., Zhou, W., Zhang, Y., & Li, H. (2020). Voxel R-CNN: Towards high performance voxel-based 3d object detection. arXiv:2012.15712","DOI":"10.1609\/aaai.v35i2.16207"},{"issue":"12","key":"1784_CR30","doi-asserted-by":"publisher","first-page":"4722","DOI":"10.1109\/TCSVT.2021.3100848","volume":"31","author":"J Deng","year":"2021","unstructured":"Deng, J., Zhou, W., Zhang, Y., & Li, H. (2021). From multi-view to hollow-3d: Hallucinated hollow-3d R-CNN for 3d object detection. IEEE Transactions on Circuits and Systems for Video Technology (TCSVT), 31(12), 4722\u20134734.","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology (TCSVT)"},{"key":"1784_CR31","unstructured":"Denninger, M., Sundermeyer, M., Winkelbauer, D., Zidan, Y., Olefir, D., Elbadrawy, M., Lodhi, A., & Katam, H. (2019). BlenderProc. CoRR. arXiv:1911.01911."},{"key":"1784_CR32","unstructured":"Deschaud, J. E. (2021). KITTI-CARLA: A KITTI-like dataset generated by CARLA simulator. arXiv preprint arXiv:2109.00892"},{"key":"1784_CR33","unstructured":"Ding, Z., Hu, Y., Ge, R., Huang, L., Chen, S., Wang, Y., & Liao, J. (2020). 1st place solution for Waymo open dataset challenge: 3d detection and domain adaptation. CoRR abs arXiv:2006.15505"},{"key":"1784_CR34","unstructured":"Dosovitskiy, A., Ros, G., Codevilla, F., Lopez, A., & Koltun, V. (2017). CARLA: An open urban driving simulator. In Proceedings of the annual conference on robot learning (pp. 1\u201316)"},{"key":"1784_CR35","unstructured":"Engelberg, T., & Niem, W. (2009). Method for classifying an object using a stereo camera. U.S. Patent App. 10\/589,641."},{"key":"1784_CR36","doi-asserted-by":"publisher","first-page":"2179","DOI":"10.1109\/TPAMI.2008.260","volume":"31","author":"M Enzweiler","year":"2009","unstructured":"Enzweiler, M., & Gavrila, D. M. (2009). Monocular pedestrian detection: Survey and experiments. IEEE Transactions on Pattern Recognition and Machine Intelligence (TPAMI), 31, 2179\u20132195.","journal-title":"IEEE Transactions on Pattern Recognition and Machine Intelligence (TPAMI)"},{"issue":"2","key":"1784_CR37","doi-asserted-by":"publisher","first-page":"303","DOI":"10.1007\/s11263-009-0275-4","volume":"88","author":"M Everingham","year":"2010","unstructured":"Everingham, M., Van Gool, L., Williams, C. K., Winn, J., & Zisserman, A. (2010). The pascal visual object classes (VOC) challenge. International Journal of Computer Vision, 88(2), 303\u2013338.","journal-title":"International Journal of Computer Vision"},{"key":"1784_CR38","doi-asserted-by":"crossref","unstructured":"Fan, L., Pang, Z., Zhang, T., Wang, Y. X., Zhao, H., Wang, F., Wang, N., & Zhang, Z. (2022). Embracing single stride 3d object detector with sparse transformer. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 8458\u20138468).","DOI":"10.1109\/CVPR52688.2022.00827"},{"key":"1784_CR39","doi-asserted-by":"crossref","unstructured":"Fan, L., Xiong, X., Wang, F., Wang, N., & Zhang, Z. (2021). RangeDet: In defense of range view for lidar-based 3d object detection. CoRR abs arXiv:2103.10039","DOI":"10.1109\/ICCV48922.2021.00291"},{"key":"1784_CR40","doi-asserted-by":"publisher","first-page":"4220","DOI":"10.3390\/s20154220","volume":"20","author":"J Fayyad","year":"2020","unstructured":"Fayyad, J., Jaradat, M., Gruyer, D., & Najjaran, H. (2020). Deep learning sensor fusion for autonomous vehicle perception and localization: A review. Sensors, 20, 4220.","journal-title":"Sensors"},{"key":"1784_CR41","doi-asserted-by":"crossref","unstructured":"Feng, D., Haase-Sch\u00fctz, C., Rosenbaum, L., Hertlein, H., Gl\u00e4ser, C., Timm, F., Wiesbeck, W., & Dietmayer, K. (2021). Deep multi-modal object detection and semantic segmentation for autonomous driving: Datasets, methods, and challenges. IEEE Transactions on Intelligent Transportation Systems (TITS), 22(3), 1341\u20131360.","DOI":"10.1109\/TITS.2020.2972974"},{"key":"1784_CR42","unstructured":"G\u00e4hlert, N., Jourdan, N., Cordts, M., Franke, U., & Denzler, J. (2020). Cityscapes 3d: Dataset and benchmark for 9 DoF vehicle detection. CoRR. arXiv:2006.07864."},{"key":"1784_CR43","doi-asserted-by":"crossref","unstructured":"Gaidon, A., Wang, Q., Cabon, Y., & Vig, E. (2016). Virtual worlds as proxy for multi-object tracking analysis. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 4340\u20134349).","DOI":"10.1109\/CVPR.2016.470"},{"key":"1784_CR44","doi-asserted-by":"crossref","unstructured":"Geiger, A., Lenz, P., & Urtasun, R. (2012). Are we ready for autonomous driving? The KITTI vision benchmark suite. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 3354\u20133361).","DOI":"10.1109\/CVPR.2012.6248074"},{"issue":"11","key":"1784_CR45","doi-asserted-by":"publisher","first-page":"1231","DOI":"10.1177\/0278364913491297","volume":"32","author":"A Geiger","year":"2013","unstructured":"Geiger, A., Lenz, P., Stiller, C., & Urtasun, R. (2013). Vision meets robotics: The KITTI dataset. The International Journal of Robotics Research (IJRR), 32(11), 1231\u20131237.","journal-title":"The International Journal of Robotics Research (IJRR)"},{"issue":"3","key":"1784_CR46","doi-asserted-by":"publisher","first-page":"227","DOI":"10.1007\/BF00115697","volume":"6","author":"D Geiger","year":"1991","unstructured":"Geiger, D., & Yuille, A. L. (1991). A common framework for image segmentation. International Journal on Computer Vision (IJCV), 6(3), 227\u2013243.","journal-title":"International Journal on Computer Vision (IJCV)"},{"key":"1784_CR47","doi-asserted-by":"crossref","unstructured":"Girshick, R. (2015). Fast R-CNN. In IEEE international conference on computer vision (ICCV) (pp. 1440\u20131448).","DOI":"10.1109\/ICCV.2015.169"},{"key":"1784_CR48","doi-asserted-by":"crossref","unstructured":"Girshick, R., Donahue, J., Darrell, T., & Malik, J. (2014). Rich feature hierarchies for accurate object detection and semantic segmentation. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 580\u2013587).","DOI":"10.1109\/CVPR.2014.81"},{"key":"1784_CR49","doi-asserted-by":"crossref","unstructured":"Goodfellow, I., Pouget-Abadie, J., Mirza, M., Xu, B., Warde-Farley, D., Ozair, S., Courville, A., & Bengio, Y. (2020). Generative adversarial networks. Communications of the ACM, 63(11), 139\u2013144.","DOI":"10.1145\/3422622"},{"key":"1784_CR50","doi-asserted-by":"crossref","unstructured":"Guan, T., Wang, J., Lan, S., Chandra, R., Wu, Z., Davis, L., & Manocha, D. (2022). M3DETR: Multi-representation, multi-scale, mutual-relation 3d object detection with transformers. In Proceedings of the IEEE\/CVF winter conference on applications of computer vision (pp. 772\u2013782).","DOI":"10.1109\/WACV51458.2022.00235"},{"key":"1784_CR51","doi-asserted-by":"crossref","unstructured":"Guizilini, V., Li, J., Ambru\u015f, R., & Gaidon, A. (2021). Geometric unsupervised domain adaptation for semantic segmentation. In Proceedings of the IEEE\/CVF international conference on computer vision (pp. 8537\u20138547).","DOI":"10.1109\/ICCV48922.2021.00842"},{"key":"1784_CR52","doi-asserted-by":"crossref","unstructured":"Guo, X., Shi, S., Wang, X., & Li, H. (2021). LIGA-Stereo: Learning lidar geometry aware representations for stereo-based 3d detector. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 3153\u20133163).","DOI":"10.1109\/ICCV48922.2021.00314"},{"issue":"8","key":"1784_CR53","doi-asserted-by":"publisher","first-page":"3135","DOI":"10.1109\/TITS.2019.2926042","volume":"21","author":"J Guo","year":"2019","unstructured":"Guo, J., Kurup, U., & Shah, M. (2019). Is it safe to drive? An overview of factors, metrics, and datasets for driveability assessment in autonomous driving. IEEE Transactions on Intelligent Transportation Systems, 21(8), 3135\u20133151.","journal-title":"IEEE Transactions on Intelligent Transportation Systems"},{"key":"1784_CR54","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., & Girshick, R. B. (2017). Mask R-CNN. In IEEE international conference on computer vision (ICCV) (pp. 2980\u20132988).","DOI":"10.1109\/ICCV.2017.322"},{"key":"1784_CR55","doi-asserted-by":"crossref","unstructured":"He, C., Zeng, H., Huang, J., Hua, X. S., & Zhang, L. (2020). Structure aware single-stage 3d object detection from point cloud. In 2020 IEEE\/CVF conference on computer vision and pattern recognition (CVPR), Seattle, WA, USA, June 13\u201319,2020 (pp. 11870\u201311879). Computer Vision Foundation\/IEEE.","DOI":"10.1109\/CVPR42600.2020.01189"},{"key":"1784_CR56","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2016). Deep residual learning for image recognition. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 770\u2013778).","DOI":"10.1109\/CVPR.2016.90"},{"key":"1784_CR57","first-page":"8409","volume":"33","author":"T He","year":"2019","unstructured":"He, T., & Soatto, S. (2019). Mono3d++: Monocular 3d vehicle detection with two-scale 3d hypotheses and task priors. Association for the Advancement of Artificial Intelligence (AAAI), 33, 8409\u20138416.","journal-title":"Association for the Advancement of Artificial Intelligence (AAAI)"},{"key":"1784_CR58","unstructured":"Hinton, G., Vinyals, O., & Dean, J. (2015). Distilling the knowledge in a neural network. arXiv preprint arXiv:1503.02531"},{"key":"1784_CR59","doi-asserted-by":"crossref","unstructured":"Hoda\u0148, T., Vineet, V., Gal, R., Shalev, E., Hanzelka, J., Connell, T., Urbina, P., Sinha, S. N., & Guenter, B. (2019). Photorealistic image synthesis for object instance detection. In 2019 IEEE international conference on image processing (ICIP), IEEE (pp. 66\u201370).","DOI":"10.1109\/ICIP.2019.8803821"},{"key":"1784_CR60","doi-asserted-by":"crossref","unstructured":"Hu, Y., Ding, Z., Ge, R., Shao, W., Huang, L., Li, K., & Liu, Q. (2021). AFDetV2: Rethinking the necessity of the second stage for object detection from point clouds. arXiv preprint arXiv:2112.09205","DOI":"10.1609\/aaai.v36i1.19980"},{"key":"1784_CR61","doi-asserted-by":"crossref","unstructured":"Hu, P., Ziglar, J., Held, D., & Ramanan, D. (2020). What you see is what you get: Exploiting visibility for 3d object detection. In IEEE conference on computer vision and pattern recognition (CVPR), computer vision foundation\/IEEE (pp. 10998\u201311006).","DOI":"10.1109\/CVPR42600.2020.01101"},{"key":"1784_CR62","unstructured":"Huang, J., & Huang, G. (2022). BEVDet4D: Exploit temporal cues in multi-camera 3d object detection. arXiv preprint arXiv:2203.17054"},{"key":"1784_CR63","unstructured":"Huang, J., Huang, G., Zhu, Z., & Du, D. (2021). BEVDet: High-performance multi-camera 3d object detection in bird-eye-view. arXiv preprint arXiv:2112.11790"},{"key":"1784_CR64","doi-asserted-by":"crossref","unstructured":"Huang, G., Liu, Z., Van Der\u00a0Maaten, L., & Weinberger, K. Q. (2017a). Densely connected convolutional networks. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 2261\u20132269).","DOI":"10.1109\/CVPR.2017.243"},{"issue":"9","key":"1784_CR65","doi-asserted-by":"publisher","first-page":"2364","DOI":"10.1109\/TITS.2016.2639582","volume":"18","author":"P Huang","year":"2017","unstructured":"Huang, P., Cheng, M., Chen, Y., Luo, H., Wang, C., & Li, J. (2017). Traffic sign occlusion detection using mobile laser scanning point clouds. IEEE Transactions on Intelligent Transportation Systems, 18(9), 2364\u20132376.","journal-title":"IEEE Transactions on Intelligent Transportation Systems"},{"key":"1784_CR66","first-page":"35","volume":"12360","author":"T Huang","year":"2020","unstructured":"Huang, T., Liu, Z., Chen, X., & Bai, X. (2020). EPNet: Enhancing point features with image semantics for 3d object detection. European Conference on Computer Vision (ECCV), 12360, 35\u201352.","journal-title":"European Conference on Computer Vision (ECCV)"},{"issue":"10","key":"1784_CR67","doi-asserted-by":"publisher","first-page":"2702","DOI":"10.1109\/TPAMI.2019.2926463","volume":"42","author":"X Huang","year":"2019","unstructured":"Huang, X., Wang, P., Cheng, X., Zhou, D., Geng, Q., & Yang, R. (2019). The apolloscape open dataset for autonomous driving and its application. IEEE Transactions on Pattern Recognition and Machine Intelligence (TPAMI), 42(10), 2702\u20132719.","journal-title":"IEEE Transactions on Pattern Recognition and Machine Intelligence (TPAMI)"},{"issue":"2","key":"1784_CR68","first-page":"20:1","volume":"50","author":"A Ioannidou","year":"2017","unstructured":"Ioannidou, A., Chatzilari, E., Nikolopoulos, S., & Kompatsiaris, I. (2017). Deep learning advances in computer vision with 3d data: A survey. ACM Computing Survey, 50(2), 20:1-20:38.","journal-title":"ACM Computing Survey"},{"key":"1784_CR69","doi-asserted-by":"crossref","unstructured":"Jiang, M., Wu, Y., & Lu, C. (2018). PointSIFT: A sift-like network module for 3d point cloud semantic segmentation. CoRR abs arXiv:1807.00652","DOI":"10.1109\/IGARSS.2019.8900102"},{"key":"1784_CR70","unstructured":"Jiao, Y., Jie, Z., Chen, S., Chen, J., Wei, X., Ma, L., & Jiang, Y. G. (2022). MSMDfusion: Fusing lidar and camera at multiple scales with multi-depth seeds for 3d object detection. arXiv preprint arXiv:2209.03102"},{"key":"1784_CR71","doi-asserted-by":"crossref","unstructured":"Kar, A., Prakash, A., Liu, M. Y., Cameracci, E., Yuan, J., Rusiniak, M., Acuna, D., Torralba, A., & Fidler, S. (2019). Meta-Sim: Learning to generate synthetic datasets. In IEEE international conference on computer vision (ICCV) (pp. 4550\u20134559).","DOI":"10.1109\/ICCV.2019.00465"},{"key":"1784_CR72","doi-asserted-by":"crossref","unstructured":"Kellner, D., Klappstein, J., & Dietmayer, K. (2012). Grid-based DBSCAN for clustering extended objects in radar data. In IEEE intelligent vehicles symposium (IV) (pp. 365\u2013370).","DOI":"10.1109\/IVS.2012.6232167"},{"key":"1784_CR73","unstructured":"Kesten, R., Usman, M., Houston, J., Pandya, T., Nadhamuni, K., Ferreira, A., Yuan, M., Low, B., Jain, A., Ondruska, P., Omari, S., Shah, S., Kulkarni, A., Kazakova, A., Tao, C., Platinsky, L., Jiang, W., & Shet, V. (2019). Level 5 perception dataset 2020. https:\/\/level-5.global\/level5\/data\/"},{"key":"1784_CR74","doi-asserted-by":"publisher","unstructured":"Kim, Y. (2014). Convolutional Neural Networks for Sentence Classification. In Proceedings of the 2014 conference on empirical methods in natural language processing (EMNLP), Doha, Qatar, October 25\u201329, 2014 (pp. 1746\u20131751). ACL. https:\/\/doi.org\/10.3115\/v1\/d14-1181","DOI":"10.3115\/v1\/d14-1181"},{"key":"1784_CR75","doi-asserted-by":"crossref","unstructured":"Kim, K., & Woo, W. (2005a). A multi-view camera tracking for modeling of indoor environment. Berlin.","DOI":"10.1007\/978-3-540-30541-5_36"},{"key":"1784_CR76","doi-asserted-by":"crossref","unstructured":"Kim, K., & Woo, W. (2005b). A multi-view camera tracking for modeling of indoor environment. In K. Aizawa, Y. Nakamura & S. Satoh (Eds.), Advances in multimedia information processing\u2014PCM 2004 (pp. 288\u2013297).","DOI":"10.1007\/978-3-540-30541-5_36"},{"key":"1784_CR77","doi-asserted-by":"crossref","unstructured":"Kim, Y., Choi, J.W., & Kum, D. (2020). GRIF Net: Gated region of interest fusion network for robust 3d object detection from radar point cloud and monocular image. In IROS (pp. 10857\u201310864).","DOI":"10.1109\/IROS45743.2020.9341177"},{"key":"1784_CR78","unstructured":"Krizhevsky, A., Sutskever, I., & Hinton, G. E. (2012). ImageNet classification with deep convolutional neural networks. In Advances in neural information processing systems (NeurIPS) (vol. 25)."},{"key":"1784_CR79","doi-asserted-by":"crossref","unstructured":"Ku, J., Mozifian, M., Lee, J., Harakeh, A., & Waslander, S. L. (2018). Joint 3d proposal generation and object detection from view aggregation. In IEEE international conference on intelligent robots and systems (IROS) (pp. 1\u20138).","DOI":"10.1109\/IROS.2018.8594049"},{"key":"1784_CR80","doi-asserted-by":"crossref","unstructured":"Lang, A. H., Vora, S., Caesar, H., Zhou, L., Yang, J., & Beijbom, O. (2019). PointPillars: Fast encoders for object detection from point clouds. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 12697\u201312705).","DOI":"10.1109\/CVPR.2019.01298"},{"key":"1784_CR81","unstructured":"Lee, S. (2020). Deep learning on radar centric 3d object detection. CoRR abs arXiv:2003.00851"},{"issue":"2","key":"1784_CR82","doi-asserted-by":"publisher","first-page":"027004","DOI":"10.1117\/1.3535590","volume":"50","author":"CH Lee","year":"2011","unstructured":"Lee, C. H., Lim, Y. C., Kwon, S., & Lee, J. H. (2011). Stereo vision-based vehicle detection using a road feature and disparity histogram. Optical Engineering, 50(2), 027004\u2013027004.","journal-title":"Optical Engineering"},{"key":"1784_CR83","doi-asserted-by":"crossref","unstructured":"Levinson, J., & Thrun, S. (2013). Automatic online calibration of cameras and lasers. In Robotics: Science and systems (vol. 2, p. 7).","DOI":"10.15607\/RSS.2013.IX.029"},{"key":"1784_CR84","doi-asserted-by":"crossref","unstructured":"Li, P., Chen, X., & Shen, S. (2019). Stereo R-CNN based 3d object detection for autonomous driving. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 7644\u20137652).","DOI":"10.1109\/CVPR.2019.00783"},{"key":"1784_CR85","doi-asserted-by":"crossref","unstructured":"Li, Y., Yu, A. W., Meng, T., Caine, B., Ngiam, J., Peng, D., Shen, J., Lu, Y., Zhou, D., Le, Q. V., & Yuille, A. (2022). DeepFusion: Lidar-camera deep fusion for multi-modal 3d object detection. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 17182\u201317191).","DOI":"10.1109\/CVPR52688.2022.01667"},{"key":"1784_CR86","doi-asserted-by":"crossref","unstructured":"Liang, M., Yang, B., Chen, Y., Hu, R., & Urtasun, R. (2019). Multi-task multi-sensor fusion for 3d object detection. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 7337\u20137345).","DOI":"10.1109\/CVPR.2019.00752"},{"key":"1784_CR87","doi-asserted-by":"crossref","unstructured":"Liang, M., Yang, B., Wang, S., & Urtasun, R. (2018). Deep continuous fusion for multi-sensor 3d object detection. In European conference on computer vision (ECCV) (pp. 663\u2013678).","DOI":"10.1007\/978-3-030-01270-0_39"},{"key":"1784_CR88","unstructured":"Liang, Z., Zhang, M., Zhang, Z., Zhao, X., & Pu, S. (2020). RangeRCNN: Towards fast and accurate 3d object detection with range image representation. CoRR abs arXiv:2009.00206"},{"key":"1784_CR89","doi-asserted-by":"crossref","unstructured":"Lin, T.Y., Doll\u00e1r, P., Girshick, R., He, K., Hariharan, B., & Belongie, S. (2017a). Feature pyramid networks for object detection. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 936\u2013944).","DOI":"10.1109\/CVPR.2017.106"},{"key":"1784_CR90","unstructured":"Lin, T. Y., Goyal, P., Girshick, R., He, K., & Doll\u00e1r, P. (2017b). Focal loss for dense object detection. IEEE Transactions on Pattern Recognition and Machine Intelligence (TPAMI), P.P.(99), 2999\u20133007"},{"key":"1784_CR91","doi-asserted-by":"crossref","unstructured":"Lin, T. Y., Maire, M., Belongie, S., Hays, J., Perona, P., Ramanan, D., Doll\u00e1r, P., & Zitnick, C. L. (2014). Microsoft COCO: Common objects in context. In European conference on computer vision (ECCV) (pp. 740\u2013755).","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"1784_CR92","doi-asserted-by":"crossref","unstructured":"Liu, W., Anguelov, D., Erhan, D., Szegedy, C., Reed, S., Fu, C. Y., & Berg, A. C. (2016). SSD: Single shot multibox detector. In B. Leibe, J. Matas, N. Sebe, & M. Welling (Eds.), European conference on computer vision (ECCV) (pp. 21\u201337).","DOI":"10.1007\/978-3-319-46448-0_2"},{"key":"1784_CR93","unstructured":"Liu, H., Simonyan, K., & Yang, Y. (2018). DARTS: Differentiable architecture search. CoRR. arXiv:1806.09055"},{"key":"1784_CR94","doi-asserted-by":"crossref","unstructured":"Liu, Z., Tang, H., Amini, A., Yang, X., Mao, H., Rus, D., & Han, S. (2022c). BEVFusion: Multi-task multi-sensor fusion with unified bird\u2019s-eye view representation. arXiv preprint arXiv:2205.13542","DOI":"10.1109\/ICRA48891.2023.10160968"},{"key":"1784_CR95","doi-asserted-by":"crossref","unstructured":"Liu, Y., Wang, T., Zhang, X., & Sun, J. (2022a). PETR: Position embedding transformation for multi-view 3d object detection. arXiv preprint arXiv:2203.05625","DOI":"10.1007\/978-3-031-19812-0_31"},{"key":"1784_CR96","doi-asserted-by":"crossref","unstructured":"Liu, Z., Wu, Z., & T\u00f3th, R. (2020). SMOKE: Single-stage monocular 3d object detection via keypoint estimation. In IEEE conference on computer vision and pattern recognition workshops (CVPRW) (pp. 4289\u20134298).","DOI":"10.1109\/CVPRW50498.2020.00506"},{"key":"1784_CR97","doi-asserted-by":"crossref","unstructured":"Liu, Y., Yan, J., Jia, F., Li, S., Gao, Q., Wang, T., Zhang, X., & Sun, J. (2022b). PETRv2: A unified framework for 3d perception from multi-camera images. arXiv preprint arXiv:2206.01256","DOI":"10.1109\/ICCV51070.2023.00302"},{"issue":"4","key":"1784_CR98","first-page":"640","volume":"39","author":"J Long","year":"2015","unstructured":"Long, J., Shelhamer, E., & Darrell, T. (2015). Fully convolutional networks for semantic segmentation. IEEE Transactions on Pattern Recognition and Machine Intelligence (TPAMI), 39(4), 640\u2013651.","journal-title":"IEEE Transactions on Pattern Recognition and Machine Intelligence (TPAMI)"},{"key":"1784_CR99","doi-asserted-by":"crossref","unstructured":"Lu, H., Chen, X., Zhang, G., Zhou, Q., Ma, Y., & Zhao, Y. (2019). SCANet: Spatial-channel attention network for 3d object detection. In IEEE international conference on acoustics, speech and, S.P. (ICASSP) (pp. 1992\u20131996).","DOI":"10.1109\/ICASSP.2019.8682746"},{"key":"1784_CR100","doi-asserted-by":"crossref","unstructured":"Ma, X., Wang, Z., Li, H., Zhang, P., Ouyang, W., & Fan, X. (2019). Accurate monocular 3d object detection via color-embedded 3d reconstruction for autonomous driving. In IEEE international conference on computer vision (ICCV) (pp. 6851\u20136860).","DOI":"10.1109\/ICCV.2019.00695"},{"key":"1784_CR101","doi-asserted-by":"crossref","unstructured":"Mahmoud, A., Hu, J. S., & Waslander, S. L. (2022). Dense voxel fusion for 3d object detection. arXiv preprint arXiv:2203.00871","DOI":"10.1109\/WACV56688.2023.00073"},{"key":"1784_CR102","doi-asserted-by":"crossref","unstructured":"Major, B., Fontijne, D., Ansari, A., Sukhavasi, R. T., Gowaiker, R., Hamilton, M., Lee, S., & Grzechnik, S. K., Subramanian, S. (2019). Vehicle detection with automotive radar using deep learning on range-azimuth-doppler tensors. In IEEE international conference on computer vision workshop (ICCVW) (pp. 924\u2013932).","DOI":"10.1109\/ICCVW.2019.00121"},{"key":"1784_CR103","doi-asserted-by":"crossref","unstructured":"Manivasagam, S., Wang, S., Wong, K., Zeng, W., Sazanovich, M., Tan, S., Yang, B., Ma, W. C., & Urtasun, R. (2020). LiDARsim: Realistic lidar simulation by leveraging the real world. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 11167\u201311176).","DOI":"10.1109\/CVPR42600.2020.01118"},{"key":"1784_CR104","doi-asserted-by":"publisher","unstructured":"Mao, J., Xue, Y., Niu, M., Bai, H., Feng, J., Liang, X., Xu, H., & Xu, C. (2021). Voxel transformer for 3d object detection. In 2021 IEEE\/CVF international conference on computer vision (ICCV), Montreal, QC, Canada, October 10\u201317, 2021 (pp. 3144\u20133153). IEEE. https:\/\/doi.org\/10.1109\/ICCV48922.2021.00315.","DOI":"10.1109\/ICCV48922.2021.00315."},{"issue":"3","key":"1784_CR105","doi-asserted-by":"publisher","first-page":"171","DOI":"10.1023\/A:1008161528515","volume":"32","author":"R Marchand","year":"1999","unstructured":"Marchand, R., & Chaumette, F. (1999). An autonomous active vision system for complete and accurate 3d scene reconstruction. International Journal on Computer Vision (IJCV), 32(3), 171\u2013194.","journal-title":"International Journal on Computer Vision (IJCV)"},{"key":"1784_CR106","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih, V., Kavukcuoglu, K., Silver, D., Rusu, A., Veness, J., Bellemare, M., Graves, A., Riedmiller, M., Fidjeland, A., Ostrovski, G., Petersen, S., Beattie, C., Sadik, A., Antonoglou, I., King, H., Kumaran, D., Wierstra, D., Legg, S., & Hassabis, D. (2015). Human-level control through deep reinforcement learning. Nature, 518, 529\u201333.","journal-title":"Nature"},{"key":"1784_CR107","doi-asserted-by":"crossref","unstructured":"Mousavian, A., Anguelov, D., Flynn, J., & Ko\u0161eck\u00e1, J. (2017). 3d bounding box estimation using deep learning and geometry. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 5632\u20135640).","DOI":"10.1109\/CVPR.2017.597"},{"key":"1784_CR108","doi-asserted-by":"crossref","unstructured":"Nabati, R., & Qi, H. (2019). RRPN: Radar region proposal network for object detection in autonomous vehicles. In IEEE international conference on image processing (ICIP) (pp. 3093\u20133097).","DOI":"10.1109\/ICIP.2019.8803392"},{"key":"1784_CR109","doi-asserted-by":"crossref","unstructured":"Nabati, R., & Qi, H. (2021). CenterFusion: Center-based radar and camera fusion for 3d object detection. In IEEE winter conference on applications of computer vision (WACV) (pp. 1527\u20131536).","DOI":"10.1109\/WACV48630.2021.00157"},{"issue":"6","key":"1784_CR110","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/2508363.2508374","volume":"32","author":"M Nie\u00dfner","year":"2013","unstructured":"Nie\u00dfner, M., Zollh\u00f6fer, M., Izadi, S., & Stamminger, M. (2013). Real-time 3d reconstruction at scale using voxel hashing. ACM Transactions on Graphics (TOG), 32(6), 1\u201311.","journal-title":"ACM Transactions on Graphics (TOG)"},{"key":"1784_CR111","doi-asserted-by":"crossref","unstructured":"Pan, X., Xia, Z., Song, S., Li, L.E., & Huang, G. (2021). 3d object detection with pointformer. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 7463\u20137472).","DOI":"10.1109\/CVPR46437.2021.00738"},{"key":"1784_CR112","doi-asserted-by":"crossref","unstructured":"Pandey, G., McBride, J. R., Savarese, S., & Eustice, R. M. (2012). Automatic targetless extrinsic calibration of a 3d lidar and camera by maximizing mutual information. In Association for the advancement of artificial intelligence (AAAI) (pp. 2053\u20132059).","DOI":"10.1609\/aaai.v26i1.8379"},{"key":"1784_CR113","doi-asserted-by":"crossref","unstructured":"Pang, S., Morris, D., & Radha, H. (2020). CLOCs: Camera-lidar object candidates fusion for 3d object detection. In IEEE international conference on intelligent robots and systems (IROS) (pp. 10386\u201310393).","DOI":"10.1109\/IROS45743.2020.9341791"},{"key":"1784_CR114","doi-asserted-by":"crossref","unstructured":"Park, D., Ambrus, R., Guizilini, V., Li, J., & Gaidon, A. (2021). Is pseudo-lidar needed for monocular 3d object detection? In IEEE international conference on computer vision (ICCV) (pp. 3142\u20133152).","DOI":"10.1109\/ICCV48922.2021.00313"},{"key":"1784_CR115","unstructured":"Park, J. Y., Chu, C. W., Kim, H. W., Lim, S. J., Park, J. C., & Koo, B. K. (2009). Multi-view camera color calibration method using color checker chart. US Patent 12\/334,095"},{"key":"1784_CR116","doi-asserted-by":"crossref","unstructured":"Patil, A., Malla, S., Gang, H., & Chen, Y. T. (2019). The H3D dataset for full-surround 3D multi-object detection and tracking in crowded urban scenes. In IEEE international conference on robotics and automation (ICRA) (pp. 9552\u20139557).","DOI":"10.1109\/ICRA.2019.8793925"},{"issue":"2","key":"1784_CR117","doi-asserted-by":"publisher","first-page":"22","DOI":"10.1109\/MSP.2016.2628914","volume":"34","author":"SM Patole","year":"2017","unstructured":"Patole, S. M., Torlak, M., Wang, D., & Ali, M. (2017). Automotive radars: A review of signal processing techniques. IEEE Signal Processing Magazine, 34(2), 22\u201335.","journal-title":"IEEE Signal Processing Magazine"},{"key":"1784_CR118","doi-asserted-by":"crossref","unstructured":"Pham, Q. H., Sevestre, P., Pahwa, R. S., Zhan, H., Pang, C. H., Chen, Y., Mustafa, A., Chandrasekhar, V., & Lin, J. (2020). A* 3d dataset: Towards autonomous driving in challenging environments. In IEEE international conference on robotics and automation (ICRA) (pp. 2267\u20132273).","DOI":"10.1109\/ICRA40945.2020.9197385"},{"key":"1784_CR119","doi-asserted-by":"crossref","unstructured":"Philion, J., & Fidler, S. (2020). Lift, splat, shoot: Encoding images from arbitrary camera rigs by implicitly unprojecting to 3d. In European conference on computer vision (pp. 194\u2013210). Springer.","DOI":"10.1007\/978-3-030-58568-6_12"},{"key":"1784_CR120","doi-asserted-by":"crossref","unstructured":"Pon, A. D., Ku, J., Li, C., & Waslander, S. L. (2020). Object-centric stereo matching for 3d object detection. In IEEE international conference on robotics and automation (ICRA) (pp. 8383\u20138389).","DOI":"10.1109\/ICRA40945.2020.9196660"},{"key":"1784_CR121","doi-asserted-by":"crossref","unstructured":"Prakash, A., Boochoon, S., Brophy, M., Acuna, D., Cameracci, E., State, G., Shapira, O., & Birchfield, S. (2019). Structured domain randomization: Bridging the reality gap by context-aware synthetic data. In IEEE international conference on robotics and automation (ICRA) (pp. 7249\u20137255).","DOI":"10.1109\/ICRA.2019.8794443"},{"key":"1784_CR122","doi-asserted-by":"crossref","unstructured":"Qi, C. R., Litany, O., He, K., & Guibas, L. (2019). Deep Hough voting for 3d object detection in point clouds. In International conference on computer vision (ICCV) (pp. 9276\u20139285).","DOI":"10.1109\/ICCV.2019.00937"},{"key":"1784_CR123","doi-asserted-by":"crossref","unstructured":"Qi, C. R., Liu, W., Wu, C., Su, H., & Guibas, L. J. (2018). Frustum PointNets for 3d object detection from RGB-D data. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 918\u2013927).","DOI":"10.1109\/CVPR.2018.00102"},{"key":"1784_CR124","unstructured":"Qi, C.R., Yi, L., Su, H., & Guibas, L. J. (2017). PointNet++: Deep hierarchical feature learning on point sets in a metric space. In Advances in neural information processing systems (NeurIPS) (vol. 30)."},{"key":"1784_CR125","doi-asserted-by":"crossref","unstructured":"Qian, K., Zhu, S., Zhang, X., & Li, L. E. (2021). Robust multimodal vehicle detection in foggy weather using complementary lidar and radar signals. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 444\u2013453).","DOI":"10.1109\/CVPR46437.2021.00051"},{"key":"1784_CR126","doi-asserted-by":"crossref","unstructured":"Qin, Z., Wang, J., & Lu, Y. (2019b). Triangulation learning network: From monocular to stereo 3d object detection. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 7615\u20137623).","DOI":"10.1109\/CVPR.2019.00780"},{"key":"1784_CR127","first-page":"8851","volume":"33","author":"Z Qin","year":"2019","unstructured":"Qin, Z., Wang, J., & Lu, Y. (2019a). Monogrnet: A geometric reasoning network for monocular 3d object localization. Association for the Advancement of Artificial Intelligence (AAAI), 33, 8851\u20138858.","journal-title":"Association for the Advancement of Artificial Intelligence (AAAI)"},{"key":"1784_CR128","doi-asserted-by":"crossref","unstructured":"Redmon, J., Divvala, S., Girshick, R., & Farhadi, A. (2016). You only look once: Unified, real-time object detection. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 779\u2013788).","DOI":"10.1109\/CVPR.2016.91"},{"issue":"6","key":"1784_CR129","doi-asserted-by":"publisher","first-page":"1137","DOI":"10.1109\/TPAMI.2016.2577031","volume":"39","author":"S Ren","year":"2017","unstructured":"Ren, S., He, K., Girshick, R., & Sun, J. (2017). Faster R-CNN: Towards real-time object detection with region proposal networks. IEEE Transactions on Pattern Recognition and Machine Intelligence (TPAMI), 39(6), 1137\u20131149.","journal-title":"IEEE Transactions on Pattern Recognition and Machine Intelligence (TPAMI)"},{"key":"1784_CR130","unstructured":"Repairer Driven News (2018). Velodyne: Leading LIDAR price halved, new high-res product to improve self-driving cars. https:\/\/www.repairerdrivennews.com\/2018\/01\/02\/velodyne-leading-lidar-price-halved-new-high-res-product-to-improve-self-driving-cars\/"},{"key":"1784_CR131","doi-asserted-by":"crossref","unstructured":"Richter, S. R., Vineet, V., Roth, S., & Koltun, V. (2016). Playing for data: Ground truth from computer games. In B. Leibe, J. Matas, N. Sebe, & M. Welling (Eds.), European conference on computer vision (ECCV) (pp. 102\u2013118).","DOI":"10.1007\/978-3-319-46475-6_7"},{"issue":"2","key":"1784_CR132","doi-asserted-by":"publisher","first-page":"1700","DOI":"10.1109\/TPAMI.2022.3166687","volume":"45","author":"SR Richter","year":"2022","unstructured":"Richter, S. R., Al Haija, H. A., & Koltun, V. (2022). Enhancing photorealism enhancement. IEEE Transactions on Pattern Analysis and Machine Intelligence, 45(2), 1700\u20131715.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"1784_CR133","doi-asserted-by":"crossref","unstructured":"Riegler, G., Ulusoy, A. O., & Geiger, A. (2017). OctNet: Learning deep 3d representations at high resolutions. In IEEE conference on computer vision and pattern recognition (CVPR) IEEE Computer Society (pp. 6620\u20136629).","DOI":"10.1109\/CVPR.2017.701"},{"key":"1784_CR134","doi-asserted-by":"crossref","unstructured":"Roddick, T., & Cipolla, R. (2020). Predicting semantic map representations from images using pyramid occupancy networks. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 11138\u201311147).","DOI":"10.1109\/CVPR42600.2020.01115"},{"key":"1784_CR135","doi-asserted-by":"crossref","unstructured":"Ronneberger, O., Fischer, P., & Brox, T. (2015). U-Net: Convolutional networks for biomedical image segmentation. In: Medical image computing and computer-assisted intervention (MICCAI), (vol. 9351, pp. 234\u2013241).","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"1784_CR136","doi-asserted-by":"crossref","unstructured":"Ros, G., Sellart, L., Materzynska, J., Vazquez, D., & Lopez, A. M. (2016). The SYNTHIA dataset: A large collection of synthetic images for semantic segmentation of urban scenes. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 3234\u20133243).","DOI":"10.1109\/CVPR.2016.352"},{"key":"1784_CR137","doi-asserted-by":"crossref","unstructured":"Schlosser, J., Chow, C. K., & Kira, Z. (2016). Fusing lidar and images for pedestrian detection using convolutional neural networks. In IEEE international conference on robotics and automation (ICRA) (pp. 2198\u20132205).","DOI":"10.1109\/ICRA.2016.7487370"},{"key":"1784_CR138","doi-asserted-by":"publisher","first-page":"85","DOI":"10.1016\/j.neunet.2014.09.003","volume":"61","author":"J Schmidhuber","year":"2015","unstructured":"Schmidhuber, J. (2015). Deep learning in neural networks: An overview. Neural Networks, 61, 85\u2013117.","journal-title":"Neural Networks"},{"key":"1784_CR139","doi-asserted-by":"crossref","unstructured":"Schneider, N., Piewak, F., Stiller, C., & Franke, U. (2017). RegNet: Multimodal sensor registration using deep neural networks. In IEEE intelligent vehicles symposium (IV) (pp. 1803\u20131810).","DOI":"10.1109\/IVS.2017.7995968"},{"key":"1784_CR140","doi-asserted-by":"publisher","unstructured":"Sheeny, M., Pellegrin, E. D., Mukherjee, S., Ahrabian, A., Wang, S., & Wallace, A. M. (2021). RADIATE: A radar dataset for automotive perception. In IEEE international conference on robotics and automation (ICRA), Xi\u2019an, China, May 30\u2013June 5, 2021 (pp. 1\u20137). IEEE. https:\/\/doi.org\/10.1109\/ICRA48506.2021.9562089","DOI":"10.1109\/ICRA48506.2021.9562089"},{"key":"1784_CR141","doi-asserted-by":"crossref","unstructured":"Shi, W., & Rajkumar, R. (2020). Point-GNN: Graph neural network for 3d object detection in a point cloud. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 1711\u20131719).","DOI":"10.1109\/CVPR42600.2020.00178"},{"key":"1784_CR142","doi-asserted-by":"crossref","unstructured":"Shi, S., Guo, C., Jiang, L., Wang, Z., Shi, J., Wang, X., & Li, H. (2020a). PV-RCNN: Point-voxel feature set abstraction for 3d object detection. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 10526\u201310535).","DOI":"10.1109\/CVPR42600.2020.01054"},{"key":"1784_CR143","doi-asserted-by":"crossref","unstructured":"Shi, S., Wang, X., & Li, H. (2019). PointRCNN: 3d object proposal generation and detection from point cloud. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 770\u2013779).","DOI":"10.1109\/CVPR.2019.00086"},{"key":"1784_CR144","doi-asserted-by":"crossref","unstructured":"Shi, S., Wang, Z., Shi, J., Wang, X., & Li, H. (2020b). From points to parts: 3d object detection from point cloud with part-aware and part-aggregation network. IEEE Transactions on Pattern Recognition and Machine Intelligence (TPAMI), 43, 1\u20131.","DOI":"10.1109\/TPAMI.2020.2977026"},{"key":"1784_CR145","doi-asserted-by":"crossref","unstructured":"Shin, K., Kwon, Y. P., & Tomizuka, M. (2019). RoarNet: A robust 3d object detection based on region approximation refinement. In IEEE intelligent vehicles symposium (IV) (pp. 2510\u20132515).","DOI":"10.1109\/IVS.2019.8813895"},{"key":"1784_CR146","doi-asserted-by":"crossref","unstructured":"Silver, D., Huang, A., Maddison, C., Guez, A., Sifre, L., Driessche, G., Schrittwieser, J., Antonoglou, I., Panneershelvam, V., Lanctot, M., & Dieleman, S. (2016). Mastering the game of go with deep neural networks and tree search. Nature, 529, 484\u2013489.","DOI":"10.1038\/nature16961"},{"key":"1784_CR147","unstructured":"Simonyan, K., & Zisserman, A. (2014). Very deep convolutional networks for large-scale image recognition. In: 3rd International Conference on Learning Representations (ICLR) San Diego, CA, USA, May 7\u20139, 2015, Conference Track Proceedings. arXiv:1409.1556"},{"key":"1784_CR148","doi-asserted-by":"crossref","unstructured":"Sindagi, V. A., Zhou, Y., & Tuzel, O. (2019). MVX-Net: Multimodal voxelnet for 3d object detection. In IEEE international conference on robotics and automation (ICRA) (pp. 7276\u20137282).","DOI":"10.1109\/ICRA.2019.8794195"},{"key":"1784_CR149","doi-asserted-by":"crossref","unstructured":"Strecha, C., von Hansen, W., Van\u00a0Gool, L., Fua, P., & Thoennessen, U. (2008). On benchmarking camera calibration and multi-view stereo for high resolution imagery. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 1\u20138).","DOI":"10.1109\/CVPR.2008.4587706"},{"key":"1784_CR150","doi-asserted-by":"publisher","unstructured":"Sun, P., Kretzschmar, H., Dotiwalla, X., Chouard, A., Patnaik, V., Tsui, P., Guo J, Zhou, Y., Chai, Y., Caine, B., Vasudevan, V., Han, W., Ngiam, J., Zhao, H., Timofeev, A., Ettinger, S., Krivokon, M., Gao, A., Joshi, A., Zhang, Y., Shlens J, Chen, Z., & Anguelov, D. (2020a). Scalability in perception for autonomous driving: Waymo open dataset. In 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), Seattle, WA, USA, June 13\u201319, 2020 (pp. 2443\u20132451). Computer Vision Foundation\/IEEE. https:\/\/doi.org\/10.1109\/CVPR42600.2020.00252","DOI":"10.1109\/CVPR42600.2020.00252"},{"key":"1784_CR151","unstructured":"Sun, Y., Zuo, W., Yun, P., Wang, H., & Liu, M. (2020b). FuseSeg: Semantic segmentation of urban scenes based on RGB and thermal data fusion. IEEE Transactions on Automation Science and Engineering, P.P.(99), 1\u201312."},{"key":"1784_CR152","doi-asserted-by":"crossref","unstructured":"Tang, H., Liu, Z., Zhao, S., Lin, Y., Lin, J., Wang, H., & Han, S. (2020). Searching efficient 3d architectures with sparse point-voxel convolution. In European conference on computer vision (ECCV) (pp. 685\u2013702).","DOI":"10.1007\/978-3-030-58604-1_41"},{"key":"1784_CR153","doi-asserted-by":"crossref","unstructured":"Urmson, C., Anhalt, J., Bagnell, D., Baker, C., Bittner, R., Clark, M., Dolan, J., Duggins, D., Galatali, T., Geyer, C. & Gittleman, M. (2008). Autonomous driving in urban environments: Boss and the urban challenge. Journal of Field Robotics, 25(8), 425\u2013466.","DOI":"10.1002\/rob.20255"},{"key":"1784_CR154","doi-asserted-by":"crossref","unstructured":"Urmson, C., Baker, C., Dolan, J., Rybski, P., Salesky, B., Whittaker, W. R., Ferguson, D., & Darms, M. (2009). Autonomous driving in traffic: Boss and the urban challenge. AI Magazine, 30(2), 17\u201328.","DOI":"10.1609\/aimag.v30i2.2238"},{"key":"1784_CR155","first-page":"6000","volume":"30","author":"A Vaswani","year":"2017","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A. N., Kaiser, Lu., & Polosukhin, I. (2017). Attention is all you need. Advances in Neural Information Processing Systems (NeurIPS), 30, 6000\u20136010.","journal-title":"Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"1784_CR156","doi-asserted-by":"crossref","unstructured":"Vora, S., Lang, A. H., Helou, B., & Beijbom, O. (2020). PointPainting: Sequential fusion for 3d object detection. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 4603\u20134611).","DOI":"10.1109\/CVPR42600.2020.00466"},{"issue":"7","key":"1784_CR157","doi-asserted-by":"publisher","first-page":"7064","DOI":"10.1109\/TVT.2020.2989148","volume":"69","author":"AM Wallace","year":"2020","unstructured":"Wallace, A. M., Halimi, A., & Buller, G. S. (2020). Full waveform lidar for adverse weather conditions. IEEE Transactions on Vehicular Technology (TVT), 69(7), 7064\u20137077.","journal-title":"IEEE Transactions on Vehicular Technology (TVT)"},{"key":"1784_CR158","doi-asserted-by":"crossref","unstructured":"Wandinger, U. (2005). Introduction to lidar. Brooks\/Cole Pub. Co.","DOI":"10.1007\/0-387-25101-4_1"},{"key":"1784_CR159","doi-asserted-by":"crossref","unstructured":"Wang, Z., & Jia, K. (2019a). Frustum ConvNet: Sliding frustums to aggregate local point-wise features for amodal. In IEEE international conference on intelligent robots and systems (IROS) (pp. 1742\u20131749).","DOI":"10.1109\/IROS40897.2019.8968513"},{"key":"1784_CR160","doi-asserted-by":"crossref","unstructured":"Wang, Z., & Jia, K. (2019b). Frustum ConvNet: Sliding frustums to aggregate local point-wise features for amodal 3d object detection. In IEEE international conference on intelligent robots and systems (IROS) (pp. 1742\u20131749).","DOI":"10.1109\/IROS40897.2019.8968513"},{"key":"1784_CR161","doi-asserted-by":"crossref","unstructured":"Wang, Y., Chao, W. L., Garg, D., Hariharan, B., Campbell, M., & Weinberger, K. Q. (2019). Pseudo-lidar from visual depth estimation: Bridging the gap in 3d object detection for autonomous driving. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 8437\u20138445).","DOI":"10.1109\/CVPR.2019.00864"},{"key":"1784_CR162","doi-asserted-by":"crossref","unstructured":"Wang, X., Girshick, R. B., Gupta, A., & He, K. (2018). Non-local neural networks. In IEEE conference on computer vision and pattern recognition (CVPR), Computer Vision Foundation\/IEEE Computer Society (pp. 7794\u20137803).","DOI":"10.1109\/CVPR.2018.00813"},{"key":"1784_CR163","doi-asserted-by":"crossref","unstructured":"Wang, C., Ma, C., Zhu, M., & Yang, X. (2021). PointAugmenting: Cross-modal augmentation for 3d object detection. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 11794\u201311803).","DOI":"10.1109\/CVPR46437.2021.01162"},{"key":"1784_CR164","doi-asserted-by":"crossref","unstructured":"Wang, S., Suo, S., Ma, W., Pokrovsky, A., & Urtasun, R. (2018). Deep parametric continuous convolutional neural networks. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 2589\u20132597).","DOI":"10.1109\/CVPR.2018.00274"},{"key":"1784_CR165","unstructured":"Wang, G., Tian, B., Zhang, Y., Chen, L., Cao, D., & Wu, J. (2020). Multi-view adaptive fusion network for 3D object detection. arXiv e-prints p arXiv:2011.00652"},{"key":"1784_CR166","doi-asserted-by":"publisher","first-page":"135","DOI":"10.1016\/j.neucom.2018.05.083","volume":"312","author":"M Wang","year":"2018","unstructured":"Wang, M., & Deng, W. (2018). Deep visual domain adaptation: A survey. Neurocomputing, 312, 135\u2013153.","journal-title":"Neurocomputing"},{"issue":"4","key":"1784_CR167","doi-asserted-by":"publisher","first-page":"1341","DOI":"10.1109\/TITS.2018.2849505","volume":"20","author":"J Wang","year":"2019","unstructured":"Wang, J., & Zhou, L. (2019). Traffic light recognition with high dynamic range imaging and deep learning. IEEE Transactions on Intelligent Transportation Systems, 20(4), 1341\u20131352.","journal-title":"IEEE Transactions on Intelligent Transportation Systems"},{"key":"1784_CR168","unstructured":"Weng, X., Man, Y., Cheng, D., Park, J., O.\u2019Toole, M., & Kitani, K. (2020). All-in-one drive: A large-scale comprehensive perception dataset with high-density long-range point clouds. arXiv"},{"key":"1784_CR169","unstructured":"Wilson, B., Qi, W., Agarwal, T., Lambert, J., Singh, J., Khandelwal, S., Pan, B., Kumar, R., Hartnett, A., Pontes, J. K., Ramanan, D., Carr, P., & Hays, J. (2021). Argoverse 2: Next generation datasets for self-driving perception and forecasting. In Proceedings of the neural information processing systems track on datasets and benchmarks (NeurIPS Datasets and Benchmarks 2021)."},{"key":"1784_CR170","doi-asserted-by":"crossref","unstructured":"Wu, X., Peng, L., Yang, H., Xie, L., Huang, C., Deng, C., Liu, H., & Cai, D. (2022). Sparse fuse dense: Towards high quality 3d detection with depth completion. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 5418\u20135427).","DOI":"10.1109\/CVPR52688.2022.00534"},{"key":"1784_CR171","doi-asserted-by":"crossref","unstructured":"Xie, J., Kiefel, M., Sun, M. T., & Geiger, A. (2016). Semantic instance annotation of street scenes by 3d to 2d label transfer. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 3688\u20133697).","DOI":"10.1109\/CVPR.2016.401"},{"key":"1784_CR172","first-page":"12460","volume":"34","author":"L Xie","year":"2020","unstructured":"Xie, L., Xiang, C., Yu, Z., Xu, G., Yang, Z., Cai, D., & He, X. (2020). PI-RCNN: An efficient multi-sensor 3d object detector with point-based attentive cont-conv fusion module. Association for the Advancement of Artificial Intelligence (AAAI), 34, 12460\u201312467.","journal-title":"Association for the Advancement of Artificial Intelligence (AAAI)"},{"key":"1784_CR173","doi-asserted-by":"crossref","unstructured":"Xu, D., Anguelov, D., & Jain, A. (2018). PointFusion: Deep sensor fusion for 3d bounding box estimation. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 244\u2013253).","DOI":"10.1109\/CVPR.2018.00033"},{"key":"1784_CR174","doi-asserted-by":"crossref","unstructured":"Xu, Q., Zhong, Y., & Neumann, U. (2021). Behind the curtain: Learning occluded shapes for 3d object detection. In Thirty-Sixth AAAI Conference on Artificial Intelligence, AAAI Thirty-Fourth Conference on InnovativeApplications of Artificial Intelligence (IAAI), The Twelveth Symposium on Educational Advances in Artificial Intelligence (EAAI) 2022 Virtual Event, February 22\u2013March 1, 2022 (pp. 2893\u20132901). AAAI Press.","DOI":"10.1609\/aaai.v36i3.20194"},{"key":"1784_CR175","unstructured":"Yang, Z., Chen, J., Miao, Z., Li, W., Zhu, X., & Zhang, L. (2022b). DeepInteraction: 3d object detection via modality interaction. arXiv preprint arXiv:2208.11112"},{"key":"1784_CR176","doi-asserted-by":"crossref","unstructured":"Yang, W., Li, Q., Liu, W., Yu, Y., Ma, Y., He, S., & Pan, J. (2021). Projecting your view attentively: Monocular road scene layout estimation via cross-view transformation. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 15536\u201315545).","DOI":"10.1109\/CVPR46437.2021.01528"},{"key":"1784_CR177","doi-asserted-by":"publisher","unstructured":"Yang, H., Liu, Z., Wu, X., Wang, W., Qian, W., He, X., & Cai, D. (2022a). Graph R-CNN: Towards accurate 3d object detection with semantic-decorated local graph. In Computer Vision - ECCV 2022 - 17th European Conference, Tel Aviv, Israel, October 23\u201327, 2022, Proceedings, Part VIII. Lecture Notes in Computer Science (vol. 13668, pp. 662\u2013679). Springer. https:\/\/doi.org\/10.1007\/978-3-031-20074-8_38","DOI":"10.1007\/978-3-031-20074-8_38"},{"key":"1784_CR178","doi-asserted-by":"crossref","unstructured":"Yang, B., Luo, W., & Urtasun, R. (2018a). PIXOR: Real-time 3d object detection from point clouds. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 7652\u20137660).","DOI":"10.1109\/CVPR.2018.00798"},{"key":"1784_CR179","doi-asserted-by":"crossref","unstructured":"Yang, B., Luo, W., & Urtasun, R. (2018b). PIXOR: Real-time 3d object detection from point clouds. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 7652\u20137660).","DOI":"10.1109\/CVPR.2018.00798"},{"key":"1784_CR180","doi-asserted-by":"crossref","unstructured":"Yang, Z., Sun, Y., Liu, S., & Jia, J. (2020). 3DSSD: Point-based 3d single stage object detector. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 11037\u201311045).","DOI":"10.1109\/CVPR42600.2020.01105"},{"key":"1784_CR181","doi-asserted-by":"crossref","unstructured":"Yang, Z., Sun, Y., Liu, S., Shen, X., & Jia, J. (2018). IPOD: Intensive point-based object detector for point cloud. CoRR. arXiv:1812.05276","DOI":"10.1109\/ICCV.2019.00204"},{"key":"1784_CR182","doi-asserted-by":"crossref","unstructured":"Yang, Z., Sun, Y., Liu, S., Shen, X., & Jia, J. (2019). STD: Sparse-to-dense 3d object detector for point cloud. In IEEE international conference on computer vision (ICCV) (pp. 1951\u20131960).","DOI":"10.1109\/ICCV.2019.00204"},{"key":"1784_CR183","first-page":"496","volume":"12363","author":"B Yang","year":"2020","unstructured":"Yang, B., Guo, R., Liang, M., Casas, S., & Urtasun, R. (2020). RadarNet: Exploiting radar for robust perception of dynamic objects. European Conference on Computer Vision (ECCV), 12363, 496\u2013512.","journal-title":"European Conference on Computer Vision (ECCV)"},{"issue":"10","key":"1784_CR184","doi-asserted-by":"publisher","first-page":"3337","DOI":"10.3390\/s18103337","volume":"18","author":"Y Yan","year":"2018","unstructured":"Yan, Y., Mao, Y., & Li, B. (2018). SECOND: Sparsely embedded convolutional detection. Sensors, 18(10), 3337.","journal-title":"Sensors"},{"key":"1784_CR185","doi-asserted-by":"crossref","unstructured":"Yin, T., Zhou, X., & Kr\u00e4henb\u00fchl, P. (2021). Center-based 3d object detection and tracking. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 11784\u201311793).","DOI":"10.1109\/CVPR46437.2021.01161"},{"key":"1784_CR186","doi-asserted-by":"crossref","unstructured":"Yoo, J., Ahn, N., & Sohn, K. (2020a). Rethinking data augmentation for image super-resolution: A comprehensive analysis and a new strategy. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 8372\u20138381).","DOI":"10.1109\/CVPR42600.2020.00840"},{"key":"1784_CR187","doi-asserted-by":"crossref","unstructured":"Yoo, J. H., Kim, Y., Kim, J., & Choi, J. W. (2020b). 3D-CVF: Generating joint camera and lidar features using cross-view spatial feature fusion for 3d object detection. In European conference on computer vision (ECCV) (pp. 720\u2013736).","DOI":"10.1007\/978-3-030-58583-9_43"},{"key":"1784_CR188","unstructured":"Yosinski, J., Clune, J., Bengio, Y., & Lipson, H. (2014). How transferable are features in deep neural networks? In Advances in neural information processing systems (NeurIPS) (vol. 27)."},{"key":"1784_CR189","unstructured":"You, Y., Wang, Y., Chao, W., Garg, D., Pleiss, G., Hariharan, B., Campbell, M. E., & Weinberger, K. Q. (2020). Pseudo-lidar++: Accurate depth for 3d object detection in autonomous driving. In 8th International Conference on Learning Representations (ICLR), Addis Ababa, Ethiopia, April 26\u201330, 2020. OpenReview.net"},{"key":"1784_CR190","doi-asserted-by":"crossref","unstructured":"Zewge, N. S., Kim, Y., Kim, J., & Kim, J. H. (2019). Millimeter-wave radar and RGB-D camera sensor fusion for real-time people detection and tracking. In 2019 7th international conference on robot intelligence technology and applications (RiTA) (pp. 93\u201398).","DOI":"10.1109\/RITAPP.2019.8932892"},{"key":"1784_CR191","unstructured":"Zhang, Y., Carballo, A., Yang, H., & Takeda, K. (2021b). Autonomous driving in adverse weather conditions: A survey. arXiv preprint arXiv:2112.08936"},{"key":"1784_CR192","unstructured":"Zhang, Y., Carballo, A., Yang, H., & Takeda, K. (2021c). Autonomous driving in adverse weather conditions: A survey. CoRR abs arXiv:2112.08936"},{"key":"1784_CR193","unstructured":"Zhang, W., Wang, Z., & Loy, C. C. (2020a). Multi-modality cut and paste for 3d object detection. arXiv:2012.12741"},{"key":"1784_CR194","doi-asserted-by":"publisher","unstructured":"Zhang, H., Yang, D., Yurtsever, E., Redmill, K. A., & \u00d6zg\u00fcner, \u00dc. (2021a). Faraway-Frustum: Dealing with lidar sparsity for 3d object detection using fusion. In 24th IEEE international intelligent Transportation tystems conference (ITSC), Indianapolis, IN, USA, September 19\u201322, 2021 (pp. 2646\u20132652). IEEE. https:\/\/doi.org\/10.1109\/ITSC48978.2021.9564990","DOI":"10.1109\/ITSC48978.2021.9564990"},{"issue":"9","key":"1784_CR195","first-page":"1781","volume":"57","author":"Y Zhang","year":"2020","unstructured":"Zhang, Y., Zhang, S., Zhang, Y., Ji, J., Duan, Y., Huang, Y., Peng, J., & Zhang, Y. (2020). Multi-modality fusion perception and computing in autonomous driving. Journal of Computer Research and Development, 57(9), 1781.","journal-title":"Journal of Computer Research and Development"},{"key":"1784_CR196","doi-asserted-by":"crossref","unstructured":"Zhao, X., Liu, Z., Hu, R., & Huang, K. (2019). 3d object detection using scale invariant and feature reweighting networks. In Association for the advancement of artificial intelligence (AAAI) (pp. 9267\u20139274).","DOI":"10.1609\/aaai.v33i01.33019267"},{"key":"1784_CR197","doi-asserted-by":"crossref","unstructured":"Zhou, B., & Kr\u00e4henb\u00fchl, P. (2022). Cross-view transformers for real-time map-view semantic segmentation. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 13760\u201313769).","DOI":"10.1109\/CVPR52688.2022.01339"},{"key":"1784_CR198","doi-asserted-by":"crossref","unstructured":"Zhou, Y., & Tuzel, O. (2018). VoxelNet: End-to-end learning for point cloud based 3d object detection. In IEEE conference on computer vision and pattern recognition (CVPR) (pp. 4490\u20134499).","DOI":"10.1109\/CVPR.2018.00472"},{"key":"1784_CR199","doi-asserted-by":"crossref","unstructured":"Zhou, Y., Wan, G., Hou, S., Yu, L., Wang, G., Rui, X., & Song, S. (2020). DA4AD: End-to-end deep attention-based visual localization for autonomous driving. In European conference on computer vision (ECCV) (pp. 271\u2013289).","DOI":"10.1007\/978-3-030-58604-1_17"},{"key":"1784_CR200","doi-asserted-by":"publisher","unstructured":"Zhu, H., Deng, J., Zhang, Y., Ji, J., Mao, Q., Li, H., & Zhang, Y. (2022). VPFNet: Improving 3d object detection with virtual point based lidar and stereo data fusion. IEEE Transactions on Multimedia (TMM). https:\/\/doi.org\/10.1109\/TMM.2022.3189778","DOI":"10.1109\/TMM.2022.3189778"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-023-01784-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-023-01784-z\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-023-01784-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,20]],"date-time":"2024-10-20T15:34:50Z","timestamp":1729438490000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-023-01784-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,5,17]]},"references-count":200,"journal-issue":{"issue":"8","published-print":{"date-parts":[[2023,8]]}},"alternative-id":["1784"],"URL":"https:\/\/doi.org\/10.1007\/s11263-023-01784-z","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,5,17]]},"assertion":[{"value":"2 March 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 March 2023","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 May 2023","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}