{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,17]],"date-time":"2026-02-17T16:07:12Z","timestamp":1771344432632,"version":"3.50.1"},"reference-count":66,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2026,1,20]],"date-time":"2026-01-20T00:00:00Z","timestamp":1768867200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,20]],"date-time":"2026-01-20T00:00:00Z","timestamp":1768867200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2026,2]]},"DOI":"10.1007\/s11263-025-02704-z","type":"journal-article","created":{"date-parts":[[2026,1,20]],"date-time":"2026-01-20T03:51:41Z","timestamp":1768881101000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["CAT++: Enhancing 3D Annotations with Hierarchical-Interleaved Encoding and Attention-Conditioned Implicit Representation"],"prefix":"10.1007","volume":"134","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-4426-0667","authenticated-orcid":false,"given":"Xiaoyan","family":"Qian","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chang","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaojuan","family":"Qi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Siewchong","family":"Tan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Edmund Y.","family":"Lam","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ngai","family":"Wong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,1,20]]},"reference":[{"key":"2704_CR1","doi-asserted-by":"crossref","unstructured":"Caesar, H., Bankiti, V., Lang, A.H., Vora, S., Liong, V.E., Xu, Q., Krishnan, A., Pan, Y., Baldan, G., & Beijbom, O. (2020). nuscenes: A multimodal dataset for autonomous driving. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11621\u201311631.","DOI":"10.1109\/CVPR42600.2020.01164"},{"key":"2704_CR2","doi-asserted-by":"crossref","unstructured":"Chen, X., Ma, H., Wan, J., Li, B., & Xia, T. (2017). Multi-view 3d object detection network for autonomous driving. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1907\u20131915.","DOI":"10.1109\/CVPR.2017.691"},{"key":"2704_CR3","doi-asserted-by":"crossref","unstructured":"Dai, A., Chang, A.X., Savva, M., Halber, M., Funkhouser, T., & Niessner, M. (2017). Scannet: Richly-annotated 3d reconstructions of indoor scenes. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR.2017.261"},{"key":"2704_CR4","unstructured":"De\u00a0Vries, H., Strub, F., Mary, J., Larochelle, H., Pietquin, O., & Courville, A.C. (2017). Modulating early visual processing by language. Advances in Neural Information Processing Systems 30."},{"key":"2704_CR5","unstructured":"Dumoulin, V., Belghazi, I., Poole, B., Mastropietro, O., Lamb, A., Arjovsky, M., & Courville, A. (2016). Adversarially learned inference. arXiv preprint arXiv:1606.00704."},{"key":"2704_CR6","doi-asserted-by":"publisher","first-page":"303","DOI":"10.1007\/s11263-009-0275-4","volume":"88","author":"M Everingham","year":"2010","unstructured":"Everingham, M., Van Gool, L., Williams, C. K., Winn, J., & Zisserman, A. (2010). The pascal visual object classes (voc) challenge. International journal of computer vision,88, 303\u2013338.","journal-title":"International journal of computer vision"},{"key":"2704_CR7","doi-asserted-by":"crossref","unstructured":"Geiger, A., Lenz, P., & Urtasun, R. (2012). Are we ready for autonomous driving? the kitti vision benchmark suite. In: Conference on Computer Vision and Pattern Recognition (CVPR).","DOI":"10.1109\/CVPR.2012.6248074"},{"issue":"2","key":"2704_CR8","doi-asserted-by":"publisher","first-page":"187","DOI":"10.1007\/s41095-021-0229-5","volume":"7","author":"M-H Guo","year":"2021","unstructured":"Guo, M.-H., Cai, J.-X., Liu, Z.-N., Mu, T.-J., Martin, R. R., & Hu, S.-M. (2021). Pct: Point cloud transformer. Computational Visual Media,7(2), 187\u2013199.","journal-title":"Computational Visual Media"},{"key":"2704_CR9","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2016). Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778.","DOI":"10.1109\/CVPR.2016.90"},{"key":"2704_CR10","doi-asserted-by":"crossref","unstructured":"Hou, Z., Yu, B., & Tao, D. (2022). Batchformer: Learning to explore sample relationships for robust representation learning. In: CVPR.","DOI":"10.1109\/CVPR52688.2022.00711"},{"key":"2704_CR11","doi-asserted-by":"crossref","unstructured":"Hou, Z., Yu, B., Wang, C., Zhan, Y., & Tao, D. (2022). Batchformerv2: Exploring sample relationships for dense representation learning. arXiv preprint arXiv:2204.01254.","DOI":"10.1109\/CVPR52688.2022.00711"},{"issue":"10","key":"2704_CR12","doi-asserted-by":"publisher","first-page":"2702","DOI":"10.1109\/TPAMI.2019.2926463","volume":"42","author":"X Huang","year":"2019","unstructured":"Huang, X., Wang, P., Cheng, X., Zhou, D., Geng, Q., & Yang, R. (2019). The apolloscape open dataset for autonomous driving and its application. IEEE Transactions on Pattern Analysis and Machine Intelligence,42(10), 2702\u20132719.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2704_CR13","doi-asserted-by":"crossref","unstructured":"Kolodiazhnyi, M., Vorontsova, A., Skripkin, M., Rukhovich, D., & Konushin, A. (2024). Unidet3d: Multi-dataset indoor 3d object detection. arXiv preprint arXiv:2409.04234.","DOI":"10.1609\/aaai.v39i4.32459"},{"key":"2704_CR14","doi-asserted-by":"crossref","unstructured":"Ku, J., Mozifian, M., Lee, J., Harakeh, A., & Waslander, S.L. (2018). Joint 3d proposal generation and object detection from view aggregation. In: 2018 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS), pp. 1\u20138. IEEE.","DOI":"10.1109\/IROS.2018.8594049"},{"key":"2704_CR15","doi-asserted-by":"crossref","unstructured":"Lai, X., Liu, J., Jiang, L., Wang, L., Zhao, H., Liu, S., Qi, X., & Jia, J. (2022). Stratified transformer for 3d point cloud segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8500\u20138509.","DOI":"10.1109\/CVPR52688.2022.00831"},{"key":"2704_CR16","doi-asserted-by":"crossref","unstructured":"Lang, A.H., Vora, S., Caesar, H., Zhou, L., Yang, J., & Beijbom, O. (2019). Pointpillars: Fast encoders for object detection from point clouds. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12697\u201312705.","DOI":"10.1109\/CVPR.2019.01298"},{"key":"2704_CR17","doi-asserted-by":"crossref","unstructured":"Liu, C., Qian, X., Huang, B., Qi, X., Lam, E., Tan, S.-C., & Wong, N. (2022). Multimodal transformer for automatic 3d annotation and object detection. In: Computer Vision\u2013ECCV 2022: 17th European Conference, Tel Aviv, Israel, October 23\u201327, 2022, Proceedings, Part XXXVIII, pp. 657\u2013673. Springer.","DOI":"10.1007\/978-3-031-19839-7_38"},{"key":"2704_CR18","doi-asserted-by":"crossref","unstructured":"Liu, C., Qian, X., Qi, X., Lam, E.Y., Tan, S.-C., & Wong, N. (2022). Map-gen: An automated 3d-box annotation flow with multimodal attention point generator. In: 2022 26th International Conference on Pattern Recognition (ICPR), pp. 1148\u20131155. IEEE","DOI":"10.1109\/ICPR56361.2022.9956415"},{"key":"2704_CR19","doi-asserted-by":"crossref","unstructured":"Liu, Z., Tang, H., Amini, A., Yang, X., Mao, H., Rus, D., & Han, S. (2022). Bevfusion: Multi-task multi-sensor fusion with unified bird\u2019s-eye view representation. arXiv preprint arXiv:2205.13542.","DOI":"10.1109\/ICRA48891.2023.10160968"},{"key":"2704_CR20","doi-asserted-by":"crossref","unstructured":"Liu, Z., Zhao, X., Huang, T., Hu, R., Zhou, Y., & Bai, X. (2020). Tanet: Robust 3d object detection from point clouds with triple attention. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 34, pp. 11677\u201311684.","DOI":"10.1609\/aaai.v34i07.6837"},{"key":"2704_CR21","doi-asserted-by":"publisher","unstructured":"Maturana, D., & Scherer, S. (2015). Voxnet: A 3d convolutional neural network for real-time object recognition. In: 2015 IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS), pp. 922\u2013928 https:\/\/doi.org\/10.1109\/IROS.2015.7353481.","DOI":"10.1109\/IROS.2015.7353481"},{"key":"2704_CR22","doi-asserted-by":"crossref","unstructured":"McCraith, R., Insafutdinov, E., Neumann, L., & Vedaldi, A. (2021). Lifting 2d object locations to 3d by discounting lidar outliers across objects and views. arXiv preprint arXiv:2109.07945.","DOI":"10.1109\/ICRA46639.2022.9811693"},{"key":"2704_CR23","doi-asserted-by":"crossref","unstructured":"Meng, Q., Wang, W., Zhou, T., Shen, J., Jia, Y., & Van\u00a0Gool, L. (2021). Towards a weakly supervised framework for 3d point cloud object detection and annotation. IEEE Transactions on Pattern Analysis and Machine Intelligence.","DOI":"10.1109\/TPAMI.2021.3063611"},{"key":"2704_CR24","doi-asserted-by":"crossref","unstructured":"Meng, Q., Wang, W., Zhou, T., Shen, J., Jia, Y., & Van\u00a0Gool, L. (2021). Towards a weakly supervised framework for 3d point cloud object detection and annotation. IEEE Transactions on Pattern Analysis and Machine Intelligence.","DOI":"10.1109\/TPAMI.2021.3063611"},{"key":"2704_CR25","doi-asserted-by":"crossref","unstructured":"Meng, Q., Wang, W., Zhou, T., Shen, J., Van\u00a0Gool, L., & Dai, D. (2020). Weakly supervised 3d object detection from lidar point cloud. In: European Conference on Computer Vision, pp. 515\u2013531. Springer.","DOI":"10.1007\/978-3-030-58601-0_31"},{"key":"2704_CR26","doi-asserted-by":"crossref","unstructured":"Mescheder, L., Oechsle, M., Niemeyer, M., Nowozin, S., & Geiger, A. (2019). Occupancy networks: Learning 3d reconstruction in function space. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4460\u20134470.","DOI":"10.1109\/CVPR.2019.00459"},{"key":"2704_CR27","doi-asserted-by":"crossref","unstructured":"Misra, I., Girdhar, R., Joulin, A. (2021). An end-to-end transformer model for 3d object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2906\u20132917.","DOI":"10.1109\/ICCV48922.2021.00290"},{"key":"2704_CR28","doi-asserted-by":"crossref","unstructured":"Niemeyer, M., Mescheder, L., Oechsle, M., & Geiger, A. (2020). Differentiable volumetric rendering: Learning implicit 3d representations without 3d supervision. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3504\u20133515.","DOI":"10.1109\/CVPR42600.2020.00356"},{"key":"2704_CR29","doi-asserted-by":"crossref","unstructured":"Paat, H., Lian, Q., Yao, W., & Zhang, T. (2024). Medl-u: Uncertainty-aware 3d automatic annotation based on evidential deep learning. In: 2024 IEEE International Conference on Robotics and Automation (ICRA), pp. 13976\u201313982. IEEE.","DOI":"10.1109\/ICRA57147.2024.10610597"},{"key":"2704_CR30","doi-asserted-by":"crossref","unstructured":"Pan, X., Xia, Z., Song, S., Li, L.E., & Huang, G. (2021). 3d object detection with pointformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 7463\u20137472.","DOI":"10.1109\/CVPR46437.2021.00738"},{"key":"2704_CR31","doi-asserted-by":"crossref","unstructured":"Park, C., Jeong, Y., Cho, M., & Park, J. (2022). Fast point transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16949\u201316958.","DOI":"10.1109\/CVPR52688.2022.01644"},{"key":"2704_CR32","unstructured":"Paszke, A., Gross, S., Massa, F., Lerer, A., Bradbury, J., Chanan, G., Killeen, T., Lin, Z., Gimelshein, N., Antiga, L., & Chintala, S. (2019). Pytorch: An imperative style, high-performance deep learning library. Advances in neural information processing systems 32."},{"key":"2704_CR33","doi-asserted-by":"crossref","unstructured":"Peng, S., Niemeyer, M., Mescheder, L., Pollefeys, M., & Geiger, A. (2020). Convolutional occupancy networks. In: European Conference on Computer Vision, pp. 523\u2013540. Springer.","DOI":"10.1007\/978-3-030-58580-8_31"},{"key":"2704_CR34","doi-asserted-by":"crossref","unstructured":"Qi, C.R., Liu, W., Wu, C., Su, H., & Guibas, L.J. (2018). Frustum pointnets for 3d object detection from rgb-d data. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 918\u2013927.","DOI":"10.1109\/CVPR.2018.00102"},{"key":"2704_CR35","unstructured":"Qi, C. R., Su, H., Mo, K., & Guibas, L.J. (2017). Pointnet: Deep learning on point sets for 3d classification and segmentation. (pp. 652\u2013660)."},{"key":"2704_CR36","unstructured":"Qi, C.R., Yi, L., Su, H., & Guibas, L.J. (2017). Pointnet++: Deep hierarchical feature learning on point sets in a metric space. arXiv preprint arXiv:1706.02413."},{"key":"2704_CR37","doi-asserted-by":"crossref","unstructured":"Qian, X., Liu, C., Qi, X., Tan, S.-C., Lam, E., & Wong, N. (2023). Context-Aware Transformer for 3D Point Cloud Automatic Annotation.","DOI":"10.1609\/aaai.v37i2.25301"},{"key":"2704_CR38","doi-asserted-by":"crossref","unstructured":"Qin, Z., Wang, J., & Lu, Y. (2020). Weakly supervised 3d object detection from point clouds. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 4144\u20134152.","DOI":"10.1145\/3394171.3413805"},{"key":"2704_CR39","unstructured":"Shen, Y., Geng, Z., Yuan, Y., Lin, Y., Liu, Z., Wang, C., Hu, H., Zheng, N., & Guo, B. (2023). V-detr: Detr with vertex relative position encoding for 3d object detection. arXiv preprint arXiv:2308.04409."},{"key":"2704_CR40","doi-asserted-by":"crossref","unstructured":"Shi, S., Guo, C., Jiang, L., Wang, Z., Shi, J., Wang, X., & Li, H. (2020). Pv-rcnn: Point-voxel feature set abstraction for 3d object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10529\u201310538.","DOI":"10.1109\/CVPR42600.2020.01054"},{"key":"2704_CR41","doi-asserted-by":"crossref","unstructured":"Shi, S., Wang, X., & Li, H. (2019). Pointrcnn: 3d object proposal generation and detection from point cloud. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 770\u2013779.","DOI":"10.1109\/CVPR.2019.00086"},{"key":"2704_CR42","doi-asserted-by":"crossref","unstructured":"Shi, S., Wang, Z., Shi, J., Wang, X., & Li, H. (2020). From points to parts: 3d object detection from point cloud with part-aware and part-aggregation network. IEEE Transactions on Pattern Analysis and Machine Intelligence.","DOI":"10.1109\/TPAMI.2020.2977026"},{"key":"2704_CR43","first-page":"7462","volume":"33","author":"V Sitzmann","year":"2020","unstructured":"Sitzmann, V., Martel, J., Bergman, A., Lindell, D., & Wetzstein, G. (2020). Implicit neural representations with periodic activation functions. Advances in Neural Information Processing Systems,33, 7462\u20137473.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2704_CR44","doi-asserted-by":"crossref","unstructured":"Song, S., Lichtenberg, S.P., & Xiao, J. (2015). Sun rgb-d:A rgb-d scene understanding benchmark suite. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 567\u2013576.","DOI":"10.1109\/CVPR.2015.7298655"},{"key":"2704_CR45","doi-asserted-by":"crossref","unstructured":"Sun, P., Kretzschmar, H., Dotiwalla, X., Chouard, A., Patnaik, V., Tsui, P., Guo, J., Zhou, Y., Chai, Y., Caine, B., & Anguelov, D. (2020). Scalability in perception for autonomous driving: Waymo open dataset. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2446\u20132454.","DOI":"10.1109\/CVPR42600.2020.00252"},{"key":"2704_CR46","unstructured":"Team, O.D. (2020). OpenPCDet: An Open-source Toolbox for 3D Object Detection from Point Clouds. https:\/\/github.com\/open-mmlab\/OpenPCDet."},{"key":"2704_CR47","unstructured":"Wang, C., Yang, W., Liu, X., & Zhang, T. (2025). State space model meets transformer: A new paradigm for 3d object detection. arXiv preprint arXiv:2503.14493."},{"key":"2704_CR48","doi-asserted-by":"crossref","unstructured":"Wei, Y., Su, S., Lu, J., & Zhou, J. (2021). Fgr: Frustum-aware geometric reasoning for weakly supervised 3d vehicle detection. In: 2021 IEEE International Conference on Robotics and Automation (ICRA), pp. 4348\u20134354. IEEE","DOI":"10.1109\/ICRA48506.2021.9561245"},{"key":"2704_CR49","unstructured":"Wilson, B., Kira, Z., & Hays, J. (2020). 3d for free: Crossmodal transfer learning using hd maps. arXiv preprint arXiv:2008.10592."},{"key":"2704_CR50","doi-asserted-by":"crossref","unstructured":"Wu, Z., Song, S., Khosla, A., Yu, F., Zhang, L., Tang, X., & Xiao, J. (2015). 3d shapenets: A deep representation for volumetric shapes. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1912\u20131920.","DOI":"10.1109\/CVPR.2015.7298801"},{"key":"2704_CR51","first-page":"33330","volume":"35","author":"X Wu","year":"2022","unstructured":"Wu, X., Lao, Y., Jiang, L., Liu, X., & Zhao, H. (2022). Point transformer v2: Grouped vector attention and partition-based pooling. Advances in Neural Information Processing Systems,35, 33330\u201333342.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2704_CR52","doi-asserted-by":"crossref","unstructured":"Xie, J., Zheng, Z., Gao, R., Wang, W., Zhu, S.-C., & Wu, Y.N. (2018). Learning descriptor networks for 3d shape synthesis and analysis. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 8629\u20138638.","DOI":"10.1109\/CVPR.2018.00900"},{"key":"2704_CR53","doi-asserted-by":"crossref","unstructured":"Xu, M., Ding, R., Zhao, H., & Qi, X. (2021). Paconv: Position adaptive convolution with dynamic kernel assembling on point clouds. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3173\u20133182.","DOI":"10.1109\/CVPR46437.2021.00319"},{"key":"2704_CR54","doi-asserted-by":"crossref","unstructured":"Yan, S., Yang, Z., Li, H., Guan, L., Kang, H., Hua, G., & Huang, Q. (2022). Implicit autoencoder for point cloud self-supervised representation learning. arXiv preprint arXiv:2201.00785.","DOI":"10.1109\/ICCV51070.2023.01336"},{"key":"2704_CR55","doi-asserted-by":"crossref","unstructured":"Yang, Y., Feng, C., Shen, Y., & Tian, D. (2018). Foldingnet: Point cloud auto-encoder via deep grid deformation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 206\u2013215.","DOI":"10.1109\/CVPR.2018.00029"},{"key":"2704_CR56","doi-asserted-by":"crossref","unstructured":"Yang, Z., Sun, Y., Liu, S., Shen, X., & Jia, J. (2019). Std: Sparse-to-dense 3d object detector for point cloud. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1951\u20131960.","DOI":"10.1109\/ICCV.2019.00204"},{"issue":"10","key":"2704_CR57","doi-asserted-by":"publisher","first-page":"3337","DOI":"10.3390\/s18103337","volume":"18","author":"Y Yan","year":"2018","unstructured":"Yan, Y., Mao, Y., & Li, B. (2018). Second: Sparsely embedded convolutional detection. Sensors,18(10), 3337.","journal-title":"Sensors"},{"key":"2704_CR58","first-page":"1","volume":"71","author":"T Ye","year":"2022","unstructured":"Ye, T., Yan, X., Wang, S., Li, Y., & Zhou, F. (2022). An efficient 3-d point cloud place recognition approach based on feature point extraction and transformer. IEEE Transactions on Instrumentation and Measurement,71, 1\u20139.","journal-title":"IEEE Transactions on Instrumentation and Measurement"},{"key":"2704_CR59","doi-asserted-by":"crossref","unstructured":"Yi, H., Shi, S., Ding, M., Sun, J., Xu, K., Zhou, H., Wang, Z., Li, S., & Wang, G. (2020). Segvoxelnet: Exploring semantic context and depth-aware features for 3d vehicle detection from point cloud. In: 2020 IEEE International Conference on Robotics and Automation (ICRA), pp. 2274\u20132280. IEEE.","DOI":"10.1109\/ICRA40945.2020.9196556"},{"key":"2704_CR60","doi-asserted-by":"crossref","unstructured":"Yu, X., Rao, Y., Wang, Z., Liu, Z., Lu, J., & Zhou, J. (2021). Pointr: Diverse point cloud completion with geometry-aware transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 12498\u201312507.","DOI":"10.1109\/ICCV48922.2021.01227"},{"key":"2704_CR61","doi-asserted-by":"crossref","unstructured":"Yu, X., Tang, L., Rao, Y., Huang, T., Zhou, J., & Lu, J. (2022). Point-bert: Pre-training 3d point cloud transformers with masked point modeling. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 19313\u201319322","DOI":"10.1109\/CVPR52688.2022.01871"},{"key":"2704_CR62","doi-asserted-by":"crossref","unstructured":"Zakharov, S., Kehl, W., Bhargava, A., & Gaidon, A. (2020). Autolabeling 3d objects with differentiable rendering of sdf shape priors. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12224\u201312233","DOI":"10.1109\/CVPR42600.2020.01224"},{"key":"2704_CR63","doi-asserted-by":"crossref","unstructured":"Zhang, W., & Xiao, C. (2019). Pcan: 3d attention map learning using contextual information for point cloud based retrieval. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12436\u201312445.","DOI":"10.1109\/CVPR.2019.01272"},{"key":"2704_CR64","doi-asserted-by":"crossref","unstructured":"Zhao, H., Jiang, L., Jia, J., Torr, P.H.S., & Koltun, V. (2021). Point transformer. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 16259\u201316268.","DOI":"10.1109\/ICCV48922.2021.01595"},{"key":"2704_CR65","doi-asserted-by":"crossref","unstructured":"Zheng, Z., Wang, P., Liu, W., Li, J., Ye, R., & Ren, D. (2020). Distance-iou loss: Faster and better learning for bounding box regression. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 34, pp. 12993\u201313000","DOI":"10.1609\/aaai.v34i07.6999"},{"key":"2704_CR66","doi-asserted-by":"crossref","unstructured":"Zhou, Y., & Tuzel, O. (2018). Voxelnet: End-to-end learning for point cloud based 3d object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4490\u20134499.","DOI":"10.1109\/CVPR.2018.00472"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-025-02704-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-025-02704-z","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-025-02704-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,17]],"date-time":"2026-02-17T15:21:07Z","timestamp":1771341667000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-025-02704-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,1,20]]},"references-count":66,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2026,2]]}},"alternative-id":["2704"],"URL":"https:\/\/doi.org\/10.1007\/s11263-025-02704-z","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,1,20]]},"assertion":[{"value":"22 October 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 September 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"20 January 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"71"}}