{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,28]],"date-time":"2026-03-28T06:25:55Z","timestamp":1774679155530,"version":"3.50.1"},"publisher-location":"Singapore","reference-count":40,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819698622","type":"print"},{"value":"9789819698639","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-96-9863-9_39","type":"book-chapter","created":{"date-parts":[[2025,7,23]],"date-time":"2025-07-23T14:38:54Z","timestamp":1753281534000},"page":"460-475","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Pose-Enhanced 3D Rotary Embedding for Multi-View 3D Object Detection"],"prefix":"10.1007","author":[{"given":"Ke","family":"Sheng","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Huiying","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xinzhong","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,7,24]]},"reference":[{"key":"39_CR1","doi-asserted-by":"publisher","unstructured":"Philion, J., Fidler, S.: Lift, splat, shoot: Encoding images from arbitrary camera rigs by implicitly unprojecting to 3d. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XIV 16, pp. 194\u2013210. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58568-6_12","DOI":"10.1007\/978-3-030-58568-6_12"},{"key":"39_CR2","unstructured":"Huang, J., Huang, G., Zhu, Z., Ye, Y., Du, D.: BEVDet: high-performance multi-camera 3d object detection in bird-eye-view. arXiv preprint arXiv: 2112.11790 (2021)"},{"key":"39_CR3","doi-asserted-by":"publisher","unstructured":"Li, Z., et al.: BEVFormer: learning Bird\u2019s-eye-view representation from multi-camera images via spatiotemporal transformers. In: European Conference on Computer Vision, pp. 1\u201318. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-20077-9_1","DOI":"10.1007\/978-3-031-20077-9_1"},{"key":"39_CR4","doi-asserted-by":"crossref","unstructured":"Wang, S., Liu, Y., Wang, T., Li, Y., Zhang, X.: Exploring object-centric temporal modeling for efficient multi-view 3D object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3621\u20133631 (2023)","DOI":"10.1109\/ICCV51070.2023.00335"},{"key":"39_CR5","doi-asserted-by":"publisher","unstructured":"Liu, Y., Wang, T., Zhang, X., Sun, J.: PETR: position embedding transformation for multi-view 3D object detection. In: European Conference on Computer Vision, pp. 531\u2013548. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19812-0_31","DOI":"10.1007\/978-3-031-19812-0_31"},{"key":"39_CR6","doi-asserted-by":"crossref","unstructured":"Li, Y., Bao, H., Ge, Z., Yang, J., Sun, J., Li, Z.: BEVStereo: enhancing depth estimation in multi-view 3D object detection with temporal stereo. In: Proceedings of the AAAI Conference on Artificial Intelligence. vol. 37, pp. 1486\u20131494 (2023)","DOI":"10.1609\/aaai.v37i2.25234"},{"key":"39_CR7","doi-asserted-by":"crossref","unstructured":"Liu, Z., et al.: BEVFusion: multi-task multi-sensor fusion with unified bird\u2019s-eye view representation. In: 2023 IEEE International Conference on Robotics and Automation (ICRA), pp. 2774\u20132781. IEEE (2023)","DOI":"10.1109\/ICRA48891.2023.10160968"},{"key":"39_CR8","doi-asserted-by":"crossref","unstructured":"Li, Y., et al.: BEVDepth: acquisition of reliable depth for multi-view 3d object detection. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 37, pp. 1477\u20131485 (2023)","DOI":"10.1609\/aaai.v37i2.25233"},{"key":"39_CR9","doi-asserted-by":"crossref","unstructured":"Yang, C., et al.: BEVFormer v2: adapting modern image backbones to bird\u2019s-eye-view recognition via perspective supervision. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 17830\u201317839 (2023)","DOI":"10.1109\/CVPR52729.2023.01710"},{"key":"39_CR10","unstructured":"Wang, Y., Guizilini, V.C., Zhang, T., Wang, Y., Zhao, H., Solomon, J.: DETR3D: 3D object detection from multi-view images via 3D-to-2D queries. In: Conference on Robot Learning, pp. 180\u2013191. PMLR (2022)"},{"key":"39_CR11","doi-asserted-by":"crossref","unstructured":"Liu, H., Teng, Y., Lu, T., Wang, H., Wang, L.: SparseBEV: high-performance sparse 3D object detection from multi-camera videos. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 18580\u201318590 (2023)","DOI":"10.1109\/ICCV51070.2023.01703"},{"key":"39_CR12","doi-asserted-by":"crossref","unstructured":"Shu, C., Deng, J., Yu, F., Liu, Y.: 3DPPE: 3D point positional encoding for transformer-based multi-camera 3D object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3580\u20133589 (2023)","DOI":"10.1109\/ICCV51070.2023.00331"},{"key":"39_CR13","doi-asserted-by":"crossref","unstructured":"Jiang, X., et al.: Far3D: expanding the horizon for surround-view 3D object detection. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 38, pp. 2561\u20132569 (2024)","DOI":"10.1609\/aaai.v38i3.28033"},{"key":"39_CR14","doi-asserted-by":"publisher","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., Zagoruyko, S.: End- to-end object detection with transformers. In: European Conference on Computer Vision. pp. 213\u2013229. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58452-8_13","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"39_CR15","doi-asserted-by":"crossref","unstructured":"Hou, J., et al.: OPEN: object-wise position embedding for multi-view 3D object detection. ArXiv preprint arXiv:2407.10753 (2024)","DOI":"10.1007\/978-3-031-73347-5_9"},{"key":"39_CR16","doi-asserted-by":"crossref","unstructured":"Chen, D., Li, J., Guizilini, V., Ambrus, R.A., Gaidon, A.: Viewpoint equivariance for multi-view 3D object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9213\u20139222 (2023)","DOI":"10.1109\/CVPR52729.2023.00889"},{"key":"39_CR17","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2023.127063","volume":"568","author":"J Su","year":"2024","unstructured":"Su, J., Ahmed, M., Lu, Y., Pan, S., Bo, W., Liu, Y.: RoFormer: enhanced transformer with rotary position embedding. Neurocomputing 568, 127063 (2024)","journal-title":"Neurocomputing"},{"key":"39_CR18","doi-asserted-by":"crossref","unstructured":"Caesar, H., et al: nuScenes: a multimodal dataset for autonomous driving. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11621\u201311631 (2020)","DOI":"10.1109\/CVPR42600.2020.01164"},{"key":"39_CR19","doi-asserted-by":"crossref","unstructured":"Wang, T., Zhu, X., Pang, J., Lin, D.: FCOS3D: fully convolutional one-stage monocular 3D object detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 913\u2013922 (2021","DOI":"10.1109\/ICCVW54120.2021.00107"},{"key":"39_CR20","doi-asserted-by":"crossref","unstructured":"Qi, Z., Wang, J., Wu, X., Zhao, H.: OCBEV: object-centric BEV transformer for multi-view 3D object detection. In: 2024 International Conference on 3D Vision (3DV), pp. 1188\u20131197. IEEE (2024)","DOI":"10.1109\/3DV62453.2024.00098"},{"key":"39_CR21","doi-asserted-by":"crossref","unstructured":"Jiang, Y., et al.: PolarFormer: multi-camera 3D object detection with polar transformer. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 37, pp. 1042\u20131050 (2023)","DOI":"10.1609\/aaai.v37i1.25185"},{"key":"39_CR22","doi-asserted-by":"crossref","unstructured":"Li, H., et al: DFA3D: 3D deformable attention for 2D-to-3D feature lifting. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6684\u20136693 (2023)","DOI":"10.1109\/ICCV51070.2023.00615"},{"key":"39_CR23","doi-asserted-by":"crossref","unstructured":"Wang, S., Jiang, X., Li, Y.: Focal-PETR: embracing foreground for efficient multi camera 3D object detection. IEEE Trans. Intell. Veh. (2023)","DOI":"10.1109\/TIV.2023.3332608"},{"key":"39_CR24","doi-asserted-by":"crossref","unstructured":"Yang, Z., Yu, Z., Choy, C., Wang, R., Anandkumar, A., Alvarez, J.M.: Improving distant 3D object detection using 2D box supervision. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14853\u201314863 (2024)","DOI":"10.1109\/CVPR52733.2024.01407"},{"key":"39_CR25","doi-asserted-by":"crossref","unstructured":"Wu, K., Peng, H., Chen, M., Fu, J., Chao, H.: Rethinking and improving relative position encoding for vision transformer. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 10033\u201310041 (2021)","DOI":"10.1109\/ICCV48922.2021.00988"},{"key":"39_CR26","doi-asserted-by":"crossref","unstructured":"Liu, Y., et al.: PETRv2: a unified framework for 3d perception from multi-camera images. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3262\u20133272 (2023)","DOI":"10.1109\/ICCV51070.2023.00302"},{"key":"39_CR27","doi-asserted-by":"crossref","unstructured":"Wang, Z., Huang, Z., Fu, J., Wang, N., Liu, S.: Object as query: lifting any 2D object detector to 3D detection. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3791\u20133800 (2023)","DOI":"10.1109\/ICCV51070.2023.00351"},{"key":"39_CR28","doi-asserted-by":"crossref","unstructured":"Xiong, K., et al.: CAPE: camera view position embedding for multi-view 3d object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 21570\u201321579 (2023)","DOI":"10.1109\/CVPR52729.2023.02066"},{"key":"39_CR29","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Advances in Neural Information Processing Systems, vol. 30 (2017)"},{"key":"39_CR30","doi-asserted-by":"crossref","unstructured":"Shaw, P., Uszkoreit, J., Vaswani, A.: Self-attention with relative position representations. arXiv preprint arXiv:1803.02155 (2018)","DOI":"10.18653\/v1\/N18-2074"},{"key":"39_CR31","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: BERT: pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)"},{"key":"39_CR32","unstructured":"Dosovitskiy, A., et al.: An image is worth 16 \u00d7 16 words: transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)"},{"key":"39_CR33","doi-asserted-by":"crossref","unstructured":"Liu, Z., et al.: Swin transformer: hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 10012\u201310022 (2021)","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"39_CR34","unstructured":"Chu, X., Tian, Z., Zhang, B., Wang, X., Shen, C.: Conditional positional encodings for vision transformers. arXiv preprint arXiv:2102.10882 (2021)"},{"key":"39_CR35","unstructured":"Touvron, H., et al.: Llama 2: Open foundation and fine-tuned chat models. arXiv preprint arXiv:2307.09288 (2023)"},{"key":"39_CR36","doi-asserted-by":"crossref","unstructured":"Heo, B., Park, S., Han, D., Yun, S.: Rotary position embedding for vision transformer. arXiv preprint arXiv:2403.13298 (2024)","DOI":"10.1007\/978-3-031-72684-2_17"},{"key":"39_CR37","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"39_CR38","doi-asserted-by":"crossref","unstructured":"Lin, T.Y., Doll\u00e1r, P., Girshick, R., He, K., Hariharan, B., Belongie, S.: Feature pyramid networks for object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2117\u20132125 (2017)","DOI":"10.1109\/CVPR.2017.106"},{"key":"39_CR39","doi-asserted-by":"crossref","unstructured":"Mildenhall, B., Srinivasan, P.P., Tancik, M., Barron, J.T., Ramamoorthi, R., Ng, R.: Nerf: Representing scenes as neural radiance fields for view synthesis. Commun. ACM, 99\u2013106 (2021)","DOI":"10.1145\/3503250"},{"key":"39_CR40","doi-asserted-by":"crossref","unstructured":"Lee, Y., Hwang, J.w., Lee, S., Bae, Y., Park, J.: An energy and GPU-computation efficient backbone network for real-time object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshops, pp. 752\u2013760 (2019)","DOI":"10.1109\/CVPRW.2019.00103"}],"container-title":["Lecture Notes in Computer Science","Advanced Intelligent Computing Technology and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-96-9863-9_39","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,28]],"date-time":"2026-03-28T04:11:30Z","timestamp":1774671090000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-96-9863-9_39"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9789819698622","9789819698639"],"references-count":40,"URL":"https:\/\/doi.org\/10.1007\/978-981-96-9863-9_39","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"24 July 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"ICIC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Intelligent Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Ningbo","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 July 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 July 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"21","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icic2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.ic-icc.cn\/icg\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}