{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T18:35:56Z","timestamp":1787510156499,"version":"build-2736575974"},"reference-count":45,"publisher":"Springer Science and Business Media LLC","issue":"9","license":[{"start":{"date-parts":[[2025,5,21]],"date-time":"2025-05-21T00:00:00Z","timestamp":1747785600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,5,21]],"date-time":"2025-05-21T00:00:00Z","timestamp":1747785600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"National Natural Science Foundation of China (NSFC)\/Research Grants Council (RGC) of Hong Kong Joint Research Scheme","award":["62361166630"],"award-info":[{"award-number":["62361166630"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62273323"],"award-info":[{"award-number":["62273323"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Shen-zhen Key Basic Research Project","award":["JCYJ2024120212442703"],"award-info":[{"award-number":["JCYJ2024120212442703"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2025,9]]},"DOI":"10.1007\/s11263-025-02469-5","type":"journal-article","created":{"date-parts":[[2025,5,21]],"date-time":"2025-05-21T09:25:02Z","timestamp":1747819502000},"page":"6015-6024","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Spatial-Temporal Transformer for Single RGB-D Camera Synchronous Tracking and Reconstruction of Non-rigid Dynamic Objects"],"prefix":"10.1007","volume":"133","author":[{"given":"Xiaofei","family":"Liu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhengkun","family":"Yi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xinyu","family":"Wu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wanfeng","family":"Shang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,5,21]]},"reference":[{"key":"2469_CR1","doi-asserted-by":"crossref","unstructured":"Alcantarilla, P. F, Bartoli, A., & Davison, A. J. (2012). Kaze features. In: Computer Vision\u2013ECCV 2012: 12th European Conference on Computer Vision, Florence, Italy, October 7-13, 2012, Proceedings, Part VI 12, Springer (pp. 214\u2013227).","DOI":"10.1007\/978-3-642-33783-3_16"},{"key":"2469_CR2","doi-asserted-by":"crossref","unstructured":"Amberg, B., Romdhani, S., & Vetter, T. (2007). Optimal step nonrigid icp algorithms for surface registration. In: 2007 IEEE conference on computer vision and pattern recognition, IEEE (pp. 1\u20138).","DOI":"10.1109\/CVPR.2007.383165"},{"key":"2469_CR3","unstructured":"Bertasius, G., Wang, H., & Torresani, L. (2021). Is space-time attention all you need for video understanding? In: ICML (pp.\u00a04)."},{"key":"2469_CR4","unstructured":"Besl, P. J., & McKay, N. D. (1992). Method for registration of 3-d shapes. In: Sensor fusion IV: control paradigms and data structures. Spie (pp. 586\u2013606)."},{"key":"2469_CR5","first-page":"18727","volume":"33","author":"A Bozic","year":"2020","unstructured":"Bozic, A., Palafox, P., Zollh\u00f6fer, M., et al. (2020a). Neural non-rigid tracking. Advances in Neural Information Processing Systems, 33, 18727\u201318737.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2469_CR6","doi-asserted-by":"crossref","unstructured":"Bozic, A., Zollhofer, M., & Theobalt, C., et\u00a0al. (2020b). Deepdeform: Learning non-rigid rgb-d reconstruction with semi-supervised data. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (pp. 7002\u20137012).","DOI":"10.1109\/CVPR42600.2020.00703"},{"key":"2469_CR7","first-page":"1403","volume":"34","author":"A Bozic","year":"2021","unstructured":"Bozic, A., Palafox, P., Thies, J., et al. (2021). Transformerfusion: Monocular rgb scene reconstruction using transformers. Advances in Neural Information Processing Systems, 34, 1403\u20131414.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2469_CR8","first-page":"1877","volume":"33","author":"T Brown","year":"2020","unstructured":"Brown, T., Mann, B., Ryder, N., et al. (2020). Language models are few-shot learners. Advances in neural information processing systems, 33, 1877\u20131901.","journal-title":"Advances in neural information processing systems"},{"key":"2469_CR9","unstructured":"Cai, H., Feng, W., & Feng, X., et\u00a0al. (2022). Neural surface reconstruction of dynamic scenes with monocular rgb-d camera. arXiv preprint arXiv:2206.15258."},{"key":"2469_CR10","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., & Synnaeve, G., et\u00a0al. (2020). End-to-end object detection with transformers. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part I 16, Springer (pp. 213\u2013229).","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"2469_CR11","doi-asserted-by":"crossref","unstructured":"Curless, B., & Levoy, M. (1996). A volumetric method for building complex models from range images. In: Proceedings of the 23rd annual conference on Computer graphics and interactive techniques (pp .303\u2013312).","DOI":"10.1145\/237170.237269"},{"key":"2469_CR12","unstructured":"Devlin, J., Chang, M. W., & Lee, K. et\u00a0al. (2018). Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805."},{"key":"2469_CR13","doi-asserted-by":"crossref","unstructured":"Dosovitskiy, A., Fischer, P., & Ilg, E., et\u00a0al. (2015). Flownet: Learning optical flow with convolutional networks. In: Proceedings of the IEEE international conference on computer vision (pp. 2758\u20132766).","DOI":"10.1109\/ICCV.2015.316"},{"key":"2469_CR14","unstructured":"Dosovitskiy, A., Beyer, L., & Kolesnikov, A., et\u00a0al. (2020). An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929."},{"key":"2469_CR15","doi-asserted-by":"crossref","unstructured":"Gu, X., Wang, Y., & Wu, C., et\u00a0al. (2019). Hplflownet: Hierarchical permutohedral lattice flownet for scene flow estimation on large-scale point clouds. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (pp. 3254\u20133263).","DOI":"10.1109\/CVPR.2019.00337"},{"key":"2469_CR16","unstructured":"Hao, Y., Song, H., & Dong, L., et\u00a0al. (2022). Language models are general-purpose interfaces. arXiv preprint arXiv:2206.06336."},{"key":"2469_CR17","doi-asserted-by":"crossref","unstructured":"He, K., Chen, X., & Xie, S., et\u00a0al. (2022). Masked autoencoders are scalable vision learners. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (pp. 16000\u201316009).","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"2469_CR18","doi-asserted-by":"crossref","unstructured":"Innmann, M., Zollh\u00f6fer, M., & Nie\u00dfner, M., et\u00a0al. (2016). Volumedeform: Real-time volumetric non-rigid reconstruction. In: Computer Vision\u2013ECCV 2016: 14th European Conference, Amsterdam, The Netherlands, October 11-14, 2016, Proceedings, Part VIII 14, Springer (pp. 362\u2013379).","DOI":"10.1007\/978-3-319-46484-8_22"},{"key":"2469_CR19","doi-asserted-by":"publisher","unstructured":"Kazhdan, M., & Hoppe, H. (2013). Screened poisson surface reconstruction. ACM Trans Graph, 32(3). https:\/\/doi.org\/10.1145\/2487228.2487237.","DOI":"10.1145\/2487228.2487237"},{"key":"2469_CR20","unstructured":"Li, J., Li, D., & Xiong, C., et\u00a0al. (2022). Blip: Bootstrapping language-image pre-training for unified vision-language understanding and generation. In: International Conference on Machine Learning, PMLR (pp. 12888\u201312900)."},{"key":"2469_CR21","doi-asserted-by":"crossref","unstructured":"Li, Y., Bozic, A., & Zhang, T., et\u00a0al. (2020). Learning to optimize non-rigid tracking. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (pp. 4910\u20134918).","DOI":"10.1109\/CVPR42600.2020.00496"},{"key":"2469_CR22","doi-asserted-by":"crossref","unstructured":"Li, Y., Takehara, H., & Taketomi, T., et\u00a0al. (2021). 4dcomplete: Non-rigid motion estimation beyond the observable surface. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (pp. 12706\u201312716).","DOI":"10.1109\/ICCV48922.2021.01247"},{"key":"2469_CR23","doi-asserted-by":"crossref","unstructured":"Lin, W., Zheng, C., & Yong, J. H., et\u00a0al. (2022). Occlusionfusion: Occlusion-aware motion estimation for real-time dynamic 3d reconstruction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (pp. 1736\u20131745).","DOI":"10.1109\/CVPR52688.2022.00178"},{"key":"2469_CR24","doi-asserted-by":"crossref","unstructured":"Liu, X., Qi, C. R., & Guibas, L. J. (2019). Flownet3d: Learning scene flow in 3d point clouds. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (pp. 529\u2013537).","DOI":"10.1109\/CVPR.2019.00062"},{"key":"2469_CR25","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., & Cao, Y., et\u00a0al. (2021). Swin transformer: Hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF international conference on computer vision (pp. 10012\u201310022).","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"2469_CR26","unstructured":"Loshchilov, I., & Hutter, F. (2017). Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101."},{"key":"2469_CR27","doi-asserted-by":"publisher","first-page":"91","DOI":"10.1023\/B:VISI.0000029664.99615.94","volume":"60","author":"DG Lowe","year":"2004","unstructured":"Lowe, D. G. (2004). Distinctive image features from scale-invariant keypoints. International journal of computer vision, 60, 91\u2013110.","journal-title":"International journal of computer vision"},{"key":"2469_CR28","unstructured":"Lu, J., Clark, C., & Zellers, R., et\u00a0al. (2022). Unified-io: A unified model for vision, language, and multi-modal tasks. arXiv preprint arXiv:2206.08916."},{"key":"2469_CR29","doi-asserted-by":"crossref","unstructured":"Ma, W. C., Wang, S., & Hu, R., et\u00a0al. (2019). Deep rigid instance scene flow. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (pp. 3614\u20133622).","DOI":"10.1109\/CVPR.2019.00373"},{"issue":"1","key":"2469_CR30","doi-asserted-by":"publisher","first-page":"99","DOI":"10.1145\/3503250","volume":"65","author":"B Mildenhall","year":"2021","unstructured":"Mildenhall, B., Srinivasan, P. P., Tancik, M., et al. (2021). Nerf: Representing scenes as neural radiance fields for view synthesis. Communications of the ACM, 65(1), 99\u2013106.","journal-title":"Communications of the ACM"},{"key":"2469_CR31","doi-asserted-by":"crossref","unstructured":"Newcombe, R. A., Fox, D., & Seitz, S. M. (2015). Dynamicfusion: Reconstruction and tracking of non-rigid scenes in real-time. In: Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 343\u2013352).","DOI":"10.1109\/CVPR.2015.7298631"},{"key":"2469_CR32","unstructured":"Ouyang, L., Wu, J., & Jiang, X., et\u00a0al. (2022). Training language models to follow instructions with human feedback. arXiv preprint arXiv:2203.02155."},{"key":"2469_CR33","unstructured":"Radford, A., Kim, J. W., & Hallacy, C., et\u00a0al. (2021). Learning transferable visual models from natural language supervision. In: International conference on machine learning, PMLR (pp. 8748\u20138763)."},{"key":"2469_CR34","doi-asserted-by":"crossref","unstructured":"Ronneberger, O., Fischer, P., & Brox, T. (2015). U-net: Convolutional networks for biomedical image segmentation. In: Medical Image Computing and Computer-Assisted Intervention\u2013MICCAI 2015: 18th International Conference, Munich, Germany, October 5-9, 2015, Proceedings, Part III 18, Springer (pp. 234\u2013241).","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"2469_CR35","doi-asserted-by":"crossref","unstructured":"Slavcheva, M., Baust, M., & Cremers, D., et\u00a0al. (2017). Killingfusion: Non-rigid 3d reconstruction without correspondences. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (pp. 1386\u20131395).","DOI":"10.1109\/CVPR.2017.581"},{"key":"2469_CR36","doi-asserted-by":"crossref","unstructured":"Slavcheva, M., Baust, M., & Ilic, S. (2018). Sobolevfusion: 3d reconstruction of scenes undergoing free non-rigid motion. In: Proceedings of the IEEE conference on computer vision and pattern recognition (pp. 2646\u20132655).","DOI":"10.1109\/CVPR.2018.00280"},{"key":"2469_CR37","unstructured":"Sorkine, O., & Alexa, M. (2007). As-rigid-as-possible surface modeling. In: Symposium on Geometry processing (pp. 109\u2013116)."},{"key":"2469_CR38","doi-asserted-by":"crossref","unstructured":"Sumner, R. W., Schmid, J., & Pauly, M. (2007). Embedded deformation for shape manipulation. ACM SIGGRAPH 2007 papers.","DOI":"10.1145\/1275808.1276478"},{"key":"2469_CR39","doi-asserted-by":"crossref","unstructured":"Tombari, F., Salti, S., & Di\u00a0Stefano, L. (2010). Unique signatures of histograms for local surface description. In: Computer Vision\u2013ECCV 2010: 11th European Conference on Computer Vision, Heraklion, Crete, Greece, September 5-11, 2010, Proceedings, Part III 11, Springer (pp. 356\u2013369).","DOI":"10.1007\/978-3-642-15558-1_26"},{"key":"2469_CR40","unstructured":"Vaswani, A., Shazeer, N., & Parmar, N., et al. (2017). Attention is all you need. Advances in neural information processing systems 30."},{"issue":"1","key":"2469_CR41","doi-asserted-by":"publisher","first-page":"302","DOI":"10.1007\/s11263-022-01702-9","volume":"131","author":"K Vo","year":"2023","unstructured":"Vo, K., Truong, S., Yamazaki, K., et al. (2023). Aoe-net: Entities interactions modeling with adaptive attention mechanism for temporal action proposals generation. International Journal of Computer Vision, 131(1), 302\u2013323.","journal-title":"International Journal of Computer Vision"},{"key":"2469_CR42","doi-asserted-by":"crossref","unstructured":"Wang, D., Cui, X., & Chen, X., et\u00a0al. (2021). Multi-view 3d reconstruction with transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (pp. 5722\u20135731).","DOI":"10.1109\/ICCV48922.2021.00567"},{"key":"2469_CR43","doi-asserted-by":"crossref","unstructured":"Wang, W., Bao, H., & Dong, L., et\u00a0al. (2022). Image as a foreign language: Beit pretraining for all vision and vision-language tasks. arXiv preprint arXiv:2208.10442.","DOI":"10.1109\/CVPR52729.2023.01838"},{"key":"2469_CR44","first-page":"23818","volume":"29","author":"L Yahui","year":"2021","unstructured":"Yahui, L., Sangineto, E., Wei, B., et al. (2021). Efficient training of visual transformers with small-size datasets. ADVANCES IN NEURAL INFORMATION PROCESSING SYSTEMS, 29, 23818\u201323830.","journal-title":"ADVANCES IN NEURAL INFORMATION PROCESSING SYSTEMS"},{"key":"2469_CR45","doi-asserted-by":"crossref","unstructured":"Zhao, H., Jiang, L., & Jia, J., et\u00a0al. (2021). Point transformer. In: Proceedings of the IEEE\/CVF international conference on computer vision (pp. 16259\u201316268).","DOI":"10.1109\/ICCV48922.2021.01595"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-025-02469-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-025-02469-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-025-02469-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,9]],"date-time":"2025-09-09T08:05:40Z","timestamp":1757405140000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-025-02469-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,21]]},"references-count":45,"journal-issue":{"issue":"9","published-print":{"date-parts":[[2025,9]]}},"alternative-id":["2469"],"URL":"https:\/\/doi.org\/10.1007\/s11263-025-02469-5","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,5,21]]},"assertion":[{"value":"21 June 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 May 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 May 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"This work was supported in part by the National Natural Science Foundation of China (NSFC)\/Research Grants Council (RGC) of Hong Kong Joint Research Scheme under Grant 62361166630; in part by NSFC under Grant 62273323; and in part by the Shenzhen Key Basic Research Project (Grant No. JCYJ20241202124427037). The authors have no competing interests to declare that are relevant to the content of this article.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of Interest Statement"}}]}}