{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,7]],"date-time":"2026-04-07T21:09:03Z","timestamp":1775596143395,"version":"3.50.1"},"reference-count":61,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"4","license":[{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"Natural Science Foundation of China","doi-asserted-by":"crossref","award":["62276242"],"award-info":[{"award-number":["62276242"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"name":"National Aviation Science Foundation","award":["2022Z071078001"],"award-info":[{"award-number":["2022Z071078001"]}]},{"name":"Hefei Municipal Natural Science Foundation","award":["HZR2431"],"award-info":[{"award-number":["HZR2431"]}]},{"name":"Dreams Foundation of Jianghuai Advance Technology Center","award":["2023-ZM01Z001"],"award-info":[{"award-number":["2023-ZM01Z001"]}]},{"DOI":"10.13039\/501100001809","name":"CAAI-MindSpore Open Fund","doi-asserted-by":"publisher","award":["2023-ZM01Z001"],"award-info":[{"award-number":["2023-ZM01Z001"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Circuits Syst. Video Technol."],"published-print":{"date-parts":[[2026,4]]},"DOI":"10.1109\/tcsvt.2025.3633750","type":"journal-article","created":{"date-parts":[[2025,11,17]],"date-time":"2025-11-17T18:43:02Z","timestamp":1763404982000},"page":"4693-4706","source":"Crossref","is-referenced-by-count":0,"title":["Language-Assisted Reconstruction for Self-Supervised Category-Level 6D Object Pose Estimation With Coarse-to-Fine Correspondence Optimization"],"prefix":"10.1109","volume":"36","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3197-8103","authenticated-orcid":false,"given":"Jun","family":"Yu","sequence":"first","affiliation":[{"name":"Department of Orthopedics, First Affiliated Hospital, the Division of Life Sciences and Medicine, the Department of Automation, Institute of Advanced Technology, University of Science and Technology of China (USTC), Hefei, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-1980-7859","authenticated-orcid":false,"given":"Yunxiang","family":"Zhang","sequence":"additional","affiliation":[{"name":"Department of Automation, USTC, Hefei, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-1976-0889","authenticated-orcid":false,"given":"Zerui","family":"Zhang","sequence":"additional","affiliation":[{"name":"Department of Automation, USTC, Hefei, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-4118-3936","authenticated-orcid":false,"given":"Wei","family":"Xu","sequence":"additional","affiliation":[{"name":"First Affiliated Hospital, Division of Life Sciences and Medicine, USTC, Hefei, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2337-9055","authenticated-orcid":false,"given":"Xiaolong","family":"Shi","sequence":"additional","affiliation":[{"name":"School of Computer and Artificial Intelligence, Zhengzhou University, Zhengzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.15607\/RSS.2021.XVII.025"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1111\/j.1460-2466.1992.tb00811.x"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ISMAR.2007.4538852"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681324"},{"key":"ref5","article-title":"Self-supervised geometric correspondence for category-level 6D object pose estimation in the wild","author":"Zhang","year":"2022","journal-title":"arXiv:2210.07199"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.15607\/rss.2018.xiv.019"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00346"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01217"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00660"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.413"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58580-8_38"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01634"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00469"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00275"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58589-1_32"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01199"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20077-9_2"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/IROS51168.2021.9636212"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19769-7_38"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01165"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20080-9_20"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01446"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/34.88573"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00277"},{"key":"ref25","first-page":"27469","article-title":"Category-level 6D object pose estimation in the wild: A semi-supervised learning approach and a new dataset","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"35","author":"Ze"},{"key":"ref26","first-page":"7207","article-title":"Neural view synthesis and matching for semi-supervised few-shot learning of 3D pose","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Wang"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58574-7_9"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01447"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02039"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/IROS47612.2022.9981145"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01294"},{"key":"ref32","first-page":"15370","article-title":"Leveraging SE(3) equivariance for self-supervised category-level object pose estimation from point clouds","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Li"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i2.20104"},{"key":"ref34","article-title":"Towards self-supervised category-level object pose and size estimation","author":"He","year":"2022","journal-title":"arXiv:2203.02884"},{"key":"ref35","first-page":"1","article-title":"Auto-encoding variational Bayes","volume-title":"Proc. ICLR","author":"Kingma"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1145\/3422622"},{"key":"ref37","first-page":"1","article-title":"Learning a probabilistic latent space of object shapes via 3D generative-adversarial modeling","volume-title":"Proc. NeurIPS","author":"Wu"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00459"},{"key":"ref39","first-page":"1","article-title":"Neural discrete representation learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"30","author":"Van Den Oord"},{"key":"ref40","first-page":"1","article-title":"Generating diverse high-fidelity images with VQ-VAE-2","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"32","author":"Razavi"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20062-5_6"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00040"},{"key":"ref43","article-title":"SAR3D: Autoregressive 3D object generation and understanding via multi-scale 3D VQVAE","author":"Chen","year":"2024","journal-title":"arXiv:2411.16856"},{"key":"ref44","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"Proc. Int. Conf. Mach. Learn.","volume":"139","author":"Radford"},{"key":"ref45","article-title":"Hierarchical text-conditional image generation with CLIP latents","author":"Ramesh","year":"2022","journal-title":"arXiv:2204.06125"},{"key":"ref46","first-page":"36479","article-title":"Photorealistic text-to-image diffusion models with deep language understanding","volume-title":"Proc. NIPS","volume":"35","author":"Saharia"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01805"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01753"},{"key":"ref49","article-title":"Soulstyler: Using large language model to guide image style transfer for target object","author":"Chen","year":"2023","journal-title":"arXiv:2311.13562"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01104"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00699"},{"key":"ref52","first-page":"652","article-title":"PointNet: Deep learning on point sets for 3D classification and segmentation","volume-title":"Proc. IEEE Conf. Comput. Vis. Pattern Recognit. (CVPR)","author":"Qi"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"ref54","article-title":"Estimating or propagating gradients through stochastic neurons for conditional computation","author":"Bengio","year":"2013","journal-title":"arXiv:1308.3432"},{"key":"ref55","article-title":"DINO: DETR with improved DeNoising anchor boxes for end-to-end object detection","author":"Zhang","year":"2022","journal-title":"arXiv:2203.03605"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.322"},{"key":"ref57","article-title":"ShapeNet: An information-rich 3D model repository","author":"Chang","year":"2015","journal-title":"arXiv:1512.03012"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00354"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00666"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00163"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00674"}],"container-title":["IEEE Transactions on Circuits and Systems for Video Technology"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/76\/11475579\/11251041.pdf?arnumber=11251041","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,7]],"date-time":"2026-04-07T20:02:43Z","timestamp":1775592163000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11251041\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4]]},"references-count":61,"journal-issue":{"issue":"4"},"URL":"https:\/\/doi.org\/10.1109\/tcsvt.2025.3633750","relation":{},"ISSN":["1051-8215","1558-2205"],"issn-type":[{"value":"1051-8215","type":"print"},{"value":"1558-2205","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,4]]}}}