{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T21:14:08Z","timestamp":1783718048324,"version":"3.55.0"},"reference-count":56,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Neurocomputing"],"published-print":{"date-parts":[[2026,11]]},"DOI":"10.1016\/j.neucom.2026.134379","type":"journal-article","created":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T16:23:51Z","timestamp":1783009431000},"page":"134379","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["MotionLifter: Unsupervised text-driven 3D human motion generation with semantic and layout self-supervision"],"prefix":"10.1016","volume":"700","author":[{"given":"Sheng","family":"Liu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jinghao","family":"Cao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-8030-4965","authenticated-orcid":false,"given":"Xiangxiang","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Rui","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yang","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6432-3704","authenticated-orcid":false,"given":"Sidan","family":"Du","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.neucom.2026.134379_bib0005","series-title":"European Conference on Computer Vision","first-page":"580","article-title":"Tm2t: stochastic and tokenized modeling for the reciprocal generation of 3D human motions and texts","author":"Guo","year":"2022"},{"key":"10.1016\/j.neucom.2026.134379_bib0010","series-title":"European Conference on Computer Vision","first-page":"480","article-title":"TEMOS: generating diverse human motions from textual descriptions","author":"Petrovich","year":"2022"},{"key":"10.1016\/j.neucom.2026.134379_bib0015","author":"Lu"},{"key":"10.1016\/j.neucom.2026.134379_bib0020","series-title":"European Conference on Computer Vision","first-page":"246","article-title":"Generating human interaction motions in scenes with text control","author":"Yi","year":"2025"},{"key":"10.1016\/j.neucom.2026.134379_bib0025","series-title":"The Eleventh International Conference on Learning Representations","article-title":"Human motion diffusion model","author":"Tevet","year":"2023"},{"key":"10.1016\/j.neucom.2026.134379_bib0030","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"1900","article-title":"Momask: generative masked modeling of 3D human motions","author":"Guo","year":"2024"},{"key":"10.1016\/j.neucom.2026.134379_bib0035","doi-asserted-by":"crossref","first-page":"31841","DOI":"10.52202\/068431-2308","article-title":"Get3d: a generative model of high quality 3D textured shapes learned from images","volume":"35","author":"Gao","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2026.134379_bib0040","author":"Xie"},{"key":"10.1016\/j.neucom.2026.134379_bib0045","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"5714","article-title":"Unsupervised 3D pose estimation with geometric self-supervision","author":"Chen","year":"2019"},{"key":"10.1016\/j.neucom.2026.134379_bib0050","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"8651","article-title":"Towards alleviating the modeling ambiguity of unsupervised monocular 3D human pose estimation","author":"Yu","year":"2021"},{"key":"10.1016\/j.neucom.2026.134379_bib0055","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"6635","article-title":"ElePose: unsupervised 3D human pose estimation by predicting camera elevation and learning normalizing flows on 2D poses","author":"Wandt","year":"2022"},{"key":"10.1016\/j.neucom.2026.134379_bib0060","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"18000","article-title":"Executing your commands via motion diffusion in latent space","author":"Chen","year":"2023"},{"key":"10.1016\/j.neucom.2026.134379_bib0065","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"5152","article-title":"Generating diverse and natural 3D human motions from text","author":"Guo","year":"2022"},{"key":"10.1016\/j.neucom.2026.134379_bib0070","doi-asserted-by":"crossref","first-page":"20067","DOI":"10.52202\/075280-0880","article-title":"MotionGPT: human motion as a foreign language","volume":"36","author":"Jiang","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2026.134379_bib0075","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"364","article-title":"Remodiffuse: retrieval-augmented motion diffusion model","author":"Zhang","year":"2023"},{"key":"10.1016\/j.neucom.2026.134379_bib0080","series-title":"2022 International Conference on 3D Vision (3DV)","first-page":"414","article-title":"Teach: temporal action composition for 3D humans","author":"Athanasiou","year":"2022"},{"key":"10.1016\/j.neucom.2026.134379_bib0085","series-title":"European Conference on Computer Vision","first-page":"358","article-title":"Motionclip: exposing human motion generation to clip space","author":"Tevet","year":"2022"},{"key":"10.1016\/j.neucom.2026.134379_bib0090","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"482","article-title":"Omg: towards open-vocabulary motion generation via mixture of controllers","author":"Liang","year":"2024"},{"key":"10.1016\/j.neucom.2026.134379_bib0095","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"5442","article-title":"AMASS: archive of motion capture as surface shapes","author":"Mahmood","year":"2019"},{"issue":"4","key":"10.1016\/j.neucom.2026.134379_bib0100","doi-asserted-by":"crossref","first-page":"236","DOI":"10.1089\/big.2016.0028","article-title":"The kit motion-language dataset","volume":"4","author":"Plappert","year":"2016","journal-title":"Big data"},{"key":"10.1016\/j.neucom.2026.134379_bib0105","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"11313","article-title":"Tokenpose: learning keypoint tokens for human pose estimation","author":"Li","year":"2021"},{"key":"10.1016\/j.neucom.2026.134379_bib0110","doi-asserted-by":"crossref","first-page":"38571","DOI":"10.52202\/068431-2795","article-title":"Vitpose: simple vision transformer baselines for human pose estimation","volume":"35","author":"Xu","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"issue":"8","key":"10.1016\/j.neucom.2026.134379_bib0115","doi-asserted-by":"crossref","first-page":"13498","DOI":"10.1109\/TITS.2021.3124981","article-title":"OpenPifPaf: composite fields for semantic keypoint detection and spatio-temporal association","volume":"23","author":"Kreiss","year":"2022","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"issue":"1","key":"10.1016\/j.neucom.2026.134379_bib0120","doi-asserted-by":"crossref","first-page":"681","DOI":"10.1109\/TPAMI.2021.3139918","article-title":"Investigating pose representations and motion contexts modeling for 3D motion prediction","volume":"45","author":"Liu","year":"2022","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.neucom.2026.134379_bib0125","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"9489","article-title":"Learning trajectory dependencies for human motion prediction","author":"Mao","year":"2019"},{"key":"10.1016\/j.neucom.2026.134379_bib0130","series-title":"European Conference on Computer Vision","first-page":"356","article-title":"Implicit neural representations for variable length human motion generation","author":"Cervantes","year":"2022"},{"key":"10.1016\/j.neucom.2026.134379_bib0135","series-title":"Proceedings of the 28th ACM International Conference on Multimedia","first-page":"2021","article-title":"Action2motion: conditioned generation of 3D human motions","author":"Guo","year":"2020"},{"key":"10.1016\/j.neucom.2026.134379_bib0140","series-title":"European Conference on Computer Vision","first-page":"417","article-title":"Posegpt: quantization-based 3D human motion generation and forecasting","author":"Lucas","year":"2022"},{"key":"10.1016\/j.neucom.2026.134379_bib0145","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"10985","article-title":"Action-conditioned 3D human motion synthesis with transformer VAE","author":"Petrovich","year":"2021"},{"key":"10.1016\/j.neucom.2026.134379_bib0150","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"9942","article-title":"Tm2d: bimodality driven 3D dance generation via music-text integration","author":"Gong","year":"2023"},{"key":"10.1016\/j.neucom.2026.134379_bib0155","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"11050","article-title":"Bailando: 3D dance generation by actor-critic GPT with choreographic memory","author":"Siyao","year":"2022"},{"key":"10.1016\/j.neucom.2026.134379_bib0160","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"448","article-title":"Edge: editable dance generation from music","author":"Tseng","year":"2023"},{"key":"10.1016\/j.neucom.2026.134379_bib0165","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"5632","article-title":"Ude: a unified driving engine for human motion generation","author":"Zhou","year":"2023"},{"key":"10.1016\/j.neucom.2026.134379_bib0170","author":"Liu"},{"issue":"12","key":"10.1016\/j.neucom.2026.134379_bib0175","doi-asserted-by":"crossref","first-page":"7518","DOI":"10.1109\/TVCG.2024.3352002","article-title":"GUESS: GradUally enriching SyntheSis for text-driven human motion generation","volume":"30","author":"Gao","year":"2024","journal-title":"IEEE Trans. Vis. Comput. Graph."},{"key":"10.1016\/j.neucom.2026.134379_bib0180","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"6252","article-title":"Towards detailed text-to-motion synthesis via basic-to-advanced hierarchical diffusion model","volume":"vol. 38","author":"Xie","year":"2024"},{"key":"10.1016\/j.neucom.2026.134379_bib0185","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1109\/TPAMI.2025.3573414","article-title":"EigenActor: variant body-object interaction generation evolved from invariant action Basis reasoning","author":"Gao","year":"2025","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.neucom.2026.134379_bib0190","doi-asserted-by":"crossref","first-page":"8900","DOI":"10.1109\/TMM.2025.3607810","article-title":"Jointly understand your command and intention: reciprocal co-evolution between scene-aware 3D human motion synthesis and analysis","volume":"27","author":"Gao","year":"2025","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.neucom.2026.134379_bib0195","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"1396","article-title":"Synthesis of compositional animations from textual descriptions","author":"Ghosh","year":"2021"},{"key":"10.1016\/j.neucom.2026.134379_bib0200","series-title":"International Conference on Machine Learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.neucom.2026.134379_bib0205","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"6796","article-title":"Gaussiandreamer: fast generation from text to 3D gaussians by bridging 2D and 3D diffusion models","author":"Yi","year":"2024"},{"key":"10.1016\/j.neucom.2026.134379_bib0210","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"14928","article-title":"Interdiff: generating 3D human-object interactions with physics-informed diffusion","author":"Xu","year":"2023"},{"key":"10.1016\/j.neucom.2026.134379_bib0215","first-page":"1","article-title":"Intergen: diffusion-based multi-human motion generation under complex interactions","author":"Liang","year":"2024","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.neucom.2026.134379_bib0220","author":"Raab"},{"key":"10.1016\/j.neucom.2026.134379_bib0225","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"1965","article-title":"MAS: multi-view ancestral sampling for 3D motion generation using 2D diffusion","author":"Kapon","year":"2024"},{"key":"10.1016\/j.neucom.2026.134379_bib0230","series-title":"ACM SIGGRAPH 2024 Conference Papers","first-page":"1","article-title":"Flexible motion in-betweening with diffusion models","author":"Cohan","year":"2024"},{"key":"10.1016\/j.neucom.2026.134379_bib0235","series-title":"Proceedings of the SIGGRAPH Asia 2025 Conference Papers","first-page":"1","article-title":"Uni-inter: unifying 3D human motion synthesis across diverse interaction contexts","author":"Liu","year":"2025"},{"issue":"140","key":"10.1016\/j.neucom.2026.134379_bib0240","first-page":"1","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","volume":"21","author":"Raffel","year":"2020","journal-title":"J. Mach. Learn. Res."},{"key":"10.1016\/j.neucom.2026.134379_bib0245","author":"Achiam"},{"key":"10.1016\/j.neucom.2026.134379_bib0250","author":"Song"},{"key":"10.1016\/j.neucom.2026.134379_bib0255","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"8922","article-title":"LoFTR: detector-free local feature matching with transformers","author":"Sun","year":"2021"},{"key":"10.1016\/j.neucom.2026.134379_bib0260","article-title":"Attention is all you need","volume":"30","author":"Vaswani","year":"2017","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.neucom.2026.134379_bib0265","author":"Ho"},{"key":"10.1016\/j.neucom.2026.134379_bib0270","series-title":"SIGGRAPH Asia 2024 Conference Papers","first-page":"1","article-title":"World-grounded human motion recovery via gravity-view coordinates","author":"Shen","year":"2024"},{"issue":"6","key":"10.1016\/j.neucom.2026.134379_bib0275","doi-asserted-by":"crossref","DOI":"10.1145\/2816795.2818013","article-title":"SMPL: a skinned multi-person linear model","volume":"34","author":"Loper","year":"2015","journal-title":"ACM Trans. Graph."},{"key":"10.1016\/j.neucom.2026.134379_bib0280","series-title":"1988 Anthropometric Survey of US Personnel: Summary Statistics Interim Report","author":"Gordon","year":"1989"}],"container-title":["Neurocomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0925231226017777?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0925231226017777?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T20:25:38Z","timestamp":1783715138000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0925231226017777"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,11]]},"references-count":56,"alternative-id":["S0925231226017777"],"URL":"https:\/\/doi.org\/10.1016\/j.neucom.2026.134379","relation":{},"ISSN":["0925-2312"],"issn-type":[{"value":"0925-2312","type":"print"}],"subject":[],"published":{"date-parts":[[2026,11]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"MotionLifter: Unsupervised text-driven 3D human motion generation with semantic and layout self-supervision","name":"articletitle","label":"Article Title"},{"value":"Neurocomputing","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.neucom.2026.134379","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Published by Elsevier B.V.","name":"copyright","label":"Copyright"}],"article-number":"134379"}}