{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,15]],"date-time":"2026-05-15T05:12:34Z","timestamp":1778821954269,"version":"3.51.4"},"reference-count":101,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"5","license":[{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"Damo Academy through Damo Academy Research Intern Program"},{"name":"NUS startup","award":["MOE Tier-1"],"award-info":[{"award-number":["MOE Tier-1"]}]},{"name":"NUS startup","award":["ByteDance"],"award-info":[{"award-number":["ByteDance"]}]},{"name":"NUS startup","award":["ARCTIC"],"award-info":[{"award-number":["ARCTIC"]}]},{"name":"SMI","award":["A8001104-00-00"],"award-info":[{"award-number":["A8001104-00-00"]}]},{"name":"Alibaba"},{"DOI":"10.13039\/100006785","name":"Google","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100006785","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Pattern Anal. Mach. Intell."],"published-print":{"date-parts":[[2026,5]]},"DOI":"10.1109\/tpami.2026.3654201","type":"journal-article","created":{"date-parts":[[2026,1,15]],"date-time":"2026-01-15T20:49:41Z","timestamp":1768510181000},"page":"5755-5773","source":"Crossref","is-referenced-by-count":0,"title":["DyDiT++: Diffusion Transformers With Timestep and Spatial Dynamics for Efficient Visual Generation"],"prefix":"10.1109","volume":"48","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9545-7991","authenticated-orcid":false,"given":"Wangbo","family":"Zhao","sequence":"first","affiliation":[{"name":"DAMO Academy, Alibaba Group, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5706-8784","authenticated-orcid":false,"given":"Yizeng","family":"Han","sequence":"additional","affiliation":[{"name":"DAMO Academy, Alibaba Group, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-9154-8615","authenticated-orcid":false,"given":"Jiasheng","family":"Tang","sequence":"additional","affiliation":[{"name":"DAMO Academy, Alibaba Group, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kai","family":"Wang","sequence":"additional","affiliation":[{"name":"School of Computing, National University of Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6405-4011","authenticated-orcid":false,"given":"Hao","family":"Luo","sequence":"additional","affiliation":[{"name":"DAMO Academy, Alibaba Group, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3667-531X","authenticated-orcid":false,"given":"Yibing","family":"Song","sequence":"additional","affiliation":[{"name":"DAMO Academy, Alibaba Group, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7251-0988","authenticated-orcid":false,"given":"Gao","family":"Huang","sequence":"additional","affiliation":[{"name":"Department of Automation, Tsinghua University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7320-1119","authenticated-orcid":false,"given":"Fan","family":"Wang","sequence":"additional","affiliation":[{"name":"DAMO Academy, Alibaba Group, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2816-4384","authenticated-orcid":false,"given":"Yang","family":"You","sequence":"additional","affiliation":[{"name":"School of Computing, National University of Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","first-page":"6840","article-title":"Denoising diffusion probabilistic models","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Ho"},{"key":"ref2","first-page":"8780","article-title":"Diffusion models beat GANs on image synthesis","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Dhariwal"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"ref4","article-title":"Stable video diffusion: Scaling latent video diffusion models to large datasets","author":"Blattmann","year":"2023"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref6","article-title":"An image is worth 16 \u00d7 16 words: Transformers for image recognition at scale","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Dosovitskiy"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00387"},{"key":"ref8","article-title":"PixArt-$\\alpha$\u03b1: Fast training of diffusion transformer for photorealistic text-to-image synthesis","author":"Chen","year":"2023"},{"key":"ref9","article-title":"PixArt-$\\delta$\u03b4: Fast and controllable image generation with latent consistency models","author":"Chen","year":"2024"},{"key":"ref10","first-page":"12606","article-title":"Scaling rectified flow transformers for high-resolution image synthesis","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Esser"},{"key":"ref11","article-title":"Flux","author":"Labs","year":"2024"},{"key":"ref12","article-title":"Video generation models as world simulators","author":"Brooks","year":"2024"},{"key":"ref13","article-title":"Latte: Latent diffusion transformer for video generation","author":"Ma","year":"2024"},{"key":"ref14","article-title":"Movie gen: A cast of media foundation models","author":"Polyak","year":"2024"},{"key":"ref15","article-title":"WAN: Open and advanced large-scale video generative models","author":"Team","year":"2025"},{"key":"ref16","first-page":"1877","article-title":"Language models are few-shot learners","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Brown"},{"key":"ref17","article-title":"Denoising diffusion implicit models","author":"Song","year":"2020"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1145\/960128.806879"},{"key":"ref19","article-title":"Progressive distillation for fast sampling of diffusion models","author":"Salimans","year":"2022"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01374"},{"key":"ref21","article-title":"Latent consistency models: Synthesizing high-resolution images with few-step inference","author":"Luo","year":"2023"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01492"},{"key":"ref23","article-title":"T-Stitch: Accelerating sampling in pre-trained diffusion models with trajectory stitching","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Pan"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.52202\/075280-0731"},{"key":"ref25","article-title":"Pruning convolutional neural networks for resource efficient inference","author":"Molchanov","year":"2016"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.155"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72980-5_2"},{"key":"ref28","article-title":"Flow matching for generative modeling","author":"Lipman","year":"2022"},{"key":"ref29","article-title":"Flow straight and fast: Learning to generate and transfer data with rectified flow","author":"Liu","year":"2022"},{"key":"ref30","article-title":"Lora: Low-rank adaptation of large language models","author":"Hu","year":"2021"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58583-9_15"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01199"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2024.3393530"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"ref35","article-title":"Dynamic diffusion transformer","author":"Zhao","year":"2024"},{"key":"ref36","doi-asserted-by":"crossref","DOI":"10.36227\/techrxiv.172055626.64129172\/v1","article-title":"A survey on mixture of experts","author":"Cai","year":"2024"},{"key":"ref37","first-page":"48686","article-title":"Temporal dynamic quantization for diffusion models","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"So"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00196"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02160"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01608"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02171"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2021.3117837"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/ICPR.2016.7900006"},{"key":"ref45","first-page":"527","article-title":"Adaptive neural networks for efficient inference","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Bolukbasi"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00244"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20083-0_22"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00551"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00850"},{"key":"ref50","first-page":"11960","article-title":"Not all images are worth 16 \u00d7 16 words: Dynamic transformers for efficient image recognition","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Wang"},{"key":"ref51","first-page":"5770","article-title":"Dynamic grained encoder for vision transformers","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Song"},{"key":"ref52","first-page":"13937","article-title":"DynamicViT: Efficient vision transformers with dynamic token sparsification","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Rao"},{"key":"ref53","article-title":"Not all patches are what you need: Expediting vision transformers via token reorganizations","author":"Liang","year":"2022"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01845"},{"key":"ref55","first-page":"9565","article-title":"HydraLoRA: An asymmetric LoRA architecture for efficient fine-tuning","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Tian"},{"key":"ref56","article-title":"LoRAMoE: Revolutionizing mixture of experts for maintaining world knowledge in language model alignment","author":"Dou","year":"2023"},{"key":"ref57","article-title":"MoELoRa: An MOE-based parameter efficient fine-tuning method for multi-task medical applications","author":"Liu","year":"2023"},{"key":"ref58","first-page":"8162","article-title":"Improved denoising diffusion probabilistic models","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Nichol"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.11671"},{"key":"ref60","first-page":"9782","article-title":"DynaBERT: Dynamic BERT with adaptive width and depth","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Hou"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01261-8_25"},{"key":"ref62","article-title":"Estimating or propagating gradients through stochastic neurons for conditional computation","author":"Bengio","year":"2013"},{"key":"ref63","article-title":"Categorical reparameterization with gumbel-softmax","author":"Jang","year":"2016"},{"key":"ref64","article-title":"Open-SoRA: Democratizing efficient video production for all","author":"Zheng","year":"2024"},{"key":"ref65","article-title":"Open-SoRA plan: Open-source large video generation model","author":"Lin","year":"2024"},{"key":"ref66","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Radford"},{"issue":"1","key":"ref67","first-page":"5485","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","volume":"21","author":"Raffel","year":"2020","journal-title":"J. Mach. Learn. Res."},{"key":"ref68","article-title":"HunyuanVideo: A systematic framework for large video generative models","author":"Kong","year":"2024"},{"key":"ref69","article-title":"Classifier-free diffusion guidance","author":"Ho","year":"2022"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00390"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10599-4_29"},{"key":"ref72","article-title":"The ArtBench dataset: Benchmarking generative models with artworks","author":"Liao","year":"2022"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v31i1.11174"},{"key":"ref74","article-title":"The Caltech-UCSD birds-200\u20132011 dataset","author":"Wah","year":"2011"},{"key":"ref75","article-title":"DiM: Diffusion mamba for efficient high-resolution image synthesis","author":"Teng","year":"2024"},{"key":"ref76","first-page":"6629","article-title":"GANs trained by a two time-scale update rule converge to a local Nash equilibrium","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Heusel"},{"key":"ref77","first-page":"2234","article-title":"Improved techniques for training GANs","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Salimans"},{"key":"ref78","article-title":"Generating images with sparse representations","author":"Nash","year":"2021"},{"key":"ref79","first-page":"3927","article-title":"Improved precision and recall metric for assessing generative models","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Kynk\u00e4\u00e4nniemi"},{"key":"ref80","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00787"},{"key":"ref81","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73242-3_3"},{"key":"ref82","article-title":"Alleviating distortion in image generation via multi-resolution diffusion models","author":"Liu","year":"2024"},{"key":"ref83","article-title":"Token merging: Your ViT but faster","author":"Bolya","year":"2022"},{"key":"ref84","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW59228.2023.00484"},{"key":"ref85","article-title":"Early exiting for accelerated inference in diffusion models","volume-title":"Proc. ICML 2023 Workshop Structured Probabilistic Inference Generative Model.","author":"Moon"},{"key":"ref86","doi-asserted-by":"publisher","DOI":"10.1007\/s11633-025-1562-4"},{"key":"ref87","first-page":"2793","article-title":"Attention is not all you need: Pure attention loses rank doubly exponentially with depth","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Dong"},{"key":"ref88","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2270"},{"key":"ref89","article-title":"ELLA: Equip diffusion models with LLM for enhanced semantic alignment","author":"Hu","year":"2024"},{"key":"ref90","doi-asserted-by":"publisher","DOI":"10.1109\/cvprw63382.2024.00538"},{"key":"ref91","article-title":"Flux.1 lite: Distilling flux1.dev for efficient text-to-image generation","author":"Daniel Verd","year":"2024"},{"key":"ref92","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52734.2025.00689"},{"key":"ref93","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1212.0402"},{"key":"ref94","first-page":"7137","article-title":"First order motion model for image animation","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Siarohin"},{"key":"ref95","article-title":"Faceforensics: A large-scale video dataset for forgery detection in human faces","author":"R\u00f6ssler","year":"2018"},{"key":"ref96","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00251"},{"key":"ref97","article-title":"Towards accurate generative models of video: A new metric & challenges","author":"Unterthiner","year":"2018"},{"key":"ref98","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2025.3541625"},{"key":"ref99","article-title":"A survey of multimodal-guided image editing with text-to-image diffusion models","author":"Shuai","year":"2024"},{"key":"ref100","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"ref101","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i5.28226"}],"container-title":["IEEE Transactions on Pattern Analysis and Machine Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/34\/11474534\/11355684.pdf?arnumber=11355684","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,15]],"date-time":"2026-05-15T04:41:03Z","timestamp":1778820063000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11355684\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5]]},"references-count":101,"journal-issue":{"issue":"5"},"URL":"https:\/\/doi.org\/10.1109\/tpami.2026.3654201","relation":{},"ISSN":["0162-8828","2160-9292","1939-3539"],"issn-type":[{"value":"0162-8828","type":"print"},{"value":"2160-9292","type":"electronic"},{"value":"1939-3539","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,5]]}}}