{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,22]],"date-time":"2026-07-22T03:58:43Z","timestamp":1784692723382,"version":"3.55.0"},"reference-count":112,"publisher":"Tech Science Press","issue":"1","license":[{"start":{"date-parts":[[2024,7,19]],"date-time":"2024-07-19T00:00:00Z","timestamp":1721347200000},"content-version":"vor","delay-in-days":200,"URL":"https:\/\/doi.org\/10.32604\/TSP-CROSSMARKPOLICY"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["CMC"],"published-print":{"date-parts":[[2024]]},"DOI":"10.32604\/cmc.2024.050790","type":"journal-article","created":{"date-parts":[[2024,6,27]],"date-time":"2024-06-27T08:15:13Z","timestamp":1719476113000},"page":"37-60","update-policy":"https:\/\/doi.org\/10.32604\/tsp-crossmarkpolicy","source":"Crossref","is-referenced-by-count":19,"title":["A Comprehensive Survey of Recent Transformers in Image, Video and Diffusion Models"],"prefix":"10.32604","volume":"80","author":[{"given":"Dinh Phu Cuong","family":"Le","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dong","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Viet-Tuan","family":"Le","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"17807","published-online":{"date-parts":[[2024]]},"reference":[{"key":"ref1","series-title":"31st Int. Conf. Neural Inf. Process. Syst. (NIPS\u201917)","first-page":"6000","article-title":"Attention is all you need","volume":"30","author":"Vaswani","year":"2017"},{"key":"ref2","series-title":"2019 Conf. North American Chapter Assoc. Comput. Linguist.: Human Lang. Technol.","first-page":"4171","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","volume":"1","author":"Devlin","year":"2019"},{"key":"ref3","first-page":"9","article-title":"Language models are unsupervised multitask learners","volume":"1","author":"Radford","year":"2019","journal-title":"OpenAI Blog"},{"key":"ref4","series-title":"34th Int. Conf. Neural Inf. Process. Syst.","first-page":"1877","article-title":"Language models are few-shot learners","volume":"33","author":"Brown","year":"2020"},{"key":"ref5","series-title":"Adv. Neural Inf. Process. Syst.","first-page":"1097","article-title":"Imagenet classification with deep convolutional neural networks","volume":"25","author":"Krizhevsky","year":"2012"},{"key":"ref6","series-title":"IEEE\/CVF Conf. Comput. Vis. Pattern Recognit.","first-page":"7794","article-title":"Non-local neural networks","author":"Wang","year":"2018"},{"key":"ref7","series-title":"IEEE\/CVF Conf. Comput. Vis. Pattern Recognit.","first-page":"7132","article-title":"Squeeze-and-excitation networks","author":"Hu","year":"2018"},{"key":"ref8","series-title":"IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR)","first-page":"11505","article-title":"Orthogonal convolutional neural networks","author":"Wang","year":"2020"},{"key":"ref9","series-title":"European Conf. Comput. Vis. (ECCV)","first-page":"3","article-title":"CBAM: Convolutional block attention module","volume":"11211","author":"Woo","year":"2018"},{"key":"ref10","series-title":"Int. Conf. Learn. Represent.","article-title":"An image is worth 16 \u00d716 words: Transformers for image recognition at scale","author":"Dosovitskiy","year":"2021"},{"key":"ref11","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3505244","article-title":"Transformers in vision: A survey","volume":"54","author":"Khan","year":"2022","journal-title":"ACM Comput. Surv. (CSUR)"},{"key":"ref12","first-page":"1","article-title":"A survey of visual transformers","author":"Liu","year":"2023","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"ref13","doi-asserted-by":"crossref","unstructured":"A. M. Hafiz, S. A. Parah, and R. U. A. Bhat, \u201cAttention mechanisms and deep learning for machine vision: A survey of the state of the art,\u201d arXiv preprint arXiv:2106.07550, 2021.","DOI":"10.21203\/rs.3.rs-510910\/v1"},{"key":"ref14","doi-asserted-by":"crossref","first-page":"111","DOI":"10.1016\/j.aiopen.2022.10.001","article-title":"A survey of transformers","volume":"3","author":"Lin","year":"2022","journal-title":"AI Open"},{"key":"ref15","doi-asserted-by":"crossref","first-page":"100520","DOI":"10.1016\/j.patter.2022.100520","article-title":"Are we ready for a new paradigm shift? A survey on visual deep MLP","volume":"3","author":"Liu","year":"2022","journal-title":"Patterns"},{"key":"ref16","doi-asserted-by":"crossref","first-page":"12922","DOI":"10.1109\/TPAMI.2023.3243465","article-title":"Video transformers: A survey","volume":"45","author":"Selva","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"ref17","unstructured":"E. Min et al., \u201cTransformer for graphs: An overview from architecture perspective,\u201d arXiv preprint arXiv:2202.08455, 2022."},{"key":"ref18","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1016\/j.aiopen.2022.01.001","article-title":"Survey: Transformer based video-language pre-training","volume":"3","author":"Ruan","year":"2022","journal-title":"AI Open"},{"key":"ref19","doi-asserted-by":"crossref","first-page":"87","DOI":"10.1109\/TPAMI.2022.3152247","article-title":"A survey on vision transformer","volume":"45","author":"Han","year":"2022","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"ref20","unstructured":"Y. Yang et al., \u201cTransformers meet visual learning understanding: A comprehensive review,\u201d arXiv preprint arXiv:2203.12944, 2022."},{"key":"ref21","unstructured":"K. Islam, \u201cRecent advances in vision transformer: A survey and outlook of recent work,\u201d arXiv preprint arXiv:2203.01536, 2022."},{"key":"ref22","doi-asserted-by":"crossref","first-page":"33","DOI":"10.1007\/s41095-021-0247-3","article-title":"Transformers in computational visual media: A survey","volume":"8","author":"Xu","year":"2022","journal-title":"Comput. Vis. Media"},{"key":"ref23","series-title":"Adv. Neural Inf. Process. Syst.","first-page":"24261","article-title":"MLP-Mixer: An all-MLP architecture for vision","volume":"34","author":"Tolstikhin","year":"2021"},{"key":"ref24","doi-asserted-by":"crossref","first-page":"5314","DOI":"10.1109\/TPAMI.2022.3206148","article-title":"ResMLP: Feedforward networks for image classification with data-efficient training","volume":"45","author":"Touvron","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"ref25","unstructured":"L. Melas-Kyriazi, \u201cDo you even need attention? A stack of feed-forward layers does surprisingly well on imagenet,\u201d arXiv preprint arXiv:2105.02723, 2021."},{"key":"ref26","series-title":"IEEE Conf. Comput. Vis. Pattern Recognit. (CVPR)","first-page":"770","article-title":"Deep residual learning for image recognition","author":"He","year":"2016"},{"key":"ref27","unstructured":"J. L. Ba, J. R. Kiros, and G. E. Hinton, \u201cLayer normalization,\u201d arXiv preprint arXiv:1607.06450, 2016."},{"key":"ref28","series-title":"IEEE\/CVF Int. Conf. Comput. Vis. (ICCV)","first-page":"9650","article-title":"Emerging properties in self-supervised vision transformers","author":"Caron","year":"2021"},{"key":"ref29","series-title":"IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR)","first-page":"12063","article-title":"MSG-Transformer: Exchanging local spatial information by manipulating messenger tokens","author":"Fang","year":"2022"},{"key":"ref30","series-title":"IEEE\/CVF Int. Conf. Comput. Vis. (ICCV)","first-page":"568","article-title":"Pyramid Vision Transformer: A versatile backbone for dense prediction without convolutions","author":"Wang","year":"2021"},{"key":"ref31","doi-asserted-by":"crossref","first-page":"415","DOI":"10.1007\/s41095-022-0274-8","article-title":"PVT v2: Improved baselines with pyramid vision transformer","volume":"8","author":"Wang","year":"2022","journal-title":"Comput. Vis. Media"},{"key":"ref32","series-title":"IEEE\/CVF Int. Conf. Comput. Vis. (ICCV)","first-page":"9992","article-title":"Swin transformer: Hierarchical vision transformer using shifted windows","author":"Liu","year":"2021"},{"key":"ref33","series-title":"IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR)","first-page":"11999","article-title":"Swin Transformer V2: Scaling up capacity and resolution","author":"Liu","year":"2022"},{"key":"ref34","doi-asserted-by":"crossref","first-page":"3240","DOI":"10.1007\/s10489-022-03613-1","article-title":"Attention-based residual autoencoder for video anomaly detection","volume":"53","author":"Le","year":"2023","journal-title":"Appl. Intell."},{"key":"ref35","series-title":"IEEE Conf. Comput. Vis. Pattern Recognit. (CVPR)","first-page":"17662","article-title":"Uformer: A general U-shaped transformer for image restoration","author":"Wang","year":"2022"},{"key":"ref36","unstructured":"Y. Li, K. Zhang, J. Cao, R. Timofte, and L. van Gool, \u201cLocalViT: Bringing locality to vision transformers,\u201d arXiv preprint arXiv:2104.05707, 2021."},{"key":"ref37","series-title":"IEEE\/CVF Int. Conf. Comput. Vis. (ICCV)","first-page":"559","article-title":"Incorporating convolution designs into visual transformers","author":"Yuan","year":"2021"},{"key":"ref38","series-title":"IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR)","first-page":"5718","article-title":"Restormer: Efficient transformer for high-resolution image restoration","author":"Zamir","year":"2022"},{"key":"ref39","series-title":"Adv. Neural Inf. Process. Syst.","first-page":"9355","article-title":"Twins: Revisiting the design of spatial attention in vision transformers","volume":"34","author":"Chu","year":"2021"},{"key":"ref40","series-title":"IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR)","first-page":"12114","article-title":"Cswin transformer: A general vision transformer backbone with cross-shaped windows","author":"Dong","year":"2022"},{"key":"ref41","unstructured":"Z. Huang, Y. Ben, G. Luo, P. Cheng, G. Yu and B. Fu, \u201cShuffle transformer: Rethinking spatial shuffle for vision transformer,\u201d arXiv preprint arXiv:2106.03650, 2021."},{"key":"ref42","series-title":"Adv. Neural Inf. Proce. Syst.","first-page":"12992","article-title":"Glance-and-Gaze vision transformer","volume":"34","author":"Yu","year":"2021"},{"key":"ref43","series-title":"IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR)","first-page":"6185","article-title":"Neighborhood attention transformer","author":"Hassani","year":"2023"},{"key":"ref44","series-title":"Eur. Conf. Comput. Vis. (ECCV)","first-page":"74","article-title":"DaViT: Dual attention vision transformers","author":"Ding","year":"2022"},{"key":"ref45","series-title":"IEEE\/CVF Int. Conf. Comput. Vis. (ICCV)","first-page":"2978","article-title":"Multi-scale vision longformer: A new vision transformer for high-resolution image encoding","author":"Zhang","year":"2021"},{"key":"ref46","series-title":"IEEE\/CVF Int. Conf. Comput. Vis. (ICCV)","first-page":"22","article-title":"CvT: Introducing convolutions to vision transformers","author":"Wu","year":"2021"},{"key":"ref47","series-title":"IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR)","first-page":"4804","article-title":"MViTv2: Improved multiscale vision transformers for classification and detection","author":"Li","year":"2022"},{"key":"ref48","series-title":"Adv. Neural Inf. Process. Syst.","first-page":"28522","article-title":"ViTAE: Vision transformer advanced by exploring intrinsic inductive bias","volume":"34","author":"Xu","year":"2021"},{"key":"ref49","series-title":"EEE\/CVF Int. Conf. Comput. Vis. (ICCV)","first-page":"589","article-title":"Visformer: The vision-friendly transformer","author":"Chen","year":"2021"},{"key":"ref50","series-title":"Int. Conf. Learn. Represent.","article-title":"Quadtree attention for vision transformers","author":"Tang","year":"2022"},{"key":"ref51","series-title":"IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR)","first-page":"2981","article-title":"HR-NAS: Searching efficient high-resolution neural architectures with lightweight transformers","author":"Ding","year":"2021"},{"key":"ref52","series-title":"Adv. Neural Inf. Process. Syst.","first-page":"23495","article-title":"Inception transformer","volume":"35","author":"Si","year":"2022"},{"key":"ref53","series-title":"Adv. Neural Inf. Process. Syst.","first-page":"35632","article-title":"MCMAE: Masked convolution meets masked autoencoders","volume":"35","author":"Gao","year":"2022"},{"key":"ref54","unstructured":"X. Li, W. Wang, L. Yang, and J. Yang, \u201cUniform masking: Enabling mae pre-training for pyramid-based vision transformers with locality,\u201d arXiv preprint arXiv:2205.10063, 2022."},{"key":"ref55","series-title":"The Eleventh Int. Conf. Learn. Represent. (ICLR)","article-title":"Vision transformer adapter for dense predictions","author":"Chen","year":"2023"},{"key":"ref56","first-page":"6575","article-title":"VOLO: Vision outlooker for visual recognition","volume":"45","author":"Yuan","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"ref57","series-title":"Medical Image Comput. Computer-Assisted Interven.\u2013MICCAI 2015: 18th Int. Conf.","first-page":"234","article-title":"U-Net: Convolutional networks for biomedical image segmentation","author":"Ronneberger","year":"2015"},{"key":"ref58","unstructured":"J. Chen et al., \u201cTransUNet: Transformers make strong encoders for medical image segmentation,\u201d arXiv preprint arXiv:2102.04306, 2021."},{"key":"ref59","series-title":"IEEE\/CVF Winter Conf. Appl. Comput. Vis. (WACV)","first-page":"1748","article-title":"UNETR: Transformers for 3D medical image segmentation","author":"Hatamizadeh","year":"2022"},{"key":"ref60","series-title":"Mach. Learn. Med. Imaging: 12th Int. Workshop","first-page":"267","article-title":"U-Net Transformer: Self and cross attention for medical image segmentation","author":"Petit"},{"key":"ref61","series-title":"Medical Image Comput. Computer Assisted Interven.\u2013MICCAI 2021: 24th Int. Conf.","first-page":"61","article-title":"UTNet: A hybrid transformer architecture for medical image segmentation","author":"Gao","year":"2021"},{"key":"ref62","unstructured":"Y. Gao, M. Zhou, D. Liu, and D. Metaxas, \u201cA multi-scale transformer for medical image segmentation: Architectures, model efficiency, and benchmarks,\u201d arXiv preprint arXiv:2203.00131, 2022."},{"key":"ref63","series-title":"ICASSP 2022-2022 IEEE Int. Conf. Acoust., Speech and Signal Process. (ICASSP)","first-page":"2390","article-title":"Mixed transformer U-Net for medical image segmentation","author":"Wang","year":"2022"},{"key":"ref64","series-title":"AAAI Conf. Artif. Intell.","first-page":"2441","article-title":"UCTransNet: Rethinking the skip connections in U-Net from a channel-wise perspective with transformer","volume":"36","author":"Wang","year":"2022"},{"key":"ref65","series-title":"Medical Image Comput. Comput. Assisted Interven.-MICCAI 2021: 24th Int. Conf., Proc., Part I 24","first-page":"14","article-title":"TransFuse: Fusing transformers and CNNs for medical image segmentation","author":"Zhang","year":"2021"},{"key":"ref66","series-title":"Eur. Conf. Comput. Vis. (ECCV)","first-page":"205","article-title":"Swin-Unet: Unet-like pure transformer for medical image segmentation","author":"Cao","year":"2023"},{"key":"ref67","series-title":"Int. MICCAI Brain. Workshop","first-page":"272","article-title":"Swin UNETR: Swin transformers for semantic segmentation of brain tumors in MRI images","author":"Hatamizadeh","year":"2022"},{"key":"ref68","series-title":"Medical Image Comput. Computer-Assisted Interven.\u2013MICCAI 2022","first-page":"162","article-title":"A robust volumetric transformer for accurate 3D tumor segmentation","author":"Peiris","year":"2022"},{"key":"ref69","series-title":"IEEE\/CVF Int. Conf. Comput. Vis. (ICCV)","first-page":"7242","article-title":"Segmenter: Transformer for semantic segmentation","author":"Strudel","year":"2021"},{"key":"ref70","series-title":"IEEE Conf. Comput. Vis. Pattern Recognit. (CVPR)","first-page":"12073","article-title":"TopFormer: Token pyramid transformer for mobile semantic segmentation","author":"Zhang","year":"2022"},{"key":"ref71","series-title":"IEEE Conf. Comput. Vis. Pattern Recognit. (CVPR)","first-page":"4510","article-title":"MobileNetV2: Inverted residuals and linear bottlenecks","author":"Sandler","year":"2018"},{"key":"ref72","series-title":"IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR)","first-page":"12084","article-title":"Multi-scale high-resolution vision transformer for semantic segmentation","author":"Gu","year":"2022"},{"key":"ref73","unstructured":"H. Yan, C. Zhang, and M. Wu, \u201cLawin transformer: Improving semantic segmentation transformer with multi-scale representations via large window attention,\u201d arXiv preprint arXiv:2201.01615, 2022."},{"key":"ref74","unstructured":"G. Sharir, A. Noy, and L. Zelnik-Manor, \u201cAn image is worth 16 \u00d7 16 words, what is a video worth?,\u201d arXiv preprint arXiv:2103.13915, 2021."},{"key":"ref75","series-title":"IEEE\/CVF Int. Conf. Comput. Vis. (ICCV)","first-page":"1728","article-title":"Frozen in time: A joint video and image encoder for end-to-end retrieval","author":"Bain","year":"2021"},{"key":"ref76","series-title":"Proc. Int. Conf. Mach. Learn. (ICML)","article-title":"Is space-time attention all you need for video understanding?","volume":"2","author":"Bertasius","year":"2021"},{"key":"ref77","series-title":"29th ACM Int. Conf. Multimed.","first-page":"917","article-title":"Token shift transformer for video classification","author":"Zhang","year":"2021"},{"key":"ref78","series-title":"IEEE\/CVF Int. Conf. Comput. Vis. (ICCV)","first-page":"13557","article-title":"VidTr: Video transformer without convolutions","volume":"696","author":"Zhang","year":"2021"},{"key":"ref79","series-title":"IEEE\/CVF Int. Conf. Comput. Vis. (ICCV)","first-page":"6816","article-title":"ViViT: A video vision transformer","author":"Arnab","year":"2021"},{"key":"ref80","series-title":"Adv. Neural Inf. Process. Syst.","first-page":"19594","article-title":"Space-time mixing attention for video transformer","volume":"34","author":"Bulat","year":"2021"},{"key":"ref81","series-title":"IEEE\/CVF Int. Conf. Comput. Visi. (ICCV)","first-page":"7082","article-title":"TSM: Temporal shift module for efficient video understanding","author":"Lin","year":"2019"},{"key":"ref82","unstructured":"Z. Liu et al., \u201cConvTransformer: A convolutional transformer network for video frame synthesis,\u201d arXiv preprint arXiv:2011.10185, 2011."},{"key":"ref83","series-title":"IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR)","first-page":"8737","article-title":"End-to-End video instance segmentation with transformers","author":"Wang","year":"2021"},{"key":"ref84","series-title":"IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR)","first-page":"2885","article-title":"Temporally efficient vision transformer for video instance segmentation","author":"Yang","year":"2022"},{"key":"ref85","series-title":"Adv. Neural Inf. Process. Syst.","first-page":"13352","article-title":"Video instance segmentation using inter-frame communication transformers","volume":"34","author":"Hwang","year":"2021"},{"key":"ref86","series-title":"IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR)","first-page":"3323","article-title":"Multiview transformers for video recognition","author":"Yan","year":"2022"},{"key":"ref87","series-title":"2021 IEEE\/CVF Int. Conf. Comput. Vis. Workshops (ICCVW)","first-page":"3156","article-title":"Video transformer network","author":"Neimark","year":"2021"},{"key":"ref88","unstructured":"I. Beltagy, M. E. Peters, and A. Cohan, \u201cLongformer: The long-document transformer,\u201d arXiv preprint arXiv:2004.05150, 2004."},{"key":"ref89","series-title":"IEEE\/CVF Int. Conf. Comput. Vis. (ICCV)","first-page":"13485","article-title":"Anticipative video transformer","author":"Girdhar","year":"2021"},{"key":"ref90","series-title":"IEEE\/CVF Int. Conf. Comput. Vis. (ICCV)","first-page":"6804","article-title":"Multiscale vision transformers","author":"Fan","year":"2021"},{"key":"ref91","series-title":"IEEE\/CVF Int. Conf. Comput. Vis. (ICCV)","first-page":"2563","article-title":"Event-based video reconstruction using transformer","author":"Weng","year":"2021"},{"key":"ref92","doi-asserted-by":"crossref","first-page":"359","DOI":"10.1109\/TBC.2022.3147145","article-title":"Cross-frame transformer-based spatiotemporal video super-resolution","volume":"68","author":"Zhang","year":"2022","journal-title":"IEEE Trans. Broadcast."},{"key":"ref93","series-title":"IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR)","first-page":"17420","article-title":"RSTT: Real-time spatial temporal transformer for space-time video super-resolution","author":"Geng","year":"2022"},{"key":"ref94","unstructured":"R. Liu et al., \u201cDecoupled spatial-temporal transformer for video inpainting,\u201d arXiv preprint arXiv:2104.06637, 2021."},{"key":"ref95","doi-asserted-by":"crossref","first-page":"160","DOI":"10.1109\/TCSVT.2022.3201045","article-title":"VDTR: Video deblurring with transformer","volume":"33","author":"Cao","year":"2023","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"ref96","series-title":"Adv. Neural Inf Process. Syst.","first-page":"6840","article-title":"Denoising diffusion probabilistic models","volume":"33","author":"Ho","year":"2020"},{"key":"ref97","series-title":"IEEE\/CVF Int. Conf. Comput. Vis. (ICCV)","first-page":"4172","article-title":"Scalable diffusion models with transformers","author":"Peebles","year":"2023"},{"key":"ref98","first-page":"153113","article-title":"Swinv2-Imagen: Hierarchical vision transformer diffusion models for text-to-image generation","volume":"8","author":"Li","year":"2023","journal-title":"Neural Comput. Appl."},{"key":"ref99","series-title":"Int. Conf. Mach. Learn.","first-page":"1692","article-title":"One transformer fits all distributions in multi-modal diffusion at scale","author":"Bao","year":"2023"},{"key":"ref100","doi-asserted-by":"crossref","first-page":"102568","DOI":"10.1016\/j.displa.2023.102568","article-title":"ET-DM: Text to image via diffusion model with efficient Transformer","volume":"80","author":"Li","year":"2023","journal-title":"Displays"},{"key":"ref101","series-title":"The twelfth Int. Conf. Learn. Represent.","article-title":"PixArt-\u03b1: Fast training of diffusion transformer for photorealistic text-to-image synthesis","author":"Chen","year":"2024"},{"key":"ref102","unstructured":"J. Chen et al., \u201cPIXART-\u03b4: Fast and controllable image generation with latent consistency models,\u201d arXiv preprint arXiv:2401.05252, 2024."},{"key":"ref103","series-title":"IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR)","first-page":"18349","article-title":"LayoutDM: Transformer-based diffusion model for layout generation","author":"Chai","year":"2023"},{"key":"ref104","unstructured":"H. Ali, S. Jiaming, L. Guilin, K. Jan, and V. Arash, \u201cDiffiT: Diffusion vision transformers for image generation,\u201d arXiv preprint arXiv:2312.02139, 2023."},{"key":"ref105","series-title":"IEEE\/CVF Int. Conf. Comput. Vis. (ICCV)","first-page":"23107","article-title":"Masked diffusion transformer is a strong image synthesizer","author":"Gao","year":"2023"},{"key":"ref106","series-title":"2nd Workshop on Lang. Robot Learn.: Lang. Ground.","article-title":"Multimodal diffusion transformer for learning from play","author":"Reuss","year":"2023"},{"key":"ref107","series-title":"Medical Image Comput. Comput. Assisted Interven.\u2013MICCAI 2023","first-page":"622","article-title":"Diffusion transformer U-Net for medical image segmentation","author":"Chowdary","year":"2023"},{"key":"ref108","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1109\/TGRS.2024.3510693","article-title":"Advancing realistic precipitation nowcasting with a spatiotemporal transformer-based denoising diffusion model","volume":"62","author":"Zhao","year":"2024","journal-title":"IEEE Trans. Geosci. Remote Sens."},{"key":"ref109","unstructured":"OpenAI, \u201cSora: Creating video from text,\u201d 2024. Accessed: Apr. 29, 2024. [Online]. Available: https:\/\/openai.com\/sora."},{"key":"ref110","series-title":"IEEE\/CVF Int. Conf. Comput. Vis. (ICCV)","first-page":"538","article-title":"Tokens-to-Token ViT: Training vision transformers from scratch on imagenet","author":"Yuan","year":"2021"},{"key":"ref111","series-title":"IEEE\/CVF Int. Conf. Comput. Vis. (ICCV)","first-page":"9961","article-title":"Co-scale conv-attentional image transformers","author":"Xu","year":"2021"},{"key":"ref112","doi-asserted-by":"crossref","first-page":"10990","DOI":"10.1109\/JSTARS.2021.3119654","article-title":"STransFuse: Fusing swin transformer and convolutional neural network for remote sensing image semantic segmentation","volume":"14","author":"Gao","year":"2021","journal-title":"IEEE J. Sel. Top. Appl. Earth Obs. Remote Sens."}],"container-title":["Computers, Materials &amp; Continua"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.techscience.com\/files\/cmc\/2024\/TSP_CMC-80-1\/TSP_CMC_50790\/TSP_CMC_50790.pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,3,6]],"date-time":"2025-03-06T11:34:02Z","timestamp":1741260842000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.techscience.com\/cmc\/v80n1\/57376"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"references-count":112,"journal-issue":{"issue":"1","published-online":{"date-parts":[[2024]]},"published-print":{"date-parts":[[2024]]}},"URL":"https:\/\/doi.org\/10.32604\/cmc.2024.050790","relation":{},"ISSN":["1546-2226"],"issn-type":[{"value":"1546-2226","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024]]},"assertion":[{"value":"2024-02-17","order":0,"name":"received","label":"Received","group":{"name":"publication_history","label":"Publication History"}},{"value":"2024-05-17","order":1,"name":"accepted","label":"Accepted","group":{"name":"publication_history","label":"Publication History"}},{"value":"2024-07-18","order":2,"name":"published","label":"Published Online","group":{"name":"publication_history","label":"Publication History"}}]}}