{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,22]],"date-time":"2026-06-22T18:17:36Z","timestamp":1782152256157,"version":"3.54.5"},"reference-count":88,"publisher":"Tsinghua University Press","issue":"2","funder":[{"DOI":"10.13039\/100004318","name":"Microsoft Corporation","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100004318","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Comp. Visual. Med."],"published-print":{"date-parts":[[2025,4]]},"DOI":"10.26599\/cvm.2025.9450377","type":"journal-article","created":{"date-parts":[[2025,4,9]],"date-time":"2025-04-09T17:53:35Z","timestamp":1744221215000},"page":"405-422","source":"Crossref","is-referenced-by-count":2,"title":["Text to Image Generation with Bidirectional Multiway Transformers"],"prefix":"10.26599","volume":"11","author":[{"given":"Hangbo","family":"Bao","sequence":"first","affiliation":[{"name":"Harbin Institute of Technology,Department of Computer Science and Technology,Harbin,China,150001"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Li","family":"Dong","sequence":"additional","affiliation":[{"name":"Microsoft Research Asia,Beijing,China,100080"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Songhao","family":"Piao","sequence":"additional","affiliation":[{"name":"Harbin Institute of Technology,Department of Computer Science and Technology,Harbin,China,150001"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Furu","family":"Wei","sequence":"additional","affiliation":[{"name":"Microsoft Research Asia,Beijing,China,100080"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"11138","reference":[{"key":"ref1","volume-title":"Zero-shot text-to-image generation","author":"Ramesh","year":"2021"},{"key":"ref2","first-page":"19822","article-title":"CoView: Mastering text-to-image generation via transformers","volume-title":"Proceedings of the 35th Conference on Neural Information Processing Systems","author":"Ding","year":"2021"},{"key":"ref3","volume-title":"Scaling autoregressive models for content-rich text-to-image generation","author":"Yu","year":"2022"},{"key":"ref4","volume-title":"Muse: Text-to-image generation via masked generative transformers","author":"Chang","year":"2023"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1145\/3422622"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00453"},{"key":"ref7","volume-title":"arge scale GAN training for high fidelity natural image synthesis","author":"Brock","year":"2018"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"ref9","volume-title":"Deep unsupervised learning using nonequilibrium thermodynamics","author":"Sohl-Dickstein","year":"2015"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02235"},{"key":"ref12","volume-title":"Draft-and-revise: Effective image generation with contextual RQ-transformer","author":"Lee","year":"2022"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.18653\/vl\/N19-142"},{"key":"ref14","volume-title":"CoBIT: A contrastive bi-directional image-text generation model","author":"You","year":"2023"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01838"},{"key":"ref16","volume-title":"VLMo: Unified vision-language pre-training with mixture-of-modality-experts","author":"Bao","year":"2021"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.670"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1644"},{"key":"ref19","volume-title":"An image is worth 16\u00d716 words: Transformers for image recognition at scale","author":"Dosovitskiy","year":"2020"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01268"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"ref22","first-page":"6629","article-title":"GANs trained by a two time-scale update rule converge to a local Nash equilibrium","volume-title":"Proceedings of the 31st International Conference on Neural Information Processing Systems","author":"Heusel","year":"2017"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1238"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00089"},{"key":"ref25","volume-title":"GLIDE: Towards photorealistic image generation and editing with text-guided diffusion models","author":"Nichol","year":"2021"},{"key":"ref26","volume-title":"Vector-quantized image modeling with improved VQGAN","author":"Yu","year":"2021"},{"key":"ref27","volume-title":"Discrete variational autoencoders","author":"Rolfe","year":"2017"},{"key":"ref28","volume-title":"Improving language understanding by generative pre-training","author":"Radford","year":"2018"},{"key":"ref29","volume-title":"BEiT: BERT pre-training of image transformers","author":"Bao","year":"2021"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/d16-1264"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0816-y"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-018-1140-0"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1603.08155"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00068"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.632"},{"key":"ref36","volume-title":"Categorical reparameterization with Gumbel-Softmax","author":"Jang","year":"2016"},{"key":"ref37","volume-title":"The concrete distribution: A continuous relaxation of discrete random variables","author":"Maddison","year":"2016"},{"key":"ref38","volume-title":"wav2vec 2.0: A framework for self-supervised learning of speech representations","author":"Baevski","year":"2020"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/d18-2012"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P16-1162"},{"key":"ref41","volume-title":"Hierarchical text-conditional image generation with CLIP latents","author":"Ramesh","year":"2022"},{"key":"ref42","volume-title":"Photorealistic text-to- image diffusion models with deep language understanding","author":"Saharia","year":"2022"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00143"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52688.2022.01738"},{"key":"ref45","volume-title":"OFA: Unifying architectures, tasks, and modalities through a simple sequence-to-sequence learning framework","author":"Wang","year":"2022"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19784-0_6"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01103"},{"key":"ref48","volume-title":"Non-autoregressive neural machine translation","author":"Gu","year":"2017"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1633"},{"key":"ref50","volume-title":"Auto-encoding variational Bayes","author":"Kingma","year":"2013"},{"key":"ref51","first-page":"6309","article-title":"Neural discrete representation learning","volume-title":"Proceedings of the 31st International Conference on Neural Information Processing Systems","author":"van den Oord","year":"2017"},{"key":"ref52","article-title":"Generating diverse high-fidelity images with VQ-VAE-2","volume-title":"Proceedings of the 33rd Conference on Neural Information Processing Systems","author":"Razavi","year":"2019"},{"key":"ref53","volume-title":"ImageBART: Bidirectional context with multinomial diffusion for autoregressive image synthesis","author":"Esser","year":"2021"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01123"},{"key":"ref55","volume-title":"Adam: A method for stochastic optimization","author":"Kingma","year":"2015"},{"issue":"1","key":"ref56","first-page":"1929","article-title":"Dropout: A simple way to prevent neural networks from overfitting","volume":"15","author":"Murphy","year":"2014","journal-title":"Journal of Machine Learning Research"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46493-0_39"},{"key":"ref58","volume-title":"BEiT v2: Masked image modeling with vector-quantized visual tokenizers","author":"Peng","year":"2022"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20056-4_20"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01426"},{"key":"ref61","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"Proceedings of the 38th International Conference on Machine Learning","author":"Radford","year":"2021"},{"key":"ref62","volume-title":"ELECTRA: Pre-training text encoders as discriminators rather than generators","author":"Clark","year":"2020"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00950"},{"key":"ref66","volume-title":"Training data-efficient image transformers & distillation through attention","author":"Touvron","year":"2020"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.467"},{"key":"ref68","article-title":"Improved techniques for training GANs","volume-title":"Proceedings of the 30th Conference on Neural Information Processing Systems","author":"Salimans","year":"2016"},{"key":"ref69","first-page":"4904","article-title":"Scaling up visual and vision-language representation learning with noisy text supervision","volume-title":"Proceedings of the 38th International Conference on Machine Learning","author":"Jia","year":"2021"},{"key":"ref70","volume-title":"LAION-400M: Open dataset of CLIP-filtered 400 million image-text pairs","author":"Schuhmann","year":"2021"},{"key":"ref71","first-page":"3104","article-title":"Sequence to sequence learning with neural networks","volume-title":"Proceedings of the 28th International Conference on Neural Information Processing System","author":"Sutskever","year":"2014"},{"key":"ref72","volume-title":"Auto-encoding variational bayes","author":"Kingma","year":"2013"},{"key":"ref73","first-page":"1691","article-title":"Generative pretraining from pixels","volume-title":"Proceedings of the 37th International Conference on Machine Learning","author":"Chen","year":"2020"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2016.2577031"},{"key":"ref75","volume-title":"Pixel-BERT: Aligning image pixels with text by deep multi-modal transformers","author":"Huang","year":"2020"},{"key":"ref76","first-page":"5583","article-title":"ViLT: Vision-and-language transformer without convolution or region supervision","volume-title":"Proceedings of the 38th International Conference on Machine Learning","author":"Kim","year":"2021"},{"key":"ref77","volume-title":"Align before fuse: Vision and language representation learning with momentum distillation","author":"Li","year":"2021"},{"key":"ref78","volume-title":"CoCa: Contrastive captioners are image-text foundation models","author":"Yu","year":"2022"},{"key":"ref79","volume-title":"Multi-grained vision language pre-training: Aligning texts with visual concepts","author":"Zeng","year":"2021"},{"key":"ref80","first-page":"255","article-title":"Convolutional networks for images, speech, and time series","volume-title":"The hand book of brain theory and neural networks","author":"Cun","year":"2022"},{"key":"ref81","volume-title":"A simple framework for contrastive learning of visual representations","author":"Chen","year":"2020"},{"key":"ref82","doi-asserted-by":"publisher","DOI":"10.1007\/s41095-022-0274-8"},{"key":"ref83","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"ref84","doi-asserted-by":"publisher","DOI":"10.1007\/s41095-022-0271-y"},{"key":"ref85","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00943"},{"key":"ref86","volume-title":"iBOT: Image BERT pre-training with online tokenizer","author":"Zhou","year":"2021"},{"key":"ref87","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W18-5446"},{"key":"ref88","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00604"}],"container-title":["Computational Visual Media"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10750449\/10985750\/10960474.pdf?arnumber=10960474","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,8]],"date-time":"2025-05-08T04:28:17Z","timestamp":1746678497000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10960474\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,4]]},"references-count":88,"journal-issue":{"issue":"2"},"URL":"https:\/\/doi.org\/10.26599\/cvm.2025.9450377","relation":{},"ISSN":["2096-0662","2096-0433"],"issn-type":[{"value":"2096-0662","type":"electronic"},{"value":"2096-0433","type":"print"}],"subject":[],"published":{"date-parts":[[2025,4]]}}}