{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T06:58:16Z","timestamp":1784617096586,"version":"3.55.0"},"publisher-location":"Cham","reference-count":89,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031732416","type":"print"},{"value":"9783031732423","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,10,29]],"date-time":"2024-10-29T00:00:00Z","timestamp":1730160000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,29]],"date-time":"2024-10-29T00:00:00Z","timestamp":1730160000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-73242-3_3","type":"book-chapter","created":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T09:15:43Z","timestamp":1730106943000},"page":"37-55","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":45,"title":["DiffiT: Diffusion Vision Transformers for Image Generation"],"prefix":"10.1007","author":[{"given":"Ali","family":"Hatamizadeh","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiaming","family":"Song","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Guilin","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jan","family":"Kautz","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Arash","family":"Vahdat","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,10,29]]},"reference":[{"key":"3_CR1","unstructured":"Albergo, M.S., Boffi, N.M., Vanden-Eijnden, E.: Stochastic interpolants: a unifying framework for flows and diffusions. arXiv preprint arXiv:2303.08797 (2023)"},{"key":"3_CR2","unstructured":"Albergo, M.S., Vanden-Eijnden, E.: Building normalizing flows with stochastic interpolants. arXiv preprint arXiv:2209.15571 (2022)"},{"issue":"3","key":"3_CR3","doi-asserted-by":"publisher","first-page":"313","DOI":"10.1016\/0304-4149(82)90051-5","volume":"12","author":"BD Anderson","year":"1982","unstructured":"Anderson, B.D.: Reverse-time diffusion equation models. Stoch. Process. Appl. 12(3), 313\u2013326 (1982)","journal-title":"Stoch. Process. Appl."},{"key":"3_CR4","unstructured":"Aneja, J., Schwing, A., Kautz, J., Vahdat, A.: A contrastive learning approach for training variational autoencoder priors. In: Advances in Neural Information Processing Systems, vol. 34, pp. 480\u2013493 (2021)"},{"issue":"4","key":"3_CR5","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3592450","volume":"42","author":"O Avrahami","year":"2023","unstructured":"Avrahami, O., Fried, O., Lischinski, D.: Blended latent diffusion. ACM Trans. Graph. (TOG) 42(4), 1\u201311 (2023)","journal-title":"ACM Trans. Graph. (TOG)"},{"key":"3_CR6","doi-asserted-by":"crossref","unstructured":"Avrahami, O., Lischinski, D., Fried, O.: Blended diffusion for text-driven editing of natural images. In: Proceedings of the CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.01767"},{"key":"3_CR7","unstructured":"Ba, J.L., Kiros, J.R., Hinton, G.E.: Layer normalization. arXiv preprint arXiv:1607.06450 (2016)"},{"key":"3_CR8","unstructured":"Balaji, Y., et al.: eDiff-I: text-to-image diffusion models with an ensemble of expert denoisers. arXiv preprint arXiv:2211.01324 (2022)"},{"key":"3_CR9","doi-asserted-by":"crossref","unstructured":"Bao, F., Li, C., Cao, Y., Zhu, J.: All are worth words: a ViT backbone for score-based diffusion models. In: NeurIPS 2022 Workshop on Score-Based Methods (2022)","DOI":"10.1109\/CVPR52729.2023.02171"},{"key":"3_CR10","unstructured":"Brock, A., Donahue, J., Simonyan, K.: Large scale GAN training for high fidelity natural image synthesis. arXiv preprint arXiv:1809.11096 (2018)"},{"key":"3_CR11","doi-asserted-by":"crossref","unstructured":"Chang, H., Zhang, H., Jiang, L., Liu, C., Freeman, W.T.: MaskGIT: masked generative image transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11315\u201311325 (2022)","DOI":"10.1109\/CVPR52688.2022.01103"},{"key":"3_CR12","unstructured":"Chen, M., et al.: Generative pretraining from pixels. In: International Conference on Machine Learning, pp. 1691\u20131703. PMLR (2020)"},{"key":"3_CR13","doi-asserted-by":"crossref","unstructured":"Choi, J., Lee, J., Shin, C., Kim, S., Kim, H., Yoon, S.: Perception prioritized training of diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11472\u201311481 (2022)","DOI":"10.1109\/CVPR52688.2022.01118"},{"key":"3_CR14","unstructured":"Couairon, G., Verbeek, J., Schwenk, H., Cord, M.: DiffEdit: diffusion-based semantic image editing with mask guidance. arXiv preprint arXiv:2210.11427 (2022)"},{"key":"3_CR15","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.J., Li, K., Fei-Fei, L.: ImageNet: a large-scale hierarchical image database. In: 2009 IEEE Conference on Computer Vision and Pattern Recognition, pp. 248\u2013255. IEEE (2009)","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"3_CR16","unstructured":"Dhariwal, P., Nichol, A.: Diffusion models beat GANs on image synthesis. In: Advances in Neural Information Processing Systems, vol. 34, pp. 8780\u20138794 (2021)"},{"key":"3_CR17","unstructured":"Ding, M., et al.: CogView: mastering text-to-image generation via transformers. In: Advances in Neural Information Processing Systems, vol. 34, pp. 19822\u201319835 (2021)"},{"key":"3_CR18","unstructured":"Ding, M., Zheng, W., Hong, W., Tang, J.: CogView2: faster and better text-to-image generation via hierarchical transformers. In: Advances in Neural Information Processing Systems, vol. 35, pp. 16890\u201316902 (2022)"},{"key":"3_CR19","unstructured":"Dockhorn, T., Vahdat, A., Kreis, K.: Score-based generative modeling with critically-damped langevin diffusion. arXiv preprint arXiv:2112.07068 (2021)"},{"key":"3_CR20","unstructured":"Dosovitskiy, A., et al.: An image is worth $$16\\times 16$$ words: transformers for image recognition at scale. In: International Conference on Learning Representations (2020)"},{"key":"3_CR21","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1016\/j.neunet.2017.12.012","volume":"107","author":"S Elfwing","year":"2018","unstructured":"Elfwing, S., Uchibe, E., Doya, K.: Sigmoid-weighted linear units for neural network function approximation in reinforcement learning. Neural Netw. 107, 3\u201311 (2018)","journal-title":"Neural Netw."},{"key":"3_CR22","unstructured":"Gal, R., et al.: An image is worth one word: personalizing text-to-image generation using textual inversion. arXiv preprint arXiv:2208.01618 (2022)"},{"key":"3_CR23","doi-asserted-by":"crossref","unstructured":"Gao, S., Zhou, P., Cheng, M., Yan, S.: Masked diffusion transformer is a strong image synthesizer. In: 2023 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 23107\u201323116 (2023)","DOI":"10.1109\/ICCV51070.2023.02117"},{"key":"3_CR24","doi-asserted-by":"crossref","unstructured":"Gong, X., Chang, S., Jiang, Y., Wang, Z.: AutoGAN: neural architecture search for generative adversarial networks. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3224\u20133234 (2019)","DOI":"10.1109\/ICCV.2019.00332"},{"key":"3_CR25","unstructured":"Goodfellow, I., et al.: Generative adversarial nets. In: Advances in Neural Information Processing Systems, pp. 2672\u20132680 (2014)"},{"issue":"4","key":"3_CR26","doi-asserted-by":"publisher","first-page":"549","DOI":"10.1111\/j.2517-6161.1994.tb02000.x","volume":"56","author":"U Grenander","year":"1994","unstructured":"Grenander, U., Miller, M.I.: Representations of knowledge in complex systems. J. Roy. Stat. Soc.: Ser. B (Methodol.) 56(4), 549\u2013581 (1994)","journal-title":"J. Roy. Stat. Soc.: Ser. B (Methodol.)"},{"key":"3_CR27","unstructured":"Ho, J., et al.: Imagen video: high definition video generation with diffusion models. arXiv preprint arXiv:2210.02303 (2022)"},{"key":"3_CR28","unstructured":"Ho, J., Jain, A., Abbeel, P.: Denoising diffusion probabilistic models. arXiv preprint arXiv:2006.11239 (2020)"},{"key":"3_CR29","unstructured":"Hong, W., Ding, M., Zheng, W., Liu, X., Tang, J.: CogVideo: large-scale pretraining for text-to-video generation via transformers. arXiv preprint arXiv:2205.15868 (2022)"},{"key":"3_CR30","unstructured":"Hoogeboom, E., Heek, J., Salimans, T.: Simple diffusion: end-to-end diffusion for high resolution images. arXiv preprint arXiv:2301.11093 (2023)"},{"key":"3_CR31","unstructured":"Hudson, D.A., Zitnick, L.: Generative adversarial transformers. In: International Conference on Machine Learning, pp. 4487\u20134499. PMLR (2021)"},{"issue":"24","key":"3_CR32","first-page":"695","volume":"6","author":"A Hyv\u00e4rinen","year":"2005","unstructured":"Hyv\u00e4rinen, A.: Estimation of non-normalized statistical models by score matching. JMLR 6(24), 695\u2013709 (2005)","journal-title":"JMLR"},{"key":"3_CR33","unstructured":"Jiang, Y., Chang, S., Wang, Z.: TransGAN: two transformers can make one strong GAN. arXiv preprint arXiv:2102.07074 (2021)"},{"key":"3_CR34","unstructured":"Karras, T., Aittala, M., Aila, T., Laine, S.: Elucidating the design space of diffusion-based generative models. In: Proceedings of NeurIPS (2022)"},{"key":"3_CR35","unstructured":"Karras, T., Aittala, M., Hellsten, J., Laine, S., Lehtinen, J., Aila, T.: Training generative adversarial networks with limited data. In: Advances in Neural Information Processing Systems, vol. 33, pp. 12104\u201312114 (2020)"},{"key":"3_CR36","doi-asserted-by":"crossref","unstructured":"Karras, T., Laine, S., Aila, T.: A style-based generator architecture for generative adversarial networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4401\u20134410 (2019)","DOI":"10.1109\/CVPR.2019.00453"},{"key":"3_CR37","unstructured":"Kawar, B., Elad, M., Ermon, S., Song, J.: Denoising diffusion restoration models. arXiv preprint arXiv:2201.11793 (2022)"},{"key":"3_CR38","unstructured":"Kawar, B., Song, J., Ermon, S., Elad, M.: JPEG artifact correction using denoising diffusion restoration models. arXiv preprint arXiv:2209.11888 (2022)"},{"key":"3_CR39","doi-asserted-by":"crossref","unstructured":"Kawar, B., et al.: Imagic: text-based real image editing with diffusion models. arXiv preprint arXiv:2210.09276 (2022)","DOI":"10.1109\/CVPR52729.2023.00582"},{"key":"3_CR40","unstructured":"Kim, D., Na, B., Kwon, S.J., Lee, D., Kang, W., Moon, I.C.: Maximum likelihood training of implicit nonlinear diffusion models. arXiv preprint arXiv:2205.13699 (2022)"},{"key":"3_CR41","unstructured":"Kingma, D.P., Ba, J.: Adam: a method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)"},{"key":"3_CR42","unstructured":"Kingma, D.P., Welling, M.: Auto-encoding variational bayes. arXiv preprint arXiv:1312.6114v10 (2013)"},{"key":"3_CR43","unstructured":"Kong, Z., Ping, W., Huang, J., Zhao, K., Catanzaro, B.: DiffWave: a versatile diffusion model for audio synthesis. arXiv preprint arXiv:2009.09761 (2020)"},{"key":"3_CR44","unstructured":"Kreis, K., Gao, R., Vahdat, A.: CVPR tutorial on denoising diffusion-based generative modeling: foundations and applications (2022). https:\/\/cvpr2022-tutorial-diffusion-models.github.io\/"},{"key":"3_CR45","unstructured":"Krizhevsky, A., Hinton, G., et\u00a0al.: Learning multiple layers of features from tiny images (2009)"},{"key":"3_CR46","doi-asserted-by":"crossref","unstructured":"Lee, D., Kim, C., Kim, S., Cho, M., Han, W.S.: Autoregressive image generation using residual quantization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11523\u201311532 (2022)","DOI":"10.1109\/CVPR52688.2022.01123"},{"key":"3_CR47","unstructured":"Lee, K., Chang, H., Jiang, L., Zhang, H., Tu, Z., Liu, C.: ViTGAN: training GANs with vision transformers. arXiv preprint arXiv:2107.04589 (2021)"},{"key":"3_CR48","unstructured":"Li, S., Chen, X., He, D., Hsieh, C.J.: Can vision transformers perform convolution? arXiv preprint arXiv:2111.01353 (2021)"},{"key":"3_CR49","unstructured":"Li, X.L., Thickstun, J., Gulrajani, I., Liang, P., Hashimoto, T.B.: Diffusion-LM improves controllable text generation. arXiv preprint arXiv:2205.14217 (2022)"},{"key":"3_CR50","unstructured":"Luhman, T., Luhman, E.: Improving diffusion model efficiency through patching. arXiv preprint arXiv:2207.04316 (2022)"},{"key":"3_CR51","doi-asserted-by":"crossref","unstructured":"Ma, N., Goldstein, M., Albergo, M.S., Boffi, N.M., Vanden-Eijnden, E., Xie, S.: SiT: exploring flow and diffusion-based generative models with scalable interpolant transformers. arXiv preprint arXiv:2401.08740 (2024)","DOI":"10.1007\/978-3-031-72980-5_2"},{"key":"3_CR52","unstructured":"Meng, C., et al.: SDEdit: guided image synthesis and editing with stochastic differential equations. arXiv preprint arXiv:2108.01073 (2021)"},{"key":"3_CR53","unstructured":"Nichol, A.Q., Dhariwal, P.: Improved denoising diffusion probabilistic models. In: International Conference on Machine Learning, pp. 8162\u20138171. PMLR (2021)"},{"key":"3_CR54","unstructured":"Nie, W., Guo, B., Huang, Y., Xiao, C., Vahdat, A., Anandkumar, A.: Diffusion models for adversarial purification. In: Proceedings of ICML (2022)"},{"key":"3_CR55","doi-asserted-by":"crossref","unstructured":"Park, J., Kim, Y.: Styleformer: transformer based generative adversarial networks with style vector. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8983\u20138992 (2022)","DOI":"10.1109\/CVPR52688.2022.00878"},{"key":"3_CR56","doi-asserted-by":"crossref","unstructured":"Peebles, W., Xie, S.: Scalable diffusion models with transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 4195\u20134205 (2023)","DOI":"10.1109\/ICCV51070.2023.00387"},{"key":"3_CR57","doi-asserted-by":"crossref","unstructured":"Perez, E., Strub, F., De\u00a0Vries, H., Dumoulin, V., Courville, A.: FiLM: visual reasoning with a general conditioning layer. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a032 (2018)","DOI":"10.1609\/aaai.v32i1.11671"},{"key":"3_CR58","unstructured":"Ramesh, A., Dhariwal, P., Nichol, A., Chu, C., Chen, M.: Hierarchical text-conditional image generation with CLIP latents. arXiv preprint arXiv:2204.06125 (2022)"},{"key":"3_CR59","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., Ommer, B.: High-resolution image synthesis with latent diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10684\u201310695 (2022)","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"3_CR60","unstructured":"Rombach, R., Esser, P.: Stable diffusion v1-4 (2022). https:\/\/huggingface.co\/CompVis\/stable-diffusion-v1-4"},{"key":"3_CR61","doi-asserted-by":"crossref","unstructured":"Ronneberger, O., Fischer, P., Brox, T.: U-Net: convolutional networks for biomedical image segmentation. arXiv preprint arXiv:1505.04597 (2015)","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"3_CR62","doi-asserted-by":"crossref","unstructured":"Ruiz, N., Li, Y., Jampani, V., Pritch, Y., Rubinstein, M., Aberman, K.: DreamBooth: fine tuning text-to-image diffusion models for subject-driven generation. arXiv preprint arXiv:2208.12242 (2022)","DOI":"10.1109\/CVPR52729.2023.02155"},{"key":"3_CR63","unstructured":"Saharia, C., et al.: Photorealistic text-to-image diffusion models with deep language understanding. arXiv preprint arXiv:2205.11487 (2022)"},{"key":"3_CR64","unstructured":"Salimans, T., Ho, J.: Progressive distillation for fast sampling of diffusion models. arXiv preprint arXiv:2202.00512 (2022)"},{"key":"3_CR65","doi-asserted-by":"crossref","unstructured":"Sauer, A., Schwarz, K., Geiger, A.: StyleGAN-XL: scaling StyleGAN to large diverse datasets. In: ACM SIGGRAPH 2022 Conference Proceedings, pp. 1\u201310 (2022)","DOI":"10.1145\/3528233.3530738"},{"key":"3_CR66","doi-asserted-by":"crossref","unstructured":"Shaw, P., Uszkoreit, J., Vaswani, A.: Self-attention with relative position representations. arXiv preprint arXiv:1803.02155 (2018)","DOI":"10.18653\/v1\/N18-2074"},{"key":"3_CR67","unstructured":"Sinha, A., Song, J., Meng, C., Ermon, S.: D2C: diffusion-decoding models for few-shot conditional generation. In: Advances in Neural Information Processing Systems, vol. 34, pp. 12533\u201312548 (2021)"},{"key":"3_CR68","unstructured":"Sohl-Dickstein, J., Weiss, E.A., Maheswaranathan, N., Ganguli, S.: Deep unsupervised learning using nonequilibrium thermodynamics. arXiv preprint arXiv:1503.03585 (2015)"},{"key":"3_CR69","unstructured":"Song, J., Meng, C., Ermon, S.: Denoising diffusion implicit models. In: International Conference on Learning Representations (2021)"},{"key":"3_CR70","unstructured":"Song, Y., Ermon, S.: Generative modeling by estimating gradients of the data distribution. arXiv preprint arXiv:1907.05600 (2019)"},{"key":"3_CR71","unstructured":"Song, Y., Sohl-Dickstein, J., Kingma, D.P., Kumar, A., Ermon, S., Poole, B.: Score-based generative modeling through stochastic differential equations. In: International Conference on Learning Representations (2021)"},{"key":"3_CR72","unstructured":"Tashiro, Y., Song, J., Song, Y., Ermon, S.: CSDI: conditional score-based diffusion models for probabilistic time series imputation. In: Advances in Neural Information Processing Systems, vol. 34, pp. 24804\u201324816 (2021)"},{"key":"3_CR73","unstructured":"Vahdat, A., Kautz, J.: NVAE: a deep hierarchical variational autoencoder. In: Advances in Neural Information Processing Systems, vol. 33, pp. 19667\u201319679 (2020)"},{"key":"3_CR74","unstructured":"Vahdat, A., Kreis, K., Kautz, J.: Score-based generative modeling in latent space. arXiv preprint arXiv:2106.05931 (2021)"},{"key":"3_CR75","doi-asserted-by":"crossref","unstructured":"Valevski, D., Kalman, M., Matias, Y., Leviathan, Y.: UniTune: text-driven image editing by fine tuning an image generation model on a single image. arXiv preprint arXiv:2210.09477 (2022)","DOI":"10.1145\/3592451"},{"issue":"7","key":"3_CR76","doi-asserted-by":"publisher","first-page":"1661","DOI":"10.1162\/NECO_a_00142","volume":"23","author":"P Vincent","year":"2011","unstructured":"Vincent, P.: A connection between score matching and denoising autoencoders. Neural Comput. 23(7), 1661\u20131674 (2011)","journal-title":"Neural Comput."},{"key":"3_CR77","unstructured":"Wang, S., Li, B., Khabsa, M., Fang, H., Ma, H.: Linformer: self-attention with linear complexity. arXiv preprint arXiv:2006.04768 (2020)"},{"key":"3_CR78","unstructured":"Wu, Y., He, K.: Group normalization. arXiv preprint arXiv:1803.08494 (2018)"},{"key":"3_CR79","unstructured":"Xu, M., Yu, L., Song, Y., Shi, C., Ermon, S., Tang, J.: GeoDiff: a geometric diffusion model for molecular conformation generation. In: Proceedings of ICLR (2022)"},{"key":"3_CR80","unstructured":"Xu, R., Xu, X., Chen, K., Zhou, B., Loy, C.C.: STransGAN: an empirical study on transformer in GANs. arXiv preprint arXiv:2110.13107 (2021)"},{"key":"3_CR81","unstructured":"Yang, X., Shih, S.M., Fu, Y., Zhao, X., Ji, S.: Your ViT is secretly a hybrid discriminative-generative diffusion model. arXiv preprint arXiv:2208.07791 (2022)"},{"key":"3_CR82","unstructured":"Ye, M., Wu, L., Liu, Q.: First hitting diffusion models. arXiv preprint arXiv:2209.01170 (2022)"},{"key":"3_CR83","unstructured":"Zeng, X., et al.: LION: latent point diffusion models for 3D shape generation. In: Advances in Neural Information Processing Systems (NeurIPS) (2022)"},{"key":"3_CR84","doi-asserted-by":"crossref","unstructured":"Zhang, B., et al.: StyleSwin: transformer-based GAN for high-resolution image generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11304\u201311314 (2022)","DOI":"10.1109\/CVPR52688.2022.01102"},{"key":"3_CR85","unstructured":"Zhang, H., et al.: ERNIE-ViLG: unified generative pre-training for bidirectional vision-language generation. arXiv preprint arXiv:2112.15283 (2021)"},{"key":"3_CR86","unstructured":"Zhang, Q., Tao, M., Chen, Y.: gDDIM: generalized denoising diffusion implicit models. arXiv preprint arXiv:2206.05564 (2022)"},{"key":"3_CR87","unstructured":"Zhao, L., Zhang, Z., Chen, T., Metaxas, D., Zhang, H.: Improved transformer for high-resolution GANs. In: Advances in Neural Information Processing Systems, vol. 34, pp. 18367\u201318380 (2021)"},{"key":"3_CR88","doi-asserted-by":"crossref","unstructured":"Zheng, S., et al.: Rethinking semantic segmentation from a sequence-to-sequence perspective with transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6881\u20136890 (2021)","DOI":"10.1109\/CVPR46437.2021.00681"},{"key":"3_CR89","doi-asserted-by":"crossref","unstructured":"Zhou, L., Du, Y., Wu, J.: 3D shape generation and completion through point-voxel diffusion. arXiv preprint arXiv:2104.03670 (2021)","DOI":"10.1109\/ICCV48922.2021.00577"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-73242-3_3","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,30]],"date-time":"2024-11-30T10:25:15Z","timestamp":1732962315000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-73242-3_3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,29]]},"ISBN":["9783031732416","9783031732423"],"references-count":89,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-73242-3_3","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,29]]},"assertion":[{"value":"29 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}