{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,22]],"date-time":"2026-07-22T16:06:27Z","timestamp":1784736387690,"version":"3.55.0"},"publisher-location":"Cham","reference-count":66,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031729799","type":"print"},{"value":"9783031729805","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024]]},"DOI":"10.1007\/978-3-031-72980-5_2","type":"book-chapter","created":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T09:34:20Z","timestamp":1730108060000},"page":"23-40","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":111,"title":["SiT: Exploring Flow and\u00a0Diffusion-Based Generative Models with\u00a0Scalable Interpolant Transformers"],"prefix":"10.1007","author":[{"given":"Nanye","family":"Ma","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mark","family":"Goldstein","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Michael S.","family":"Albergo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Nicholas M.","family":"Boffi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Eric","family":"Vanden-Eijnden","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Saining","family":"Xie","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,10,29]]},"reference":[{"key":"2_CR1","unstructured":"Albergo, M.S., Boffi, N.M., Lindsey, M., Vanden-Eijnden, E.: Multimarginal generative modeling with stochastic interpolants. arXiv preprint arXiv:2310.03695 (2023)"},{"key":"2_CR2","unstructured":"Albergo, M.S., Boffi, N.M., Vanden-Eijnden, E.: Stochastic interpolants: a unifying framework for flows and diffusions. arXiv preprint arXiv:2303.08797 (2023)"},{"key":"2_CR3","unstructured":"Albergo, M.S., Goldstein, M., Boffi, N.M., Ranganath, R., Vanden-Eijnden, E.: Stochastic interpolants with data-dependent couplings. arXiv preprint arXiv:2310.03725 (2023)"},{"key":"2_CR4","unstructured":"Albergo, M.S., Vanden-Eijnden, E.: Building normalizing flows with stochastic interpolants. In: ICLR (2023)"},{"key":"2_CR5","doi-asserted-by":"publisher","first-page":"313","DOI":"10.1016\/0304-4149(82)90051-5","volume":"12","author":"BD Anderson","year":"1982","unstructured":"Anderson, B.D.: Reverse-time diffusion equation models. Stochast. Process. Appl. 12, 313\u2013326 (1982)","journal-title":"Stochast. Process. Appl."},{"key":"2_CR6","unstructured":"Ben-Hamu, H., et al.: Matching normalizing flows and probability paths on manifolds. In: ICML (2022)"},{"key":"2_CR7","unstructured":"Benton, J., Deligiannidis, G., Doucet, A.: Error bounds for flow matching methods. arXiv preprint arXiv:2305.16860 (2023)"},{"key":"2_CR8","doi-asserted-by":"crossref","unstructured":"Blattmann, A., et al.: Align your latents: High-resolution video synthesis with latent diffusion models. In: CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.02161"},{"key":"2_CR9","doi-asserted-by":"crossref","unstructured":"Boffi, N.M., Vanden-Eijnden, E.: Deep learning probability flows and entropy production rates in active matter. arXiv preprint arXiv:2309.12991 (2023)","DOI":"10.1073\/pnas.2318106121"},{"key":"2_CR10","unstructured":"Brock, A., Donahue, J., Simonyan, K.: Large scale GAN training for high fidelity natural image synthesis. In: ICLR (2019)"},{"key":"2_CR11","doi-asserted-by":"publisher","first-page":"e82819","DOI":"10.7554\/eLife.82819","volume":"12","author":"A Chandra","year":"2023","unstructured":"Chandra, A., T\u00fcnnermann, L., L\u00f6fstedt, T., Gratz, R.: Transformer-based deep learning for predicting protein properties in the life sciences. Elife 12, e82819 (2023)","journal-title":"Elife"},{"key":"2_CR12","doi-asserted-by":"crossref","unstructured":"Chang, H., Zhang, H., Jiang, L., Liu, C., Freeman, W.T.: MaskGIT: masked generative image transformer. In: CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.01103"},{"key":"2_CR13","unstructured":"Chen, H., Lee, H., Lu, J.: Improved analysis of score-based generative modeling: user-friendly bounds under minimal smoothness assumptions. In: ICML (2023)"},{"key":"2_CR14","unstructured":"Chen, S., Chewi, S., Li, J., Li, Y., Salim, A., Zhang, A.: Sampling is as easy as learning the score: theory for diffusion models with minimal data assumptions. In: ICLR (2023)"},{"key":"2_CR15","unstructured":"Chen, S., Daras, G., Dimakis, A.: Restoration-degradation beyond linear diffusions: a non-asymptotic analysis for DDIM-type samplers. In: ICML (2023)"},{"key":"2_CR16","unstructured":"Chen, T.: On the importance of noise scheduling for diffusion models. arXiv preprint arXiv:2301.10972 (2023)"},{"key":"2_CR17","unstructured":"Dao, Q., Phung, H., Nguyen, B., Tran, A.: Flow matching in latent space. arXiv preprint arXiv:2307.08698 (2023)"},{"key":"2_CR18","unstructured":"De\u00a0Bortoli, V., Thornton, J., Heng, J., Doucet, A.: Diffusion Schr\u00f6dinger bridge with applications to score-based generative modeling. In: NeurIPS (2021)"},{"key":"2_CR19","unstructured":"Dhariwal, P., Nichol, A.: Diffusion models beat GANs on image synthesis. In: NIPS (2021)"},{"key":"2_CR20","unstructured":"Dosovitskiy, A., et al.: An image is worth 16$$\\times $$16 words: transformers for image recognition at scale. In: ICLR (2021)"},{"key":"2_CR21","doi-asserted-by":"crossref","unstructured":"Gao, S., Zhou, P., Cheng, M.M., Yan, S.: Masked diffusion transformer is a strong image synthesizer. arXiv preprint arXiv:2303.14389 (2023)","DOI":"10.1109\/ICCV51070.2023.02117"},{"key":"2_CR22","unstructured":"von Glehn, I., Spencer, J.S., Pfau, D.: A self-attention ansatz for Ab-initio quantum chemistry. In: ICLR (2023)"},{"key":"2_CR23","unstructured":"Gupta, A., et al.: Photorealistic video generation with diffusion models. arXiv preprint arXiv:2312.06662 (2023)"},{"key":"2_CR24","unstructured":"Ho, J., Jain, A., Abbeel, P.: Denoising diffusion probabilistic models. In: NeurIPS (2020)"},{"key":"2_CR25","unstructured":"Ho, J., Saharia, C., Chan, W., Fleet, D.J., Norouzi, M., Salimans, T.: Cascaded diffusion models for high fidelity image generation. arXiv preprint arXiv:2106.15282 (2021)"},{"key":"2_CR26","unstructured":"Ho, J., Salimans, T.: Classifier-free diffusion guidance. arXiv preprint arXiv:2207.12598 (2022)"},{"key":"2_CR27","unstructured":"Hoogeboom, E., Heek, J., Salimans, T.: Simple diffusion: end-to-end diffusion for high resolution images. In: ICML (2023)"},{"key":"2_CR28","unstructured":"Hyv\u00e4rinen, A.: Estimation of non-normalized statistical models by score matching. JMLR (2005)"},{"key":"2_CR29","doi-asserted-by":"publisher","first-page":"1739","DOI":"10.1162\/089976699300016214","volume":"11","author":"A Hyv\u00e4rinen","year":"1999","unstructured":"Hyv\u00e4rinen, A.: Sparse code shrinkage: denoising of nongaussian data by maximum likelihood estimation. Neural Comput. 11, 1739\u20131768 (1999)","journal-title":"Neural Comput."},{"key":"2_CR30","unstructured":"Jabri, A., Fleet, D., Chen, T.: Scalable adaptive computation for iterative generation. In: ICML (2023)"},{"key":"2_CR31","doi-asserted-by":"crossref","unstructured":"Jakab, T., Li, R., Wu, S., Rupprecht, C., Vedaldi, A.: Farm3D: learning articulated 3D animals by distilling 2D diffusion. In: 3DV (2024)","DOI":"10.1109\/3DV62453.2024.00051"},{"key":"2_CR32","unstructured":"Karras, T., Aittala, M., Aila, T., Laine, S.: Elucidating the design space of diffusion-based generative models. In: NeurIPS (2022)"},{"key":"2_CR33","unstructured":"Kingma, D.P., Gao, R.: Understanding the diffusion objective as a weighted integral of elbos. arXiv preprint arXiv:2303.00848 (2023)"},{"key":"2_CR34","unstructured":"Kingma, D.P., Salimans, T., Poole, B., Ho, J.: Variational diffusion models. In: NeurIPS (2021)"},{"key":"2_CR35","unstructured":"Lee, H., Lu, J., Tan, Y.: Convergence for score-based generative modeling with polynomial complexity. In: NeurIPS (2022)"},{"key":"2_CR36","unstructured":"Lee, H., Lu, J., Tan, Y.: Convergence of score-based generative modeling for general data distributions. In: ALT (2023)"},{"key":"2_CR37","unstructured":"Lee, S., Kim, B., Ye, J.C.: Minimizing trajectory curvature of ode-based generative models. In: ICML (2023)"},{"key":"2_CR38","unstructured":"Lipman, Y., Chen, R.T.Q., Ben-Hamu, H., Nickel, M., Le, M.: Flow matching for generative modeling. In: ICLR (2023)"},{"key":"2_CR39","doi-asserted-by":"crossref","unstructured":"Liu, R., Wu, R., Hoorick, B.V., Tokmakov, P., Zakharov, S., Vondrick, C.: Zero-1-to-3: zero-shot one image to 3D object. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.00853"},{"key":"2_CR40","unstructured":"Liu, X., Gong, C., Liu, Q.: Flow straight and fast: learning to generate and transfer data with rectified flow. In: ICLR (2023)"},{"key":"2_CR41","unstructured":"Liu, X., Wu, L., Ye, M., Liu, Q.: Let us build bridges: understanding and extending diffusion generative models. arXiv preprint arXiv:2208.14699 (2022)"},{"key":"2_CR42","unstructured":"Meng, C., et al.: SDEdit: guided image synthesis and editing with stochastic differential equations. In: ICLR (2022)"},{"key":"2_CR43","unstructured":"Nichol, A., Dhariwal, P.: Improved denoising diffusion probabilistic models. In: ICML (2021)"},{"key":"2_CR44","unstructured":"Parmar, N., et al.: Image transformer. In: ICML (2018)"},{"key":"2_CR45","doi-asserted-by":"crossref","unstructured":"Peebles, W., Xie, S.: Scalable diffusion models with transformers. In: ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.00387"},{"key":"2_CR46","unstructured":"Peluchetti, S.: Non-denoising forward-time diffusions. In: ICLR (2022)"},{"key":"2_CR47","unstructured":"Pooladian, A.A., Ben-Hamu, H., Domingo-Enrich, C., Amos, B., Lipman, Y., Chen, R.T.Q.: Multisample flow matching: straightening flows with minibatch couplings. In: ICML (2023)"},{"key":"2_CR48","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., Ommer, B.: High-resolution image synthesis with latent diffusion models. In: CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"2_CR49","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"234","DOI":"10.1007\/978-3-319-24574-4_28","volume-title":"Medical Image Computing and Computer-Assisted Intervention \u2013 MICCAI 2015","author":"O Ronneberger","year":"2015","unstructured":"Ronneberger, O., Fischer, P., Brox, T.: U-net: convolutional networks for biomedical image segmentation. In: Navab, N., Hornegger, J., Wells, W.M., Frangi, A.F. (eds.) MICCAI 2015. LNCS, vol. 9351, pp. 234\u2013241. Springer, Cham (2015). https:\/\/doi.org\/10.1007\/978-3-319-24574-4_28"},{"key":"2_CR50","doi-asserted-by":"publisher","first-page":"211","DOI":"10.1007\/s11263-015-0816-y","volume":"115","author":"O Russakovsky","year":"2015","unstructured":"Russakovsky, O., et al.: ImageNet large scale visual recognition challenge. IJCV 115, 211\u2013252 (2015)","journal-title":"IJCV"},{"key":"2_CR51","unstructured":"Salimans, T., Ho, J.: Progressive distillation for fast sampling of diffusion models. In: ICLR (2022)"},{"key":"2_CR52","doi-asserted-by":"crossref","unstructured":"Sauer, A., Schwarz, K., Geiger, A.: StyleGAN-XL: scaling StyleGAN to large diverse datasets. In: SIGGRAPH (2022)","DOI":"10.1145\/3528233.3530738"},{"key":"2_CR53","unstructured":"Shi, Y., Bortoli, V.D., Campbell, A., Doucet, A.: Diffusion Schr\u00f6dinger bridge matching. In: NIPS (2023)"},{"key":"2_CR54","unstructured":"Simoncelli, E.P., Adelson, E.H.: Noise removal via Bayesian wavelet coring. In: ICIP (1996)"},{"key":"2_CR55","unstructured":"Singhal, R., Goldstein, M., Ranganath, R.: Where to diffuse, how to diffuse, and how to get back: automated learning for multivariate diffusions. In: ICLR (2023)"},{"key":"2_CR56","unstructured":"Sohl-Dickstein, J., Weiss, E., Maheswaranathan, N., Ganguli, S.: Deep unsupervised learning using nonequilibrium thermodynamics. In: ICML (2015)"},{"key":"2_CR57","unstructured":"Song, J., Meng, C., Ermon, S.: Denoising diffusion implicit models. In: ICLR (2021)"},{"key":"2_CR58","unstructured":"Song, Y., Durkan, C., Murray, I., Ermon, S.: Maximum likelihood training of score-based diffusion models. In: NeurIPS (2021)"},{"key":"2_CR59","unstructured":"Song, Y., Sohl-Dickstein, J., Kingma, D.P., Kumar, A., Ermon, S., Poole, B.: Score-based generative modeling through stochastic differential equations. In: ICLR (2021)"},{"key":"2_CR60","unstructured":"Tong, A., et al.: Improving and generalizing flow-based generative models with minibatch optimal transport. In: ICML Workshop on New Frontiers in Learning, Control, and Dynamical Systems (2023)"},{"key":"2_CR61","unstructured":"Vahdat, A., Kreis, K., Kautz, J.: Score-based generative modeling in latent space. In: NIPS (2021)"},{"key":"2_CR62","unstructured":"Vaswani, A., et al.: Attention is all you need. In: NIPS (2017)"},{"key":"2_CR63","doi-asserted-by":"crossref","unstructured":"Wang, Q., et al.: Learning deep transformer models for machine translation. In: ACL (2019)","DOI":"10.18653\/v1\/P19-1176"},{"key":"2_CR64","unstructured":"Zaheer, M., et al.: Big bird: transformers for longer sequences. In: NeurIPS (2020)"},{"key":"2_CR65","unstructured":"Zheng, H., Nie, W., Vahdat, A., Anandkumar, A.: Fast training of diffusion models with masked transformers. arXiv preprint arXiv:2306.09305 (2023)"},{"key":"2_CR66","unstructured":"Zheng, K., Lu, C., Chen, J., Zhu, J.: Improved techniques for maximum likelihood estimation for diffusion odes. In: ICML (2023)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72980-5_2","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T10:07:17Z","timestamp":1730110037000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72980-5_2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"ISBN":["9783031729799","9783031729805"],"references-count":66,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72980-5_2","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024]]},"assertion":[{"value":"29 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}