{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,4]],"date-time":"2026-06-04T19:02:17Z","timestamp":1780599737675,"version":"3.54.1"},"publisher-location":"Cham","reference-count":89,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031726323","type":"print"},{"value":"9783031726330","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,11,22]],"date-time":"2024-11-22T00:00:00Z","timestamp":1732233600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,22]],"date-time":"2024-11-22T00:00:00Z","timestamp":1732233600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72633-0_24","type":"book-chapter","created":{"date-parts":[[2024,11,21]],"date-time":"2024-11-21T07:57:11Z","timestamp":1732175831000},"page":"424-441","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":9,"title":["Efficient Diffusion Transformer with\u00a0Step-Wise Dynamic Attention Mediators"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0404-1737","authenticated-orcid":false,"given":"Yifan","family":"Pu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-7965-364X","authenticated-orcid":false,"given":"Zhuofan","family":"Xia","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-7004-939X","authenticated-orcid":false,"given":"Jiayi","family":"Guo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-3431-6189","authenticated-orcid":false,"given":"Dongchen","family":"Han","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-4866-6920","authenticated-orcid":false,"given":"Qixiu","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-3524-1935","authenticated-orcid":false,"given":"Duo","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8345-4205","authenticated-orcid":false,"given":"Yuhui","family":"Yuan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4699-084X","authenticated-orcid":false,"given":"Ji","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5706-8784","authenticated-orcid":false,"given":"Yizeng","family":"Han","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7361-9283","authenticated-orcid":false,"given":"Shiji","family":"Song","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7251-0988","authenticated-orcid":false,"given":"Gao","family":"Huang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0403-1923","authenticated-orcid":false,"given":"Xiu","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,11,22]]},"reference":[{"key":"24_CR1","unstructured":"Achiam, J., et\u00a0al.: GPT-4 technical report. arXiv preprint arXiv:2303.08774 (2023)"},{"key":"24_CR2","unstructured":"Aditya, R., Prafulla, D., Alex, N., Casey, C., Mark, C.: Hierarchical text-conditional image generation with clip latents. arXiv:2204.06125 (2022)"},{"key":"24_CR3","doi-asserted-by":"crossref","unstructured":"Bao, F., et al.: All are worth words: a vit backbone for diffusion models. In: IEEE CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.02171"},{"key":"24_CR4","unstructured":"Brock, A., Donahue, J., Simonyan, K.: Large scale GAN training for high fidelity natural image synthesis. In: ICLR (2019)"},{"key":"24_CR5","unstructured":"Brooks, T., et al.: Video generation models as world simulators (2024). https:\/\/openai.com\/research\/video-generation-models-as-world-simulators"},{"key":"24_CR6","unstructured":"Brown, T., et\u00a0al.: Language models are few-shot learners. In: NeurIPS (2020)"},{"key":"24_CR7","doi-asserted-by":"crossref","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., Zagoruyko, S.: End-to-end object detection with transformers. In: ECCV (2020)","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"24_CR8","doi-asserted-by":"crossref","unstructured":"Chang, H., Zhang, H., Jiang, L., Liu, C., Freeman, W.T.: Maskgit: masked generative image transformer. In: IEEE CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.01103"},{"key":"24_CR9","doi-asserted-by":"crossref","unstructured":"Chen, J., et al.: Pixart-$$\\sigma $$: weak-to-strong training of diffusion transformer for 4k text-to-image generation. In: ECCV (2024)","DOI":"10.1007\/978-3-031-73411-3_5"},{"key":"24_CR10","unstructured":"Chen, J., et al.: Pixart-$$\\delta $$: fast and controllable image generation with latent consistency models. In: ICML (2024)"},{"key":"24_CR11","doi-asserted-by":"crossref","unstructured":"Chen, J., et\u00a0al.: Pixart-$$\\alpha $$: fast training of diffusion transformer for photorealistic text-to-image synthesis. In: ICLR (2024)","DOI":"10.1007\/978-3-031-73411-3_5"},{"key":"24_CR12","unstructured":"Crowson, K., Baumann, S.A., Birch, A., Abraham, T.M., Kaplan, D.Z., Shippole, E.: Scalable high-resolution pixel-space image synthesis with hourglass diffusion transformers. In: ICML (2024)"},{"key":"24_CR13","doi-asserted-by":"crossref","unstructured":"Deng, J., Dong, W., Socher, R., Li, L.J., Li, K., Fei-Fei, L.: ImageNet: a large-scale hierarchical image database. In: IEEE CVPR (2009)","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"24_CR14","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: Bert: pre-training of deep bidirectional transformers for language understanding. In: ACL (2019)"},{"key":"24_CR15","unstructured":"Dhariwal, P., Nichol, A.: Diffusion models beat gans on image synthesis. In: NeurIPS (2021)"},{"key":"24_CR16","unstructured":"Dosovitskiy, A., et al.: An image is worth 16x16 words: transformers for image recognition at scale. In: ICLR (2021)"},{"key":"24_CR17","unstructured":"Esser, P., et al.: Scaling rectified flow transformers for high-resolution image synthesis (2024). https:\/\/stabilityai-public-packages.s3.us-west-2.amazonaws.com\/Stable+Diffusion+3+Paper.pdf"},{"key":"24_CR18","doi-asserted-by":"crossref","unstructured":"Fang, Y., et al.: EVA: exploring the limits of masked visual representation learning at scale. In: IEEE CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.01855"},{"key":"24_CR19","doi-asserted-by":"crossref","unstructured":"Gao, S., Zhou, P., Cheng, M.M., Yan, S.: Masked diffusion transformer is a strong image synthesizer. In: IEEE ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.02117"},{"key":"24_CR20","doi-asserted-by":"crossref","unstructured":"Guo, J., et al.: Zero-shot generative model adaptation via image-specific prompt learning. In: IEEE CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.01106"},{"key":"24_CR21","doi-asserted-by":"crossref","unstructured":"Guo, J., et al.: Smooth diffusion: crafting smooth latent spaces in diffusion models. In: IEEE CVPR (2024)","DOI":"10.1109\/CVPR52733.2024.00721"},{"key":"24_CR22","doi-asserted-by":"crossref","unstructured":"Han, D., Pan, X., Han, Y., Song, S., Huang, G.: FLatten transformer: vision transformer using focused linear attention. In: IEEE ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.00548"},{"key":"24_CR23","doi-asserted-by":"crossref","unstructured":"Han, D., Ye, T., Han, Y., Xia, Z., Song, S., Huang, G.: Agent attention: On the integration of softmax and linear attention. In: ECCV (2024)","DOI":"10.1007\/978-3-031-72973-7_8"},{"key":"24_CR24","doi-asserted-by":"crossref","unstructured":"Han, Y., et al.: Dynamic perceiver for efficient visual recognition. In: IEEE ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.00551"},{"key":"24_CR25","unstructured":"Han, Y., Huang, G., Song, S., Yang, L., Wang, H., Wang, Y.: Dynamic neural networks: a survey. In: IEEE TPAMI (2021)"},{"key":"24_CR26","doi-asserted-by":"crossref","unstructured":"Han, Y., Huang, G., Song, S., Yang, L., Zhang, Y., Jiang, H.: Spatially adaptive feature refinement for efficient inference. In: IEEE TIP (2021)","DOI":"10.1109\/TIP.2021.3125263"},{"key":"24_CR27","doi-asserted-by":"crossref","unstructured":"Han, Y., et al.: Latency-aware unified dynamic networks for efficient image recognition. In: IEEE TPAMI (2024)","DOI":"10.1109\/TPAMI.2024.3393530"},{"key":"24_CR28","doi-asserted-by":"crossref","unstructured":"Han, Y., et al.: Learning to weight samples for dynamic early-exiting networks. In: ECCV (2022)","DOI":"10.1007\/978-3-031-20083-0_22"},{"key":"24_CR29","unstructured":"Han, Y., Yuan, Z., Pu, Y., Xue, C., Song, S., Sun, G., Huang, G.: Latency-aware spatial-wise dynamic networks. In: NeurIPS (2022)"},{"key":"24_CR30","unstructured":"Hansen, C., Hansen, C., Alstrup, S., Simonsen, J.G., Lioma, C.: Neural speed reading with structural-jump-lstm. In: ICLR (2019)"},{"key":"24_CR31","doi-asserted-by":"crossref","unstructured":"Hassani, A., Walton, S., Li, J., Li, S., Shi, H.: Neighborhood attention transformer. In: IEEE CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.00599"},{"key":"24_CR32","unstructured":"Heusel, M., Ramsauer, H., Unterthiner, T., Nessler, B., Hochreiter, S.: GANs trained by a two time-scale update rule converge to a local NASH equilibrium. In: NeurIPS (2017)"},{"key":"24_CR33","unstructured":"Ho, J., Jain, A., Abbeel, P.: Denoising diffusion probabilistic models. In: NeurIPS (2020)"},{"key":"24_CR34","unstructured":"Ho, J., Saharia, C., Chan, W., Fleet, D.J., Norouzi, M., Salimans, T.: Cascaded diffusion models for high fidelity image generation. In: JMLR (2022)"},{"key":"24_CR35","unstructured":"Hoogeboom, E., Heek, J., Salimans, T.: simple diffusion: end-to-end diffusion for high resolution images. In: ICML (2023)"},{"key":"24_CR36","unstructured":"Huang, G., Chen, D., Li, T., Wu, F., Van Der\u00a0Maaten, L., Weinberger, K.Q.: Multi-scale dense networks for resource efficient image classification. In: ICLR (2018)"},{"key":"24_CR37","doi-asserted-by":"crossref","unstructured":"Huang, G., et al.: Glance and focus networks for dynamic visual recognition. In: IEEE TPAMI (2022)","DOI":"10.1109\/TPAMI.2022.3196959"},{"key":"24_CR38","unstructured":"Jabri, A., Fleet, D., Chen, T.: Scalable adaptive computation for iterative generation. In: ICML (2023)"},{"key":"24_CR39","unstructured":"Katharopoulos, A., Vyas, A., Pappas, N., Fleuret, F.: Transformers are RNNs: fast autoregressive transformers with linear attention. In: ICML (2020)"},{"key":"24_CR40","unstructured":"Kingma, D.P., Ba, J.: Adam: a method for stochastic optimization. In: ICLR (2015)"},{"key":"24_CR41","unstructured":"Kingma, D.P., Gao, R.: Understanding the diffusion objective as a weighted integral of elbos. In: NeurIPS (2023)"},{"key":"24_CR42","doi-asserted-by":"crossref","unstructured":"Kirillov, A., et\u00a0al.: Segment anything. In: IEEE ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"24_CR43","doi-asserted-by":"crossref","unstructured":"Li, Y., et al.: Efficient and explicit modelling of image hierarchies for image restoration. In: IEEE CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.01753"},{"key":"24_CR44","unstructured":"Li, Z., et\u00a0al.: Hunyuan-DiT: a powerful multi-resolution diffusion transformer with fine-grained Chinese understanding. arXiv preprint arXiv:2405.08748 (2024)"},{"key":"24_CR45","doi-asserted-by":"crossref","unstructured":"Liu, Z., et al.: Swin transformer: hierarchical vision transformer using shifted windows. In: IEEE ICCV (2021)","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"24_CR46","doi-asserted-by":"crossref","unstructured":"Liu, Z., et al.: SCoFT: self-contrastive fine-tuning for equitable image generation. In: CVPR (2024)","DOI":"10.1109\/CVPR52733.2024.01029"},{"key":"24_CR47","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. In: ICLR (2019)"},{"key":"24_CR48","unstructured":"Lu, H., Yang, G., Fei, N., Huo, Y., Lu, Z., Luo, P., Ding, M.: VDT: general-purpose video diffusion transformers via mask modeling. In: ICLR (2023)"},{"key":"24_CR49","unstructured":"Lu, J., et al.: Soft: Softmax-free transformer with linear complexity. In: NeurIPS (2021)"},{"key":"24_CR50","unstructured":"Lu, Z., Wang, Z., Huang, D., Wu, C., Liu, X., Ouyang, W., Bai, L.: FiT: flexible vision transformer for diffusion model. In: ICML (2024)"},{"key":"24_CR51","doi-asserted-by":"crossref","unstructured":"Ma, N., Goldstein, M., Albergo, M.S., Boffi, N.M., Vanden-Eijnden, E., Xie, S.: SiT: exploring flow and diffusion-based generative models with scalable interpolant transformers. In: ECCV (2024)","DOI":"10.1007\/978-3-031-72980-5_2"},{"key":"24_CR52","unstructured":"Ma, X., et al.: Latte: latent diffusion transformer for video generation. arXiv preprint arXiv:2401.03048 (2024)"},{"key":"24_CR53","unstructured":"Michel, P., Levy, O., Neubig, G.: Are sixteen heads really better than one? In: NeurIPS (2019)"},{"key":"24_CR54","unstructured":"Mo, S., Xie, E., Chu, R., Hong, L., Niessner, M., Li, Z.: DiT-3D: exploring plain diffusion transformers for 3d shape generation. In: NeurIPS (2023)"},{"key":"24_CR55","unstructured":"Oquab, M., et al.: DINOv2: learning robust visual features without supervision. In: TMLR (2024)"},{"key":"24_CR56","doi-asserted-by":"crossref","unstructured":"Pan, X., Ye, T., Xia, Z., Song, S., Huang, G.: Slide-transformer: hierarchical vision transformer with local self-attention. In: IEEE CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.00207"},{"key":"24_CR57","doi-asserted-by":"crossref","unstructured":"Peebles, W., Xie, S.: Scalable diffusion models with transformers. In: IEEE ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.00387"},{"key":"24_CR58","doi-asserted-by":"crossref","unstructured":"Pu, Y., Han, Y., Wang, Y., Feng, J., Deng, C., Huang, G.: Fine-grained recognition with learnable semantic data augmentation. In: IEEE TIP (2023)","DOI":"10.1109\/TIP.2024.3364500"},{"key":"24_CR59","unstructured":"Pu, Y., et al.: Rank-detr for high quality object detection. In: NeurIPS (2024)"},{"key":"24_CR60","doi-asserted-by":"crossref","unstructured":"Pu, Y., et al.: Adaptive rotated convolution for rotated object detection. In: IEEE ICCV (2023)","DOI":"10.1109\/ICCV51070.2023.00606"},{"key":"24_CR61","unstructured":"Radford, A., et\u00a0al.: Learning transferable visual models from natural language supervision. In: ICML (2021)"},{"key":"24_CR62","unstructured":"Raffel, C., et al.: Exploring the limits of transfer learning with a unified text-to-text transformer. In: JMLR (2020)"},{"key":"24_CR63","unstructured":"Ramesh, A., et al.: Zero-shot text-to-image generation. In: ICML (2021)"},{"key":"24_CR64","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., Ommer, B.: High-resolution image synthesis with latent diffusion models. In: IEEE CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"24_CR65","doi-asserted-by":"crossref","unstructured":"Ronneberger, O., Fischer, P., Brox, T.: U-Net: Convolutional networks for biomedical image segmentation. In: MICCAI (2015)","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"24_CR66","unstructured":"Saharia, C., et\u00a0al.: Photorealistic text-to-image diffusion models with deep language understanding. In: NeurIPS (2022)"},{"key":"24_CR67","doi-asserted-by":"crossref","unstructured":"Sauer, A., Schwarz, K., Geiger, A.: Stylegan-xl: scaling stylegan to large diverse datasets. In: SIGGRAPH (2022)","DOI":"10.1145\/3528233.3530738"},{"key":"24_CR68","unstructured":"Shen, Z., Zhang, M., Zhao, H., Yi, S., Li, H.: Efficient attention: attention with linear complexities. In: WACV (2021)"},{"key":"24_CR69","unstructured":"Song, L., et al.: Dynamic grained encoder for vision transformers. In: NeurIPS (2021)"},{"key":"24_CR70","unstructured":"Touvron, H., et\u00a0al.: Llama: open and efficient foundation language models. arXiv preprint arXiv:2302.13971 (2023)"},{"key":"24_CR71","unstructured":"Vaswani, A., et al.: Attention is all you need. In: NeurIPS (2017)"},{"key":"24_CR72","unstructured":"Wang, C., Yang, Q., Huang, R., Song, S., Huang, G.: Efficient knowledge distillation from model checkpoints. In: NeurIPS (2022)"},{"key":"24_CR73","doi-asserted-by":"crossref","unstructured":"Wang, J., et al.: GRA: detecting oriented objects through group-wise rotating and attention. In: ECCV (2024)","DOI":"10.1007\/978-3-031-72643-9_18"},{"key":"24_CR74","doi-asserted-by":"crossref","unstructured":"Wang, S., Wu, L., Cui, L., Shen, Y.: Glancing at the patch: anomaly localization with global and local feature comparison. In: IEEE CVPR (2021)","DOI":"10.1109\/CVPR46437.2021.00032"},{"key":"24_CR75","doi-asserted-by":"crossref","unstructured":"Wang, Y., Chen, Z., Jiang, H., Song, S., Han, Y., Huang, G.: Adaptive focus for efficient video recognition. In: IEEE ICCV (2021)","DOI":"10.1109\/ICCV48922.2021.01594"},{"key":"24_CR76","unstructured":"Wang, Y., Han, Y., Wang, C., Song, S., Tian, Q., Huang, G.: Computation-efficient deep learning for computer vision: a survey. In: Cybernetics and Intelligence (2023)"},{"key":"24_CR77","unstructured":"Wang, Y., Huang, R., Song, S., Huang, Z., Huang, G.: Not all images are worth 16x16 words: dynamic transformers for efficient image recognition. In: NeurIPS (2021)"},{"key":"24_CR78","doi-asserted-by":"crossref","unstructured":"Xia, Z., Han, D., Han, Y., Pan, X., Song, S., Huang, G.: Gsva: generalized segmentation via multimodal large language models. In: IEEE CVPR (2024)","DOI":"10.1109\/CVPR52733.2024.00370"},{"key":"24_CR79","unstructured":"Xia, Z., Pan, X., Jin, X., He, Y., Xue\u2019, H., Song, S., Huang, G.: Budgeted training for vision transformer. In: ICLR (2023)"},{"key":"24_CR80","doi-asserted-by":"crossref","unstructured":"Xia, Z., Pan, X., Song, S., Li, L.E., Huang, G.: Vision transformer with deformable attention. In: IEEE CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.00475"},{"key":"24_CR81","doi-asserted-by":"crossref","unstructured":"Xia, Z., Pan, X., Song, S., Li, L.E., Huang, G.: Dat++: Spatially dynamic vision transformer with deformable attention. arXiv preprint arXiv:2309.01430 (2023)","DOI":"10.1109\/CVPR52688.2022.00475"},{"key":"24_CR82","doi-asserted-by":"crossref","unstructured":"Xiong, Y., Zeng, Z., Chakraborty, R., Tan, M., Fung, G., Li, Y., Singh, V.: Nystr\u00f6mformer: A nystr\u00f6m-based algorithm for approximating self-attention. In: AAAI (2021)","DOI":"10.1609\/aaai.v35i16.17664"},{"key":"24_CR83","unstructured":"Xue, S., Yi, M., Luo, W., Zhang, S., Sun, J., Li, Z., Ma, Z.M.: SA-Solver: stochastic adams solver for fast sampling of diffusion models. In: NeurIPS (2023)"},{"key":"24_CR84","unstructured":"Yang, Q., Wang, S., Lin, M.G., Song, S., Huang, G.: Boosting offline reinforcement learning with action preference query. In: ICML (2023)"},{"key":"24_CR85","doi-asserted-by":"crossref","unstructured":"Yang, Q., Wang, S., Zhang, Q., Huang, G., Song, S.: Hundreds guide millions: adaptive offline reinforcement learning with expert guidance. In: IEEE TNNLS (2023)","DOI":"10.1109\/TNNLS.2023.3293508"},{"key":"24_CR86","unstructured":"Yang, X., Shih, S.M., Fu, Y., Zhao, X., Ji, S.: Your ViT is secretly a hybrid discriminative-generative diffusion model. arXiv:2208.07791 (2022)"},{"key":"24_CR87","doi-asserted-by":"crossref","unstructured":"You, H., et al.: Castling-vit: compressing self-attention via switching towards linear-angular attention during vision transformer inference. In: IEEE CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.01387"},{"key":"24_CR88","doi-asserted-by":"crossref","unstructured":"Zhang, T., Huang, H.Y., Feng, C., Cao, L.: Enlivening redundant heads in multi-head self-attention for machine translation. In: EMNLP (2021)","DOI":"10.18653\/v1\/2021.emnlp-main.260"},{"key":"24_CR89","unstructured":"Zheng, H., Nie, W., Vahdat, A., Anandkumar, A.: Fast training of diffusion models with masked transformers. TMLR (2024)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72633-0_24","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T18:48:15Z","timestamp":1733078895000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72633-0_24"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,22]]},"ISBN":["9783031726323","9783031726330"],"references-count":89,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72633-0_24","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,22]]},"assertion":[{"value":"22 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}