{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,10]],"date-time":"2025-10-10T00:27:06Z","timestamp":1760056026901,"version":"build-2065373602"},"publisher-location":"Cham","reference-count":33,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031723377"},{"type":"electronic","value":"9783031723384"}],"license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024]]},"DOI":"10.1007\/978-3-031-72338-4_23","type":"book-chapter","created":{"date-parts":[[2024,9,16]],"date-time":"2024-09-16T10:03:01Z","timestamp":1726480981000},"page":"333-348","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["MAGIC: Multi-prompt Any Length Video Generation Model with\u00a0Controllable Inter-frame Correlation and\u00a0Low Barrier"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-0435-1779","authenticated-orcid":false,"given":"Jialiang","family":"Xu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-1559-2498","authenticated-orcid":false,"given":"Weiran","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-6723-5341","authenticated-orcid":false,"given":"Lingbing","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-9804-4277","authenticated-orcid":false,"given":"Weitao","family":"Song","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6965-4158","authenticated-orcid":false,"given":"Yi","family":"Ji","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-1669-1878","authenticated-orcid":false,"given":"Ying","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-1495-5138","authenticated-orcid":false,"given":"Chunping","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,9,17]]},"reference":[{"key":"23_CR1","doi-asserted-by":"crossref","unstructured":"Blattmann, A., Rombach, R., Ling, H., et\u00a0al.: Align your latents: high-resolution video synthesis with latent diffusion models. In: 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 22563\u201322575 (2023)","DOI":"10.1109\/CVPR52729.2023.02161"},{"key":"23_CR2","unstructured":"Brooks, T., Peebles, B., Holmes, C., et\u00a0al.: Video generation models as world simulators (2024). https:\/\/openai.com\/research\/video-generation-models-as-world-simulators"},{"key":"23_CR3","doi-asserted-by":"crossref","unstructured":"Esser, P., Chiu, J., Atighehchian, P., et\u00a0al.: Structure and content-guided video synthesis with diffusion models. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 7346\u20137356, October 2023","DOI":"10.1109\/ICCV51070.2023.00675"},{"key":"23_CR4","unstructured":"Goodfellow, I.J., Pouget-Abadie, J., Mirza, M., et\u00a0al.: Generative adversarial nets. In: Advances in Neural Information Processing Systems 27: Annual Conference on Neural Information Processing Systems 2014, 8\u201313 December 2014, Montreal, Quebec, Canada, pp. 2672\u20132680 (2014)"},{"key":"23_CR5","unstructured":"Harvey, W., Naderiparizi, S., Masrani, V., et\u00a0al.: Flexible diffusion modeling of long videos. In: NeurIPS (2022)"},{"key":"23_CR6","unstructured":"He, Y., Yang, T., Zhang, Y., et\u00a0al.: Latent video diffusion models for high-fidelity video generation with arbitrary lengths. arXiv preprint arXiv:2211.13221 (2022)"},{"key":"23_CR7","doi-asserted-by":"crossref","unstructured":"Hessel, J., Holtzman, A., Forbes, M., et\u00a0al.: CLIPScore: a reference-free evaluation metric for image captioning. In: Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing, pp. 7514\u20137528. Association for Computational Linguistics, Online and Punta Cana, Dominican Republic, November 2021","DOI":"10.18653\/v1\/2021.emnlp-main.595"},{"key":"23_CR8","unstructured":"Ho, J., Chan, W., Saharia, C., et\u00a0al.: Imagen video: high definition video generation with diffusion models. arXiv preprint arXiv:2210.02303 (2022)"},{"key":"23_CR9","unstructured":"Ho, J., Jain, A., Abbeel, P.: Denoising diffusion probabilistic models. In: Larochelle, H., Ranzato, M., Hadsell, R., Balcan, M., Lin, H. (eds.) Advances in Neural Information Processing Systems, vol.\u00a033, pp. 6840\u20136851. Curran Associates, Inc. (2020)"},{"key":"23_CR10","unstructured":"Ho, J., Salimans, T., Gritsenko, A., et\u00a0al.: Video diffusion models. In: Koyejo, S., Mohamed, S., Agarwal, A., et\u00a0al. (eds.) Advances in Neural Information Processing Systems, vol.\u00a035, pp. 8633\u20138646. Curran Associates, Inc. (2022)"},{"key":"23_CR11","doi-asserted-by":"crossref","unstructured":"Khachatryan, L., Movsisyan, A., Tadevosyan, V., et\u00a0al.: Text2Video-zero: text-to-image diffusion models are zero-shot video generators. In: 2023 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 15908\u201315918 (2023)","DOI":"10.1109\/ICCV51070.2023.01462"},{"key":"23_CR12","unstructured":"Liu, Y., Han, T., Ma, S., et\u00a0al.: Summary of ChatGPT\/GPT-4 research and perspective towards the future of large language models. arXiv preprint arXiv:2304.01852 (2023)"},{"key":"23_CR13","unstructured":"Lu, C., Zhou, Y., Bao, F., et\u00a0al.: DPM-Solver++: fast solver for guided sampling of diffusion probabilistic models. arXiv preprint arXiv:2211.01095 (2022)"},{"issue":"3","key":"23_CR14","doi-asserted-by":"publisher","first-page":"209","DOI":"10.1109\/LSP.2012.2227726","volume":"20","author":"A Mittal","year":"2013","unstructured":"Mittal, A., Soundararajan, R., Bovik, A.C.: Making a \u201ccompletely blind\u2019\u2019 image quality analyzer. IEEE Signal Process. Lett. 20(3), 209\u2013212 (2013)","journal-title":"IEEE Signal Process. Lett."},{"key":"23_CR15","doi-asserted-by":"crossref","unstructured":"Mou, C., Wang, X., Xie, L., et\u00a0al.: T2i-Adapter: learning adapters to dig out more controllable ability for text-to-image diffusion models. arXiv preprint arXiv:2302.08453 (2023)","DOI":"10.1609\/aaai.v38i5.28226"},{"key":"23_CR16","unstructured":"Nichol, A.Q., Dhariwal, P.: Improved denoising diffusion probabilistic models. In: Meila, M., Zhang, T. (eds.) Proceedings of the 38th International Conference on Machine Learning. Proceedings of Machine Learning Research, vol.\u00a0139, pp. 8162\u20138171. PMLR, 18\u201324 July 2021"},{"key":"23_CR17","unstructured":"Nichol, A.Q., Dhariwal, P., Ramesh, A., et\u00a0al.: GLIDE: towards photorealistic image generation and editing with text-guided diffusion models. In: Chaudhuri, K., Jegelka, S., Song, L., Szepesvari, C., Niu, G., Sabato, S. (eds.) Proceedings of the 39th International Conference on Machine Learning. Proceedings of Machine Learning Research, vol.\u00a0162, pp. 16784\u201316804. PMLR, 17\u201323 July 2022"},{"key":"23_CR18","unstructured":"Radford, A., Kim, J.W., Hallacy, C., et\u00a0al.: Learning transferable visual models from natural language supervision. In: Meila, M., Zhang, T. (eds.) Proceedings of the 38th International Conference on Machine Learning. Proceedings of Machine Learning Research, vol.\u00a0139, pp. 8748\u20138763. PMLR, 18\u201324 July 2021"},{"key":"23_CR19","unstructured":"Ramesh, A., Dhariwal, P., Nichol, A., et\u00a0al.: Hierarchical text-conditional image generation with CLIP latents. arXiv preprint arXiv:2204.06125 (2022)"},{"key":"23_CR20","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., et\u00a0al.: High-resolution image synthesis with latent diffusion models. In: 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 10674\u201310685 (2022)","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"23_CR21","unstructured":"Saharia, C., Chan, W., Saxena, S., et\u00a0al.: Photorealistic text-to-image diffusion models with deep language understanding. In: Koyejo, S., Mohamed, S., Agarwal, A., Belgrave, D., Cho, K., Oh, A. (eds.) Advances in Neural Information Processing Systems, vol.\u00a035, pp. 36479\u201336494. Curran Associates, Inc. (2022)"},{"key":"23_CR22","unstructured":"Schuhmann, C., Beaumont, R., Vencu, R., et\u00a0al.: LAION-5B: an open large-scale dataset for training next generation image-text models. In: Koyejo, S., Mohamed, S., Agarwal, A., Belgrave, D., Cho, K., Oh, A. (eds.) Advances in Neural Information Processing Systems, vol.\u00a035, pp. 25278\u201325294. Curran Associates, Inc. (2022)"},{"key":"23_CR23","unstructured":"Singer, U., Polyak, A., Hayes, T., et\u00a0al.: Make-A-Video: text-to-video generation without text-video data. arXiv preprint arXiv:2209.14792 (2022)"},{"key":"23_CR24","unstructured":"Song, Y., Sohl-Dickstein, J., Kingma, D.P., et\u00a0al.: Score-based generative modeling through stochastic differential equations. In: International Conference on Learning Representations (2021)"},{"key":"23_CR25","unstructured":"Villegas, R., Babaeizadeh, M., Kindermans, P.J., et\u00a0al.: Phenaki: variable length video generation from open domain textual description. arXiv preprint arXiv:2210.02399 (2022)"},{"key":"23_CR26","unstructured":"Wang, F.Y., Chen, W., Song, G., et\u00a0al.: Gen-L-Video: multi-text to long video generation via temporal co-denoising. arXiv preprint arXiv:2305.18264 (2023)"},{"key":"23_CR27","unstructured":"Wu, C., Huang, L., Zhang, Q., et\u00a0al.: GODIVA: generating open-domain videos from natural descriptions. arXiv preprint arXiv:2104.14806 (2021)"},{"key":"23_CR28","doi-asserted-by":"crossref","unstructured":"Wu, H., Zhang, E., Liao, L., et\u00a0al.: Exploring video quality assessment on user generated contents from aesthetic and technical perspectives. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 20144\u201320154, October 2023","DOI":"10.1109\/ICCV51070.2023.01843"},{"key":"23_CR29","doi-asserted-by":"crossref","unstructured":"Wu, J.Z., Ge, Y., Wang, X., et\u00a0al.: Tune-A-Video: one-shot tuning of image diffusion models for text-to-video generation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 7623\u20137633, October 2023","DOI":"10.1109\/ICCV51070.2023.00701"},{"key":"23_CR30","doi-asserted-by":"crossref","unstructured":"Yin, S., Wu, C., Yang, H., et\u00a0al.: NUWA-XL: diffusion over diffusion for eXtremely long video generation. In: Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 1309\u20131320. Association for Computational Linguistics, Toronto, Canada, July 2023","DOI":"10.18653\/v1\/2023.acl-long.73"},{"key":"23_CR31","doi-asserted-by":"crossref","unstructured":"Yu, S., Sohn, K., Kim, S., Shin, J.: Video probabilistic diffusion models in projected latent space. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18456\u201318466 (2023)","DOI":"10.1109\/CVPR52729.2023.01770"},{"key":"23_CR32","doi-asserted-by":"crossref","unstructured":"Zhang, L., Rao, A., Agrawala, M.: Adding conditional control to text-to-image diffusion models. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 3836\u20133847, October 2023","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"23_CR33","unstructured":"Zhang, Y., Wei, Y., Jiang, D., et\u00a0al.: ControlVideo: training-free controllable text-to-video generation. arXiv preprint arXiv:2305.13077 (2023)"}],"container-title":["Lecture Notes in Computer Science","Artificial Neural Networks and Machine Learning \u2013 ICANN 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72338-4_23","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,9]],"date-time":"2025-10-09T07:49:54Z","timestamp":1759996194000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72338-4_23"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"ISBN":["9783031723377","9783031723384"],"references-count":33,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72338-4_23","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2024]]},"assertion":[{"value":"17 September 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICANN","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Artificial Neural Networks","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Lugano","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Switzerland","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20 September 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"33","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icann2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}