{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,2]],"date-time":"2026-01-02T03:24:51Z","timestamp":1767324291471,"version":"3.48.0"},"publisher-location":"Cham","reference-count":51,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032128393","type":"print"},{"value":"9783032128409","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-12840-9_11","type":"book-chapter","created":{"date-parts":[[2026,1,2]],"date-time":"2026-01-02T03:21:42Z","timestamp":1767324102000},"page":"155-170","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["StorySync: Training-Free Subject Consistency via\u00a0Region Harmonization"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0183-8115","authenticated-orcid":false,"given":"Gopalji","family":"Gaur","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2973-4302","authenticated-orcid":false,"given":"Mohammadreza","family":"Zolfaghari","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6282-8861","authenticated-orcid":false,"given":"Thomas","family":"Brox","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,1,2]]},"reference":[{"key":"11_CR1","doi-asserted-by":"crossref","unstructured":"Arar, M., et al.: Domain-agnostic tuning-encoder for fast personalization of text-to-image models. In: SIGGRAPH Asia 2023 Conference Papers (2023). https:\/\/api.semanticscholar.org\/CorpusID:259847716","DOI":"10.1145\/3610548.3618173"},{"key":"11_CR2","unstructured":"Arkhipkin, V., et al.: Kandinsky 3.0 technical report (2024). https:\/\/arxiv.org\/abs\/2312.03511"},{"key":"11_CR3","doi-asserted-by":"publisher","unstructured":"Avrahami, O., et al.: The chosen one: consistent characters in text-to-image diffusion models. In: Special Interest Group on Computer Graphics and Interactive Techniques Conference Conference Papers 2024, SIGGRAPH 2024, pp. 1\u201312. ACM (2024). https:\/\/doi.org\/10.1145\/3641519.3657430","DOI":"10.1145\/3641519.3657430"},{"key":"11_CR4","unstructured":"Balaji, Y., et al.: eDiff-I: text-to-image diffusion models with an ensemble of expert denoisers. arXiv abs\/2211.01324 (2022). https:\/\/api.semanticscholar.org\/CorpusID:253254800"},{"key":"11_CR5","doi-asserted-by":"crossref","unstructured":"Cao, M., Wang, X., Qi, Z., Shan, Y., Qie, X., Zheng, Y.: Masactrl: tuning-free mutual self-attention control for consistent image synthesis and editing. In: 2023 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 22503\u201322513 (2023). https:\/\/api.semanticscholar.org\/CorpusID:258179432","DOI":"10.1109\/ICCV51070.2023.02062"},{"key":"11_CR6","unstructured":"Dong, Z., Wei, P., Lin, L.: DreamArtist: Towards Controllable One-Shot Text-to-Image Generation via Positive-Negative Prompt-Tuning (2023). http:\/\/arxiv.org\/abs\/2211.11337, arXiv:2211.11337"},{"key":"11_CR7","doi-asserted-by":"crossref","unstructured":"Feng, Z., et al.: Improved visual story generation with adaptive context modeling. arXiv preprint arXiv:2305.16811 (2023)","DOI":"10.18653\/v1\/2023.findings-acl.305"},{"key":"11_CR8","unstructured":"Fu, S., et al.: Dreamsim: learning new dimensions of human visual similarity using synthetic data. arXiv preprint arXiv:2306.09344 (2023)"},{"key":"11_CR9","unstructured":"Gal, R., et al.: An image is worth one word: personalizing text-to-image generation using textual inversion. arXiv abs\/2208.01618 (2022). https:\/\/api.semanticscholar.org\/CorpusID:251253049"},{"key":"11_CR10","unstructured":"Gal, R., et al.: An Image is Worth One Word: Personalizing Text-to-Image Generation using Textual Inversion (2022). http:\/\/arxiv.org\/abs\/2208.01618. arXiv:2208.01618"},{"key":"11_CR11","doi-asserted-by":"crossref","unstructured":"Gal, R., Arar, M., Atzmon, Y., Bermano, A.H., Chechik, G., Cohen-Or, D.: Encoder-based Domain Tuning for Fast Personalization of Text-to-Image Models (2023). http:\/\/arxiv.org\/abs\/2302.12228. arXiv:2302.12228","DOI":"10.1145\/3610548.3618173"},{"key":"11_CR12","unstructured":"Gong, Y., et al.: Talecrafter: interactive story visualization with multiple characters. arXiv preprint arXiv:2305.18247 (2023)"},{"key":"11_CR13","unstructured":"Gu, Y., et al.: Mix-of-show: decentralized low-rank adaptation for multi-concept customization of diffusion models. arXiv abs\/2305.18292 (2023). https:\/\/api.semanticscholar.org\/CorpusID:258960192"},{"key":"11_CR14","unstructured":"He, H., et al.: Dreamstory: open-domain story visualization by LLM-guided multi-subject consistent diffusion. arXiv preprint arXiv:2407.12899 (2024)"},{"key":"11_CR15","unstructured":"He, J., Tuo, Y., Chen, B., Zhong, C., Geng, Y., Bo, L.: Anystory: towards unified single and multiple subject personalization in text-to-image generation. arXiv preprint arXiv:2501.09503 (2025)"},{"key":"11_CR16","unstructured":"Ho, J., Jain, A., Abbeel, P.: Denoising diffusion probabilistic models. arXiv preprint arxiv:2006.11239 (2020)"},{"key":"11_CR17","unstructured":"Huang, Z., Wu, T., Jiang, Y., Chan, K.C.K., Liu, Z.: ReVersion: Diffusion-Based Relation Inversion from Images (2023). http:\/\/arxiv.org\/abs\/2303.13495. arXiv:2303.13495"},{"key":"11_CR18","unstructured":"Jeong, H., Kwon, G., Ye, J.C.: Zero-shot generation of coherent storybook from plain text story using diffusion models. arXiv preprint arXiv:2302.03900 (2023)"},{"key":"11_CR19","doi-asserted-by":"publisher","unstructured":"Kumari, N., Zhang, B., Zhang, R., Shechtman, E., Zhu, J.Y.: Multi-concept customization of text-to-image diffusion. In: 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 1931\u20131941. IEEE, Vancouver, BC, Canada (2023). https:\/\/doi.org\/10.1109\/CVPR52729.2023.00192. https:\/\/ieeexplore.ieee.org\/document\/10203856\/","DOI":"10.1109\/CVPR52729.2023.00192"},{"key":"11_CR20","unstructured":"Li, D., Li, J., Hoi, S.C.H.: Blip-diffusion: pre-trained subject representation for controllable text-to-image generation and editing. arXiv abs\/2305.14720 (2023). https:\/\/api.semanticscholar.org\/CorpusID:258865473"},{"key":"11_CR21","doi-asserted-by":"crossref","unstructured":"Liu, C., Wu, H., Zhong, Y., Zhang, X., Xie, W.: Intelligent grimm - open-ended visual storytelling via latent diffusion models. In: 2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 6190\u20136200 (2023). https:\/\/api.semanticscholar.org\/CorpusID:258999141","DOI":"10.1109\/CVPR52733.2024.00592"},{"key":"11_CR22","doi-asserted-by":"crossref","unstructured":"Liu, C., Wu, H., Zhong, Y., Zhang, X., Wang, Y., Xie, W.: Intelligent grimm-open-ended visual storytelling via latent diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6190\u20136200 (2024)","DOI":"10.1109\/CVPR52733.2024.00592"},{"key":"11_CR23","unstructured":"Liu, T., et al.: One-prompt-one-story: Free-lunch consistent text-to-image generation using a single prompt. arXiv preprint arXiv:2501.13554 (2025)"},{"key":"11_CR24","doi-asserted-by":"crossref","unstructured":"Otsu, N.: A threshold selection method from gray-level histograms. IEEE Trans. Syst. Man Cybern. 9, 62\u201366 (1979). https:\/\/api.semanticscholar.org\/CorpusID:15326934","DOI":"10.1109\/TSMC.1979.4310076"},{"key":"11_CR25","doi-asserted-by":"crossref","unstructured":"Pan, X., Qin, P., Li, Y., Xue, H., Chen, W.: Synthesizing coherent story with auto-regressive latent diffusion models. In: 2024 IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV), pp. 2908\u20132918 (2022). https:\/\/api.semanticscholar.org\/CorpusID:253734226","DOI":"10.1109\/WACV57701.2024.00290"},{"key":"11_CR26","doi-asserted-by":"crossref","unstructured":"Patashnik, O., Garibi, D., Azuri, I., Averbuch-Elor, H., Cohen-Or, D.: Localizing object-level shape variations with text-to-image diffusion models. In: 2023 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 22994\u201323004 (2023). https:\/\/api.semanticscholar.org\/CorpusID:257632209","DOI":"10.1109\/ICCV51070.2023.02107"},{"key":"11_CR27","doi-asserted-by":"crossref","unstructured":"Po, R., Yang, G., Aberman, K., Wetzstein, G.: Orthogonal adaptation for modular customization of diffusion models. In: 2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 7964\u20137973 (2023). https:\/\/api.semanticscholar.org\/CorpusID:265659333","DOI":"10.1109\/CVPR52733.2024.00761"},{"key":"11_CR28","unstructured":"Podell, D., et al.: SDXL: improving latent diffusion models for high-resolution image synthesis. arXiv preprint arXiv:2307.01952 (2023)"},{"key":"11_CR29","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PmLR (2021)"},{"key":"11_CR30","unstructured":"Richardson, E., Goldberg, K., Alaluf, Y., Cohen-Or, D.: Conceptlab: creative generation using diffusion prior constraints. arXiv abs\/2308.02669 (2023). https:\/\/api.semanticscholar.org\/CorpusID:260683062"},{"key":"11_CR31","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., Ommer, B.: High-resolution image synthesis with latent diffusion models. CoRR abs\/2112.10752 (2021). https:\/\/arxiv.org\/abs\/2112.10752"},{"key":"11_CR32","unstructured":"Ronneberger, O., Fischer, P., Brox, T.: U-net: convolutional networks for biomedical image segmentation. CoRR abs\/1505.04597 (2015). http:\/\/arxiv.org\/abs\/1505.04597"},{"key":"11_CR33","doi-asserted-by":"crossref","unstructured":"Ruiz, N., Li, Y., Jampani, V., Pritch, Y., Rubinstein, M., Aberman, K.: Dreambooth: fine tuning text-to-image diffusion models for subject-driven generation. In: 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 22500\u201322510 (2022). https:\/\/api.semanticscholar.org\/CorpusID:251800180","DOI":"10.1109\/CVPR52729.2023.02155"},{"key":"11_CR34","doi-asserted-by":"crossref","unstructured":"Shi, J., Xiong, W., Lin, Z., Jung, H.J.: Instantbooth: personalized text-to-image generation without test-time finetuning. In: 2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 8543\u20138552 (2023). https:\/\/api.semanticscholar.org\/CorpusID:258041269","DOI":"10.1109\/CVPR52733.2024.00816"},{"key":"11_CR35","unstructured":"Su, S., Guo, L., Gao, L., Shen, H., Song, J.: Make-a-storyboard: a general framework for storyboard with disentangled and merged control. arXiv abs\/2312.07549 (2023). https:\/\/api.semanticscholar.org\/CorpusID:266191147"},{"key":"11_CR36","unstructured":"Tang, L., Jia, M., Wang, Q., Phoo, C.P., Hariharan, B.: Emergent correspondence from image diffusion. arXiv abs\/2306.03881 (2023). https:\/\/api.semanticscholar.org\/CorpusID:259089017"},{"key":"11_CR37","unstructured":"Tewel, Y., Gal, R., Samuel, D., Atzmon, Y., Wolf, L., Chechik, G.: Add-it: training-free object insertion in images with pretrained diffusion models. arXiv preprint arXiv:2411.07232 (2024)"},{"issue":"4","key":"11_CR38","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3658157","volume":"43","author":"Y Tewel","year":"2024","unstructured":"Tewel, Y., et al.: Training-free consistent text-to-image generation. ACM Trans. Graph. (TOG) 43(4), 1\u201318 (2024)","journal-title":"ACM Trans. Graph. (TOG)"},{"key":"11_CR39","unstructured":"Wang, J., et al.: Oneactor: consistent character generation via cluster-conditioned guidance. arXiv preprint arXiv:2404.10267 (2024)"},{"key":"11_CR40","unstructured":"Wang, J., et al.: Spotactor: training-free layout-controlled consistent image generation. arXiv preprint arXiv:2409.04801 (2024)"},{"key":"11_CR41","doi-asserted-by":"crossref","unstructured":"Wang, Q., et al.: Characterfactory: sampling consistent characters with GANs for diffusion models. arXiv preprint arXiv:2404.15677 (2024)","DOI":"10.1109\/TIP.2025.3558668"},{"key":"11_CR42","doi-asserted-by":"crossref","unstructured":"Wei, Y., Zhang, Y., Ji, Z., Bai, J., Zhang, L., Zuo, W.: Elite: encoding visual concepts into textual embeddings for customized text-to-image generation. In: 2023 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 15897\u201315907 (2023). https:\/\/api.semanticscholar.org\/CorpusID:257219968","DOI":"10.1109\/ICCV51070.2023.01461"},{"key":"11_CR43","doi-asserted-by":"crossref","unstructured":"Wu, J.Z., et al.: Tune-a-video: one-shot tuning of image diffusion models for text-to-video generation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7623\u20137633 (2023)","DOI":"10.1109\/ICCV51070.2023.00701"},{"key":"11_CR44","unstructured":"Yang, S., et al.: Seed-story: multimodal long story generation with large language model. arXiv preprint arXiv:2407.08683 (2024)"},{"key":"11_CR45","unstructured":"Ye, H., Zhang, J., Liu, S., Han, X., Yang, W.: IP-adapter: text compatible image prompt adapter for text-to-image diffusion models. arXiv preprint arXiv:2308.06721 (2023)"},{"key":"11_CR46","doi-asserted-by":"crossref","unstructured":"Zhang, R., Isola, P., Efros, A.A., Shechtman, E., Wang, O.: The unreasonable effectiveness of deep features as a perceptual metric. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 586\u2013595 (2018)","DOI":"10.1109\/CVPR.2018.00068"},{"key":"11_CR47","doi-asserted-by":"publisher","unstructured":"Zhang, Y., et al.: Inversion-based style transfer with diffusion models. In: 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 10146\u201310156. IEEE, Vancouver, BC, Canada (2023). https:\/\/doi.org\/10.1109\/CVPR52729.2023.00978. https:\/\/ieeexplore.ieee.org\/document\/10203866\/","DOI":"10.1109\/CVPR52729.2023.00978"},{"key":"11_CR48","doi-asserted-by":"crossref","unstructured":"Zheng, S., Fu, Y.: Contextualstory: consistent visual storytelling with spatially-enhanced and storyline context. arXiv preprint arXiv:2407.09774 (2024)","DOI":"10.1609\/aaai.v39i10.33153"},{"key":"11_CR49","first-page":"110315","volume":"37","author":"Y Zhou","year":"2024","unstructured":"Zhou, Y., Zhou, D., Cheng, M.M., Feng, J., Hou, Q.: Storydiffusion: consistent self-attention for long-range image and video generation. Adv. Neural. Inf. Process. Syst. 37, 110315\u2013110340 (2024)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"11_CR50","unstructured":"Zhou, Z., Li, J., Li, H., Chen, N., Tang, X.: Storymaker: towards holistic consistent characters in text-to-image generation. arXiv preprint arXiv:2409.12576 (2024)"},{"key":"11_CR51","unstructured":"Zhu, J., Ma, H., Chen, J., Yuan, J.: DomainStudio: Fine-Tuning Diffusion Models for Domain-Driven Image Generation using Limited Data (2023). http:\/\/arxiv.org\/abs\/2306.14153. arXiv:2306.14153"}],"container-title":["Lecture Notes in Computer Science","Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-12840-9_11","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,1,2]],"date-time":"2026-01-02T03:21:48Z","timestamp":1767324108000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-12840-9_11"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9783032128393","9783032128409"],"references-count":51,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-12840-9_11","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"2 January 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"DAGM GCPR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"DAGM German Conference on Pattern Recognition","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Freiburg","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Germany","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 September 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"26 September 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"47","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"dagm2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.dagm-gcpr.de\/year\/2025","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}