{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,30]],"date-time":"2025-09-30T00:22:08Z","timestamp":1759191728678,"version":"3.44.0"},"publisher-location":"Cham","reference-count":39,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032059802","type":"print"},{"value":"9783032059819","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,9,23]],"date-time":"2025-09-23T00:00:00Z","timestamp":1758585600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,9,23]],"date-time":"2025-09-23T00:00:00Z","timestamp":1758585600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-05981-9_19","type":"book-chapter","created":{"date-parts":[[2025,9,29]],"date-time":"2025-09-29T19:04:07Z","timestamp":1759172647000},"page":"310-327","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Enabling ControlNet to\u00a0follow Localized Descriptions Using Cross-Attention Control"],"prefix":"10.1007","author":[{"given":"Denis","family":"Lukovnikov","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Asja","family":"Fischer","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,9,23]]},"reference":[{"key":"19_CR1","doi-asserted-by":"crossref","unstructured":"Avrahami, O., et al.: Spatext: spatio-textual representation for controllable image generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18370\u201318380 (2023)","DOI":"10.1109\/CVPR52729.2023.01762"},{"key":"19_CR2","unstructured":"Bahdanau, D., Cho, K., Bengio, Y.: Neural machine translation by jointly learning to align and translate. CoRR abs\/1409.0473 (2014). https:\/\/api.semanticscholar.org\/CorpusID:11212020"},{"key":"19_CR3","unstructured":"Balaji, Y., et\u00a0al.: ediff-i: text-to-image diffusion models with an ensemble of expert denoisers. arXiv preprint arXiv:2211.01324 (2022)"},{"key":"19_CR4","unstructured":"Bar-Tal, O., Yariv, L., Lipman, Y., Dekel, T.: Multidiffusion: fusing diffusion paths for controlled image generation (2023)"},{"key":"19_CR5","unstructured":"Bi\u0144kowski, M., Sutherland, D.J., Arbel, M., Gretton, A.: Demystifying mmd gans. arXiv preprint arXiv:1801.01401 (2018)"},{"key":"19_CR6","doi-asserted-by":"crossref","unstructured":"Chen, M., Laina, I., Vedaldi, A.: Training-free layout control with cross-attention guidance. arXiv preprint arXiv:2304.03373 (2023)","DOI":"10.1109\/WACV57701.2024.00526"},{"issue":"5","key":"19_CR7","first-page":"1","volume":"28","author":"T Chen","year":"2009","unstructured":"Chen, T., Cheng, M.M., Tan, P., Shamir, A., Hu, S.M.: Sketch2photo: internet image montage. ACM Trans. Graph. (TOG) 28(5), 1\u201310 (2009)","journal-title":"ACM Trans. Graph. (TOG)"},{"key":"19_CR8","doi-asserted-by":"crossref","unstructured":"Couairon, G., Careil, M., Cord, M., Lathuili\u00e8re, S., Verbeek, J.: Zero-shot spatial layout conditioning for text-to-image diffusion models. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2174\u20132183 (2023)","DOI":"10.1109\/ICCV51070.2023.00207"},{"key":"19_CR9","unstructured":"He, Y., Salakhutdinov, R., Kolter, J.Z.: Localized text-to-image generation for free via cross attention control. arXiv preprint arXiv:2306.14636 (2023)"},{"key":"19_CR10","unstructured":"Heusel, M., Ramsauer, H., Unterthiner, T., Nessler, B., Hochreiter, S.: Gans trained by a two time-scale update rule converge to a local Nash equilibrium. In: Advances in Neural Information Processing Systems, vol. 30 (2017)"},{"key":"19_CR11","doi-asserted-by":"crossref","unstructured":"Johnson, M., Brostow, G.J., Shotton, J., Arandjelovic, O., Kwatra, V., Cipolla, R.: Semantic photo synthesis. In: Computer Graphics Forum, vol.\u00a025, pp. 407\u2013413. Wiley Online Library (2006)","DOI":"10.1111\/j.1467-8659.2006.00960.x"},{"key":"19_CR12","doi-asserted-by":"crossref","unstructured":"Kim, Y., Lee, J., Kim, J.H., Ha, J.W., Zhu, J.Y.: Dense text-to-image generation with attention modulation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7701\u20137711 (2023)","DOI":"10.1109\/ICCV51070.2023.00708"},{"key":"19_CR13","unstructured":"Kingma, D.P., Ba, J.: Adam: a method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)"},{"key":"19_CR14","unstructured":"Li, L., et al.: Omnibooth: learning latent control for image synthesis with multi-modal instruction. arXiv preprint arXiv:2410.04932 (2024)"},{"key":"19_CR15","doi-asserted-by":"crossref","unstructured":"Li, Y., et al.: Gligen: open-set grounded text-to-image generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 22511\u201322521 (2023)","DOI":"10.1109\/CVPR52729.2023.02156"},{"key":"19_CR16","doi-asserted-by":"crossref","unstructured":"Li, Z., Wu, J., Koh, I., Tang, Y., Sun, L.: Image synthesis from layout with locality-aware mask adaption. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 13819\u201313828 (2021)","DOI":"10.1109\/ICCV48922.2021.01356"},{"key":"19_CR17","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"740","DOI":"10.1007\/978-3-319-10602-1_48","volume-title":"Computer Vision \u2013 ECCV 2014","author":"T-Y Lin","year":"2014","unstructured":"Lin, T.-Y., et al.: Microsoft COCO: common objects in context. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8693, pp. 740\u2013755. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10602-1_48"},{"key":"19_CR18","doi-asserted-by":"crossref","unstructured":"Mao, J., Wang, X.: Training-free location-aware text-to-image synthesis. arXiv preprint arXiv:2304.13427 (2023)","DOI":"10.1109\/ICIP49359.2023.10222616"},{"issue":"12","key":"19_CR19","doi-asserted-by":"publisher","first-page":"4695","DOI":"10.1109\/TIP.2012.2214050","volume":"21","author":"A Mittal","year":"2012","unstructured":"Mittal, A., Moorthy, A.K., Bovik, A.C.: No-reference image quality assessment in the spatial domain. IEEE Trans. Image Process. 21(12), 4695\u20134708 (2012)","journal-title":"IEEE Trans. Image Process."},{"key":"19_CR20","doi-asserted-by":"crossref","unstructured":"Mou, C., et al.: T2i-adapter: learning adapters to dig out more controllable ability for text-to-image diffusion models. arXiv preprint arXiv:2302.08453 (2023)","DOI":"10.1609\/aaai.v38i5.28226"},{"key":"19_CR21","doi-asserted-by":"crossref","unstructured":"Phung, Q., Ge, S., Huang, J.B.: Grounded text-to-image synthesis with attention refocusing. arXiv preprint arXiv:2306.05427 (2023)","DOI":"10.1109\/CVPR52733.2024.00758"},{"key":"19_CR22","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning (ICLR) (2021)"},{"key":"19_CR23","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., Ommer, B.: High-resolution image synthesis with latent diffusion models. In: IEEE \/ CVF Computer Vision and Pattern Recognition Conference (CVPR) (2022)","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"19_CR24","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"234","DOI":"10.1007\/978-3-319-24574-4_28","volume-title":"Medical Image Computing and Computer-Assisted Intervention \u2013 MICCAI 2015","author":"O Ronneberger","year":"2015","unstructured":"Ronneberger, O., Fischer, P., Brox, T.: U-Net: convolutional networks for biomedical image segmentation. In: Navab, N., Hornegger, J., Wells, W.M., Frangi, A.F. (eds.) MICCAI 2015. LNCS, vol. 9351, pp. 234\u2013241. Springer, Cham (2015). https:\/\/doi.org\/10.1007\/978-3-319-24574-4_28"},{"key":"19_CR25","unstructured":"Schuhmann, C., et al.: LAION-5B: an open large-scale dataset for training next generation image-text models. In: Conference on Neural Information Processing Systems (NeurIPS) (2022)"},{"key":"19_CR26","unstructured":"Simo, R.: Paint-with-words, implemented with stable diffusion (2023). https:\/\/github.com\/cloneofsimo\/paint-with-words-sd"},{"key":"19_CR27","unstructured":"Song, J., Meng, C., Ermon, S.: Denoising diffusion implicit models. In: International Conference on Learning Representations (ICLR) (2022)"},{"key":"19_CR28","unstructured":"Sushko, V., Sch\u00f6nfeld, E., Zhang, D., Gall, J., Schiele, B., Khoreva, A.: You only need adversarial supervision for semantic image synthesis. arXiv preprint arXiv:2012.04781 (2020)"},{"key":"19_CR29","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Advances in Neural Information Processing Systems, vol. 30 (2017)"},{"key":"19_CR30","doi-asserted-by":"crossref","unstructured":"Wang, X., Darrell, T., Rambhatla, S.S., Girdhar, R., Misra, I.: Instancediffusion: instance-level control for image generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6232\u20136242 (2024)","DOI":"10.1109\/CVPR52733.2024.00596"},{"key":"19_CR31","unstructured":"Xiao, J., Li, L., Lv, H., Wang, S., Huang, Q.: R &b: region and boundary aware zero-shot grounded text-to-image generation. arXiv preprint arXiv:2310.08872 (2023)"},{"key":"19_CR32","doi-asserted-by":"crossref","unstructured":"Yang, S., et al.: Maniqa: multi-dimension attention network for no-reference image quality assessment. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1191\u20131200 (2022)","DOI":"10.1109\/CVPRW56347.2022.00126"},{"key":"19_CR33","doi-asserted-by":"crossref","unstructured":"Zeng, Y., et al.: Scenecomposer: any-level semantic image synthesis. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 22468\u201322478 (2023)","DOI":"10.1109\/CVPR52729.2023.02152"},{"key":"19_CR34","doi-asserted-by":"crossref","unstructured":"Zhang, L., Rao, A., Agrawala, M.: Adding conditional control to text-to-image diffusion models. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3836\u20133847 (2023)","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"19_CR35","unstructured":"Zhao, S., et al.: Uni-controlnet: all-in-one control to text-to-image diffusion models. arXiv preprint arXiv:2305.16322 (2023)"},{"key":"19_CR36","doi-asserted-by":"crossref","unstructured":"Zheng, G., Zhou, X., Li, X., Qi, Z., Shan, Y., Li, X.: Layoutdiffusion: controllable diffusion model for layout-to-image generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 22490\u201322499 (2023)","DOI":"10.1109\/CVPR52729.2023.02154"},{"issue":"3","key":"19_CR37","doi-asserted-by":"publisher","first-page":"302","DOI":"10.1007\/s11263-018-1140-0","volume":"127","author":"B Zhou","year":"2019","unstructured":"Zhou, B., et al.: Semantic understanding of scenes through the ade20k dataset. Int. J. Comput. Vis. 127(3), 302\u2013321 (2019)","journal-title":"Int. J. Comput. Vis."},{"key":"19_CR38","doi-asserted-by":"crossref","unstructured":"Zhou, D., Li, Y., Ma, F., Yang, Z., Yang, Y.: Migc++: advanced multi-instance generation controller for image synthesis. IEEE Trans. Pattern Anal. Mach. Intell. (2024)","DOI":"10.1109\/CVPR52733.2024.00651"},{"key":"19_CR39","doi-asserted-by":"crossref","unstructured":"Zhu, P., Abdal, R., Qin, Y., Wonka, P.: Sean: image synthesis with semantic region-adaptive normalization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5104\u20135113 (2020)","DOI":"10.1109\/CVPR42600.2020.00515"}],"container-title":["Lecture Notes in Computer Science","Machine Learning and Knowledge Discovery in Databases. Research Track"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-05981-9_19","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,29]],"date-time":"2025-09-29T19:04:25Z","timestamp":1759172665000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-05981-9_19"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,9,23]]},"ISBN":["9783032059802","9783032059819"],"references-count":39,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-05981-9_19","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,9,23]]},"assertion":[{"value":"23 September 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECML PKDD","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Joint European Conference on Machine Learning and Knowledge Discovery in Databases","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Porto","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Portugal","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"15 September 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"19 September 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ecml2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ecmlpkdd.org\/2025\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}