{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T15:56:40Z","timestamp":1783439800910,"version":"3.54.6"},"publisher-location":"Cham","reference-count":74,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031730320","type":"print"},{"value":"9783031730337","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T00:00:00Z","timestamp":1730332800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T00:00:00Z","timestamp":1730332800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-73033-7_27","type":"book-chapter","created":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T00:03:55Z","timestamp":1730333035000},"page":"476-495","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":15,"title":["Thinking Outside the\u00a0BBox: Unconstrained Generative Object Compositing"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5642-8282","authenticated-orcid":false,"given":"Gemma","family":"Canet Tarr\u00e9s","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1154-9907","authenticated-orcid":false,"given":"Zhe","family":"Lin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0466-9548","authenticated-orcid":false,"given":"Zhifei","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9954-6294","authenticated-orcid":false,"given":"Jianming","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yizhi","family":"Song","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-0310-6933","authenticated-orcid":false,"given":"Dan","family":"Ruta","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3898-0596","authenticated-orcid":false,"given":"Andrew","family":"Gilbert","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3580-4685","authenticated-orcid":false,"given":"John","family":"Collomosse","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-8104-4100","authenticated-orcid":false,"given":"Soo Ye","family":"Kim","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,10,31]]},"reference":[{"key":"27_CR1","unstructured":"https:\/\/pixabay.com\/"},{"key":"27_CR2","doi-asserted-by":"crossref","unstructured":"Alaluf, Y., Tov, O., Mokady, R., Gal, R., Bermano, A.: Hyperstyle: stylegan inversion with hypernetworks for real image editing. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18511\u201318521 (2022)","DOI":"10.1109\/CVPR52688.2022.01796"},{"issue":"4","key":"27_CR3","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3592450","volume":"42","author":"O Avrahami","year":"2023","unstructured":"Avrahami, O., Fried, O., Lischinski, D.: Blended latent diffusion. ACM Trans. Graph. (TOG) 42(4), 1\u201311 (2023)","journal-title":"ACM Trans. Graph. (TOG)"},{"key":"27_CR4","doi-asserted-by":"crossref","unstructured":"Avrahami, O., Lischinski, D., Fried, O.: Blended diffusion for text-driven editing of natural images. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18208\u201318218 (2022)","DOI":"10.1109\/CVPR52688.2022.01767"},{"key":"27_CR5","doi-asserted-by":"publisher","first-page":"2570","DOI":"10.1007\/s11263-020-01336-9","volume":"128","author":"S Azadi","year":"2020","unstructured":"Azadi, S., Pathak, D., Ebrahimi, S., Darrell, T.: Compositional GAN: learning image-conditional binary composition. Int. J. Comput. Vision 128, 2570\u20132585 (2020)","journal-title":"Int. J. Comput. Vision"},{"key":"27_CR6","unstructured":"Bau, D., et al.: Semantic photo manipulation with a generative image prior. arXiv preprint arXiv:2005.07727 (2020)"},{"key":"27_CR7","unstructured":"Chen, H., Zhang, Y., Wang, X., Duan, X., Zhou, Y., Zhu, W.: Disenbooth: disentangled parameter-efficient tuning for subject-driven text-to-image generation. arXiv preprint arXiv:2305.03374 (2023)"},{"key":"27_CR8","unstructured":"Chen, W., et al.: Subject-driven text-to-image generation via apprenticeship learning. arXiv preprint arXiv:2304.00186 (2023)"},{"key":"27_CR9","doi-asserted-by":"crossref","unstructured":"Chen, X., Huang, L., Liu, Y., Shen, Y., Zhao, D., Zhao, H.: Anydoor: zero-shot object-level image customization. arXiv preprint arXiv:2307.09481 (2023)","DOI":"10.1109\/CVPR52733.2024.00630"},{"key":"27_CR10","doi-asserted-by":"crossref","unstructured":"Dvornik, N., Mairal, J., Schmid, C.: Modeling visual context is key to augmenting object detection datasets. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 364\u2013380 (2018)","DOI":"10.1007\/978-3-030-01258-8_23"},{"issue":"6","key":"27_CR11","doi-asserted-by":"publisher","first-page":"2014","DOI":"10.1109\/TPAMI.2019.2961896","volume":"43","author":"N Dvornik","year":"2019","unstructured":"Dvornik, N., Mairal, J., Schmid, C.: On the importance of visual context for data augmentation in scene understanding. IEEE Trans. Pattern Anal. Mach. Intell. 43(6), 2014\u20132028 (2019)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"27_CR12","doi-asserted-by":"crossref","unstructured":"Fang, H.S., Sun, J., Wang, R., Gou, M., Li, Y.L., Lu, C.: Instaboost: boosting instance segmentation via probability map guided copy-pasting. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 682\u2013691 (2019)","DOI":"10.1109\/ICCV.2019.00077"},{"key":"27_CR13","unstructured":"Fu, S., et al.: Dreamsim: learning new dimensions of human visual similarity using synthetic data. arXiv preprint arXiv:2306.09344 (2023)"},{"key":"27_CR14","unstructured":"Gal, R., et al.: An image is worth one word: personalizing text-to-image generation using textual inversion. arXiv preprint arXiv:2208.01618 (2022)"},{"key":"27_CR15","doi-asserted-by":"crossref","unstructured":"Gu, S., Bao, J., Yang, H., Chen, D., Wen, F., Yuan, L.: Mask-guided portrait editing with conditional GANs. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3436\u20133445 (2019)","DOI":"10.1109\/CVPR.2019.00355"},{"key":"27_CR16","unstructured":"Hertz, A., Mokady, R., Tenenbaum, J., Aberman, K., Pritch, Y., Cohen-Or, D.: Prompt-to-prompt image editing with cross attention control. arXiv preprint arXiv:2208.01626 (2022)"},{"key":"27_CR17","doi-asserted-by":"crossref","unstructured":"Hessel, J., Holtzman, A., Forbes, M., Le\u00a0Bras, R., Choi, Y.: Clipscore: a reference-free evaluation metric for image captioning. In: Proceedings of the Conference on Empirical Methods in Natural Language Processing, pp. 7514\u20137528 (2021)","DOI":"10.18653\/v1\/2021.emnlp-main.595"},{"key":"27_CR18","first-page":"6840","volume":"33","author":"J Ho","year":"2020","unstructured":"Ho, J., Jain, A., Abbeel, P.: Denoising diffusion probabilistic models. Adv. Neural. Inf. Process. Syst. 33, 6840\u20136851 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"27_CR19","unstructured":"Jia, X., et al.: Taming encoder for zero fine-tuning image customization with text-to-image diffusion models. arXiv preprint arXiv:2304.02642 (2023)"},{"issue":"6","key":"27_CR20","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/2070781.2024191","volume":"30","author":"K Karsch","year":"2011","unstructured":"Karsch, K., Hedau, V., Forsyth, D., Hoiem, D.: Rendering synthetic objects into legacy photographs. ACM Trans. Graph. (TOG) 30(6), 1\u201312 (2011)","journal-title":"ACM Trans. Graph. (TOG)"},{"key":"27_CR21","doi-asserted-by":"crossref","unstructured":"Kawar, B., et al.: Imagic: text-based real image editing with diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6007\u20136017 (2023)","DOI":"10.1109\/CVPR52729.2023.00582"},{"issue":"4","key":"27_CR22","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/2601097.2601209","volume":"33","author":"N Kholgade","year":"2014","unstructured":"Kholgade, N., Simon, T., Efros, A., Sheikh, Y.: 3D object manipulation in a single photograph using stock 3d models. ACM Trans. Graph. (TOG) 33(4), 1\u201312 (2014)","journal-title":"ACM Trans. Graph. (TOG)"},{"key":"27_CR23","doi-asserted-by":"crossref","unstructured":"Kim, G., Kwon, T., Ye, J.C.: Diffusionclip: text-guided diffusion models for robust image manipulation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2426\u20132435 (2022)","DOI":"10.1109\/CVPR52688.2022.00246"},{"key":"27_CR24","unstructured":"Kim, K., Park, S., Lee, J., Choo, J.: Reference-based image composition with sketch via structure-aware diffusion model. arXiv preprint arXiv:2304.09748 (2023)"},{"key":"27_CR25","doi-asserted-by":"crossref","unstructured":"Kulal, S., et al.: Putting people in their place: affordance-aware human insertion into scenes. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 17089\u201317099 (2023)","DOI":"10.1109\/CVPR52729.2023.01639"},{"key":"27_CR26","doi-asserted-by":"crossref","unstructured":"Lalonde, J.F., Hoiem, D., Efros, A.A., Rother, C., Winn, J., Criminisi, A.: Photo clip art. ACM Trans. Graph. (TOG) 26(3), 3-es (2007)","DOI":"10.1145\/1276377.1276381"},{"key":"27_CR27","unstructured":"Lee, D., Liu, S., Gu, J., Liu, M.Y., Yang, M.H., Kautz, J.: Context-aware synthesis and placement of object instances. In: Advances in Neural Information Processing Systems, vol. 31 (2018)"},{"key":"27_CR28","unstructured":"Li, D., Li, J., Hoi, S.C.: Blip-diffusion: pre-trained subject representation for controllable text-to-image generation and editing. arXiv preprint arXiv:2305.14720 (2023)"},{"key":"27_CR29","unstructured":"Li, T., Ku, M., Wei, C., Chen, W.: Dreamedit: subject-driven image editing. arXiv preprint arXiv:2306.12624 (2023)"},{"key":"27_CR30","doi-asserted-by":"crossref","unstructured":"Lin, C.H., Yumer, E., Wang, O., Shechtman, E., Lucey, S.: ST-GAN: spatial transformer generative adversarial networks for image compositing. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 9455\u20139464 (2018)","DOI":"10.1109\/CVPR.2018.00985"},{"key":"27_CR31","first-page":"16331","volume":"34","author":"H Ling","year":"2021","unstructured":"Ling, H., Kreis, K., Li, D., Kim, S.W., Torralba, A., Fidler, S.: Editgan: high-precision semantic image editing. Adv. Neural. Inf. Process. Syst. 34, 16331\u201316345 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"27_CR32","doi-asserted-by":"crossref","unstructured":"Liu, D., Long, C., Zhang, H., Yu, H., Dong, X., Xiao, C.: Arshadowgan: shadow generative adversarial network for augmented reality in single light scenes. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8139\u20138148 (2020)","DOI":"10.1109\/CVPR42600.2020.00816"},{"key":"27_CR33","unstructured":"Liu, L., et al.: OPA: object placement assessment dataset. arXiv preprint arXiv:2107.01889 (2021)"},{"key":"27_CR34","doi-asserted-by":"crossref","unstructured":"Liu, X., et al.: More control for free! image synthesis with semantic diffusion guidance. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 289\u2013299 (2023)","DOI":"10.1109\/WACV56688.2023.00037"},{"key":"27_CR35","unstructured":"Lu, L., Zhang, B., Niu, L.: Dreamcom: finetuning text-guided inpainting model for image composition. arXiv preprint arXiv:2309.15508 (2023)"},{"key":"27_CR36","doi-asserted-by":"crossref","unstructured":"Lu, S., Liu, Y., Kong, A.W.K.: TF-icon: diffusion-based training-free cross-domain image composition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2294\u20132305 (2023)","DOI":"10.1109\/ICCV51070.2023.00218"},{"key":"27_CR37","doi-asserted-by":"crossref","unstructured":"Lugmayr, A., Danelljan, M., Romero, A., Yu, F., Timofte, R., Van\u00a0Gool, L.: Repaint: inpainting using denoising diffusion probabilistic models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11461\u201311471 (2022)","DOI":"10.1109\/CVPR52688.2022.01117"},{"key":"27_CR38","unstructured":"Meng, C., et al.: Sdedit: guided image synthesis and editing with stochastic differential equations. arXiv preprint arXiv:2108.01073 (2021)"},{"key":"27_CR39","doi-asserted-by":"crossref","unstructured":"Miao, J., et al.: Large-scale video panoptic segmentation in the wild: a benchmark. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 21033\u201321043 (2022)","DOI":"10.1109\/CVPR52688.2022.02036"},{"key":"27_CR40","unstructured":"Niu, L., Liu, Q., Liu, Z., Li, J.: Fast object placement assessment. arXiv preprint arXiv:2205.14280 (2022)"},{"key":"27_CR41","unstructured":"Oquab, M., et al.: Dinov2: learning robust visual features without supervision. arXiv preprint arXiv:2304.07193 (2023)"},{"key":"27_CR42","unstructured":"Qi, L., et al.: Fine-grained entity segmentation. arXiv preprint arXiv:2211.05776 (2022)"},{"key":"27_CR43","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PMLR (2021)"},{"key":"27_CR44","doi-asserted-by":"crossref","unstructured":"Remez, T., Huang, J., Brown, M.: Learning to segment via cut-and-paste. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 37\u201352 (2018)","DOI":"10.1007\/978-3-030-01234-2_3"},{"key":"27_CR45","doi-asserted-by":"crossref","unstructured":"Richardson, E., et al.: Encoding in style: a stylegan encoder for image-to-image translation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2287\u20132296 (2021)","DOI":"10.1109\/CVPR46437.2021.00232"},{"key":"27_CR46","doi-asserted-by":"crossref","unstructured":"Rombach, R., Blattmann, A., Lorenz, D., Esser, P., Ommer, B.: High-resolution image synthesis with latent diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10684\u201310695 (2022)","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"27_CR47","doi-asserted-by":"crossref","unstructured":"Ruiz, N., Li, Y., Jampani, V., Pritch, Y., Rubinstein, M., Aberman, K.: Dreambooth: fine tuning text-to-image diffusion models for subject-driven generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 22500\u201322510 (2023)","DOI":"10.1109\/CVPR52729.2023.02155"},{"key":"27_CR48","unstructured":"Seyfioglu, M.S., Bouyarmane, K., Kumar, S., Tavanaei, A., Tutar, I.B.: Diffuse to choose: enriching image conditioned inpainting in latent diffusion models for virtual try-all. arXiv preprint arXiv:2401.13795 (2024)"},{"key":"27_CR49","doi-asserted-by":"crossref","unstructured":"Shi, J., Xiong, W., Lin, Z., Jung, H.J.: Instantbooth: personalized text-to-image generation without test-time finetuning. arXiv preprint arXiv:2304.03411 (2023)","DOI":"10.1109\/CVPR52733.2024.00816"},{"key":"27_CR50","unstructured":"Song, Y., et al.: Objectstitch: generative object compositing. arXiv preprint arXiv:2212.00932 (2022)"},{"key":"27_CR51","doi-asserted-by":"crossref","unstructured":"Song, Y., et al.: Imprint: generative object compositing by learning identity-preserving representation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 8048\u20138058 (2024)","DOI":"10.1109\/CVPR52733.2024.00769"},{"key":"27_CR52","doi-asserted-by":"crossref","unstructured":"Tan, F., Bernier, C., Cohen, B., Ordonez, V., Barnes, C.: Where and who? Automatic semantic-aware person composition. In: 2018 IEEE Winter Conference on Applications of Computer Vision (WACV), pp. 1519\u20131528. IEEE (2018)","DOI":"10.1109\/WACV.2018.00170"},{"key":"27_CR53","doi-asserted-by":"crossref","unstructured":"Tripathi, S., Chandra, S., Agrawal, A., Tyagi, A., Rehg, J.M., Chari, V.: Learning to generate synthetic data via compositing. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 461\u2013470 (2019)","DOI":"10.1109\/CVPR.2019.00055"},{"key":"27_CR54","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"252","DOI":"10.1007\/978-3-030-66823-5_15","volume-title":"Computer Vision \u2013 ECCV 2020 Workshops","author":"A Volokitin","year":"2020","unstructured":"Volokitin, A., Susmelj, I., Agustsson, E., Van Gool, L., Timofte, R.: Efficiently detecting plausible locations for object placement using masked convolutions. In: Bartoli, A., Fusiello, A. (eds.) ECCV 2020. LNCS, vol. 12538, pp. 252\u2013266. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-66823-5_15"},{"key":"27_CR55","unstructured":"Wang, T., et al.: Pretraining is all you need for image-to-image translation. arXiv preprint arXiv:2205.12952 (2022)"},{"issue":"3","key":"27_CR56","first-page":"3259","volume":"45","author":"T Wang","year":"2022","unstructured":"Wang, T., Hu, X., Heng, P.A., Fu, C.W.: Instance shadow detection with a single-stage detector. IEEE Trans. Pattern Anal. Mach. Intell. 45(3), 3259\u20133273 (2022)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"27_CR57","doi-asserted-by":"crossref","unstructured":"Wang, X., Yu, K., Dong, C., Tang, X., Loy, C.C.: Deep network interpolation for continuous imagery effect transition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1692\u20131701 (2019)","DOI":"10.1109\/CVPR.2019.00179"},{"key":"27_CR58","doi-asserted-by":"crossref","unstructured":"Xie, S., Zhang, Z., Lin, Z., Hinz, T., Zhang, K.: Smartbrush: text and shape guided object inpainting with diffusion model. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 22428\u201322437 (2023)","DOI":"10.1109\/CVPR52729.2023.02148"},{"key":"27_CR59","doi-asserted-by":"crossref","unstructured":"Xu, N., Price, B., Cohen, S., Huang, T.: Deep image matting. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2970\u20132979 (2017)","DOI":"10.1109\/CVPR.2017.41"},{"key":"27_CR60","doi-asserted-by":"crossref","unstructured":"Xu, N., et al.: Youtube-VOS: sequence-to-sequence video object segmentation. In: Proceedings of the European Conference on Computer Vision, pp. 585\u2013601 (2018)","DOI":"10.1007\/978-3-030-01228-1_36"},{"key":"27_CR61","series-title":"LNCS","doi-asserted-by":"publisher","first-page":"300","DOI":"10.1007\/978-3-031-20071-7_18","volume-title":"ECCV 2022","author":"B Xue","year":"2022","unstructured":"Xue, B., Ran, S., Chen, Q., Jia, R., Zhao, B., Tang, X.: DCCF: deep comprehensible color filter learning framework for high-resolution image harmonization. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13667, pp. 300\u2013316. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-20071-7_18"},{"key":"27_CR62","doi-asserted-by":"crossref","unstructured":"Yang, B., et al.: Paint by example: exemplar-based image editing with diffusion models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18381\u201318391 (2023)","DOI":"10.1109\/CVPR52729.2023.01763"},{"key":"27_CR63","doi-asserted-by":"crossref","unstructured":"Yang, H., Zhang, R., Guo, X., Liu, W., Zuo, W., Luo, P.: Towards photo-realistic virtual try-on by adaptively generating-preserving image content. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7850\u20137859 (2020)","DOI":"10.1109\/CVPR42600.2020.00787"},{"key":"27_CR64","unstructured":"Yu, T., et al.: Inpaint anything: segment anything meets image inpainting. arXiv preprint arXiv:2304.06790 (2023)"},{"key":"27_CR65","doi-asserted-by":"crossref","unstructured":"Yu, X., et al.: Mvimgnet: a large-scale dataset of multi-view images. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9150\u20139161 (2023)","DOI":"10.1109\/CVPR52729.2023.00883"},{"key":"27_CR66","doi-asserted-by":"crossref","unstructured":"Yuan, Z., Cao, M., Wang, X., Qi, Z., Yuan, C., Shan, Y.: Customnet: zero-shot object customization with variable-viewpoints in text-to-image diffusion models. arXiv preprint arXiv:2310.19784 (2023)","DOI":"10.1145\/3664647.3681396"},{"key":"27_CR67","doi-asserted-by":"crossref","unstructured":"Zhan, F., Zhu, H., Lu, S.: Spatial fusion GAN for image synthesis. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3653\u20133662 (2019)","DOI":"10.1109\/CVPR.2019.00377"},{"key":"27_CR68","unstructured":"Zhang, B., et al.: Controlcom: controllable image composition using diffusion model. arXiv preprint arXiv:2308.10040 (2023)"},{"key":"27_CR69","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"566","DOI":"10.1007\/978-3-030-58601-0_34","volume-title":"Computer Vision \u2013 ECCV 2020","author":"L Zhang","year":"2020","unstructured":"Zhang, L., Wen, T., Min, J., Wang, J., Han, D., Shi, J.: Learning object placement by inpainting for compositional data augmentation. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12358, pp. 566\u2013581. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58601-0_34"},{"key":"27_CR70","doi-asserted-by":"publisher","first-page":"79","DOI":"10.1007\/s41095-020-0158-8","volume":"6","author":"SH Zhang","year":"2020","unstructured":"Zhang, S.H., Zhou, Z.P., Liu, B., Dong, X., Hall, P.: What and where: a context-based recommendation system for object insertion. Comput. Vis. Media 6, 79\u201393 (2020)","journal-title":"Comput. Vis. Media"},{"key":"27_CR71","doi-asserted-by":"crossref","unstructured":"Zhang, X., Guo, J., Yoo, P., Matsuo, Y., Iwasawa, Y.: Paste, inpaint and harmonize via denoising: subject-driven image editing with pre-trained diffusion model. arXiv preprint arXiv:2306.07596 (2023)","DOI":"10.1109\/ICASSP48485.2024.10448510"},{"key":"27_CR72","series-title":"LNCS","doi-asserted-by":"publisher","first-page":"277","DOI":"10.1007\/978-3-031-19787-1_16","volume-title":"ECCV 2022","author":"H Zheng","year":"2022","unstructured":"Zheng, H., et al.: Image inpainting with cascaded modulation GAN and object-aware training. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13676, pp. 277\u2013296. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19787-1_16"},{"key":"27_CR73","series-title":"LNCS","doi-asserted-by":"publisher","first-page":"373","DOI":"10.1007\/978-3-031-19790-1_23","volume-title":"ECCV 2022","author":"S Zhou","year":"2022","unstructured":"Zhou, S., Liu, L., Niu, L., Zhang, L.: Learning object placement via dual-path graph completion. In: Avidan, S., Brostow, G., Ciss\u00e9, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022. LNCS, vol. 13677, pp. 373\u2013389. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-19790-1_23"},{"key":"27_CR74","doi-asserted-by":"crossref","unstructured":"Zhu, S., Lin, Z., Cohen, S., Kuen, J., Zhang, Z., Chen, C.: Topnet: transformer-based object placement network for image compositing. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1838\u20131847 (2023)","DOI":"10.1109\/CVPR52729.2023.00183"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-73033-7_27","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T00:41:48Z","timestamp":1730335308000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-73033-7_27"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,31]]},"ISBN":["9783031730320","9783031730337"],"references-count":74,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-73033-7_27","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,31]]},"assertion":[{"value":"31 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}