{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,24]],"date-time":"2026-06-24T17:02:24Z","timestamp":1782320544814,"version":"3.54.5"},"publisher-location":"Singapore","reference-count":33,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819219469","type":"print"},{"value":"9789819219476","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,6,25]],"date-time":"2026-06-25T00:00:00Z","timestamp":1782345600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,6,25]],"date-time":"2026-06-25T00:00:00Z","timestamp":1782345600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-1947-6_17","type":"book-chapter","created":{"date-parts":[[2026,6,24]],"date-time":"2026-06-24T16:12:38Z","timestamp":1782317558000},"page":"203-214","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Semantic-Driven Object Placement via\u00a0Multi-modal Large Language Models"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-8076-3845","authenticated-orcid":false,"given":"Xuzheng","family":"Chen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-9604-2823","authenticated-orcid":false,"given":"Gang","family":"Shen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-7913-7920","authenticated-orcid":false,"given":"Rundong","family":"Cao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8295-2520","authenticated-orcid":false,"given":"Guojie","family":"Song","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,6,25]]},"reference":[{"key":"17_CR1","unstructured":"Achiam, J., et al.: Gpt-4 technical report. arXiv preprint arXiv:2303.08774 (2023)"},{"key":"17_CR2","doi-asserted-by":"crossref","unstructured":"Cheng, T., Song, L., Ge, Y., Liu, W., Wang, X., Shan, Y.: Yolo-world: real-time open-vocabulary object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16901\u201316911 (2024)","DOI":"10.1109\/CVPR52733.2024.01599"},{"key":"17_CR3","doi-asserted-by":"crossref","unstructured":"Choi, M.J., Lim, J.J., Torralba, A., Willsky, A.S.: Exploiting hierarchical context on a large database of object categories. In: IEEE Computer Society Conference on Computer Vision and Pattern Recognition, pp. 129\u2013136. IEEE (2010)","DOI":"10.1109\/CVPR.2010.5540221"},{"key":"17_CR4","doi-asserted-by":"crossref","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: Bert: pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the 2019 conference of the North American chapter of the association for computational linguistics: human language technologies, volume 1 (long and short papers), pp. 4171\u20134186 (2019)","DOI":"10.18653\/v1\/N19-1423"},{"key":"17_CR5","doi-asserted-by":"crossref","unstructured":"Doersch, C., Gupta, A., Efros, A.A.: Unsupervised visual representation learning by context prediction. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 1422\u20131430 (2015)","DOI":"10.1109\/ICCV.2015.167"},{"key":"17_CR6","unstructured":"Dosovitskiy, A.: An image is worth 16x16 words: transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)"},{"key":"17_CR7","doi-asserted-by":"crossref","unstructured":"Fang, H.S., Sun, J., Wang, R., Gou, M., Li, Y.L., Lu, C.: Instaboost: boosting instance segmentation via probability map guided copy-pasting. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 682\u2013691 (2019)","DOI":"10.1109\/ICCV.2019.00077"},{"key":"17_CR8","doi-asserted-by":"crossref","unstructured":"Georgakis, G., Mousavian, A., Berg, A.C., Kosecka, J.: Synthesizing training data for object detection in indoor scenes. arXiv preprint arXiv:1702.07836 (2017)","DOI":"10.15607\/RSS.2017.XIII.043"},{"key":"17_CR9","unstructured":"Goodfellow, I., et al.: Generative adversarial nets. In: Advances in Neural Information Processing Systems, p. 27 (2014)"},{"key":"17_CR10","first-page":"6840","volume":"33","author":"J Ho","year":"2020","unstructured":"Ho, J., Jain, A., Abbeel, P.: Denoising diffusion probabilistic models. Adv. Neural. Inf. Process. Syst. 33, 6840\u20136851 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"17_CR11","doi-asserted-by":"publisher","first-page":"3","DOI":"10.1007\/s11263-008-0137-5","volume":"80","author":"D Hoiem","year":"2008","unstructured":"Hoiem, D., Efros, A.A., Hebert, M.: Putting objects in perspective. Int. J. Comput. Vision 80, 3\u201315 (2008)","journal-title":"Int. J. Comput. Vision"},{"key":"17_CR12","unstructured":"Hurst, A., et al.: Gpt-4o system card. arXiv preprint arXiv:2410.21276 (2024)"},{"key":"17_CR13","unstructured":"Kingma, D.P.: Auto-encoding variational bayes. arXiv preprint arXiv:1312.6114 (2013)"},{"key":"17_CR14","unstructured":"Lee, D., Liu, S., Gu, J., Liu, M.Y., Yang, M.H., Kautz, J.: Context-aware synthesis and placement of object instances. In: Advances in Neural Information Processing Systems, vol. 31 (2018)"},{"key":"17_CR15","unstructured":"Li, J., Li, D., Savarese, S., Hoi, S.: Blip-2: bootstrapping language-image pre-training with frozen image encoders and large language models. In: International Conference on Machine Learning, pp. 19730\u201319742. PMLR (2023)"},{"key":"17_CR16","unstructured":"Li, J., Li, D., Xiong, C., Hoi, S.: Blip: bootstrapping language-image pre-training for unified vision-language understanding and generation. In: International Conference on Machine Learning, pp. 12888\u201312900. PMLR (2022)"},{"key":"17_CR17","doi-asserted-by":"crossref","unstructured":"Lin, C.H., Yumer, E., Wang, O., Shechtman, E., Lucey, S.: St-gan: spatial transformer generative adversarial networks for image compositing. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 9455\u20139464 (2018)","DOI":"10.1109\/CVPR.2018.00985"},{"key":"17_CR18","doi-asserted-by":"crossref","unstructured":"Lin, T.Y., et al.: Microsoft coco: common objects in context. In: Computer Vision\u2013ECCV 2014: 13th European Conference, zurich, Switzerland, 6\u201312 September 2014, Proceedings, Part v 13, pp. 740\u2013755. Springer (2014)","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"17_CR19","unstructured":"Liu, L., et al.: Opa: object placement assessment dataset. arXiv preprint arXiv:2107.01889 (2021)"},{"key":"17_CR20","doi-asserted-by":"crossref","unstructured":"Niu, H., Hu, J., Lin, J., Zhang, S.: Eov-seg: efficient open-vocabulary panoptic segmentation. arXiv preprint arXiv:2412.08628 (2024)","DOI":"10.1609\/aaai.v39i6.32669"},{"key":"17_CR21","unstructured":"Niu, L., et al.: Making images real again: a comprehensive survey on deep image composition. arXiv preprint arXiv:2106.14490 (2021)"},{"key":"17_CR22","unstructured":"Niu, L., Liu, Q., Liu, Z., Li, J.: Fast object placement assessment. arXiv preprint arXiv:2205.14280 (2022)"},{"key":"17_CR23","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763. PmLR (2021)"},{"key":"17_CR24","doi-asserted-by":"crossref","unstructured":"Remez, T., Huang, J., Brown, M.: Learning to segment via cut-and-paste. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 37\u201352 (2018)","DOI":"10.1007\/978-3-030-01234-2_3"},{"key":"17_CR25","doi-asserted-by":"crossref","unstructured":"Tripathi, S., Chandra, S., Agrawal, A., Tyagi, A., Rehg, J.M., Chari, V.: Learning to generate synthetic data via compositing. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 461\u2013470 (2019)","DOI":"10.1109\/CVPR.2019.00055"},{"key":"17_CR26","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Advances in Neural Information Processing Systems, vol. 30 (2017)"},{"key":"17_CR27","unstructured":"Wang, H., Wang, Q., Yang, F., Zhang, W., Zuo, W.: Data augmentation for object detection via progressive and selective instance-switching. arXiv preprint arXiv:1906.00358 (2019)"},{"key":"17_CR28","doi-asserted-by":"crossref","unstructured":"Wang, Y., Feng, Y., Wu, J., Xu, H., Zheng, J.: Ca-gan: Object placement via coalescing attention based generative adversarial network. In: 2023 IEEE International Conference on Multimedia and Expo (ICME), pp. 2375\u20132380. IEEE (2023)","DOI":"10.1109\/ICME55011.2023.00405"},{"key":"17_CR29","doi-asserted-by":"crossref","unstructured":"Zhang, L., Wen, T., Min, J., Wang, J., Han, D., Shi, J.: Learning object placement by inpainting for compositional data augmentation. In: Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, 23\u201328 August 2020, Proceedings, Part XIII 16, pp. 566\u2013581. Springer (2020)","DOI":"10.1007\/978-3-030-58601-0_34"},{"key":"17_CR30","unstructured":"Zhang, S., et al.: Interactive object placement with reinforcement learning (2023)"},{"key":"17_CR31","doi-asserted-by":"publisher","first-page":"79","DOI":"10.1007\/s41095-020-0158-8","volume":"6","author":"SH Zhang","year":"2020","unstructured":"Zhang, S.H., Zhou, Z.P., Liu, B., Dong, X., Hall, P.: What and where: a context-based recommendation system for object insertion. Comput. Vis. Media 6, 79\u201393 (2020)","journal-title":"Comput. Vis. Media"},{"key":"17_CR32","doi-asserted-by":"crossref","unstructured":"Zhou, S., Liu, L., Niu, L., Zhang, L.: Learning object placement via dual-path graph completion. In: European Conference on Computer Vision, pp. 373\u2013389. Springer (2022)","DOI":"10.1007\/978-3-031-19790-1_23"},{"key":"17_CR33","doi-asserted-by":"crossref","unstructured":"Zhu, S., Lin, Z., Cohen, S., Kuen, J., Zhang, Z., Chen, C.: Topnet: transformer-based object placement network for image compositing. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1838\u20131847 (2023)","DOI":"10.1109\/CVPR52729.2023.00183"}],"container-title":["Lecture Notes in Computer Science","Data Science: Foundations and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-1947-6_17","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,24]],"date-time":"2026-06-24T16:13:27Z","timestamp":1782317607000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-1947-6_17"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,25]]},"ISBN":["9789819219469","9789819219476"],"references-count":33,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-1947-6_17","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,6,25]]},"assertion":[{"value":"25 June 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"PAKDD","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Pacific-Asia Conference on Knowledge Discovery and Data Mining","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Hong Kong","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"9 June 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"12 June 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"pakdd2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.pakdd2026.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}