{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,11]],"date-time":"2026-06-11T15:02:52Z","timestamp":1781190172075,"version":"3.54.1"},"publisher-location":"Singapore","reference-count":41,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819596935","type":"print"},{"value":"9789819596942","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-981-95-9694-2_7","type":"book-chapter","created":{"date-parts":[[2026,6,11]],"date-time":"2026-06-11T14:29:42Z","timestamp":1781188182000},"page":"82-96","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["FashionAtlas: Enhancing Semantics and\u00a0Control in\u00a0Multimodal Fashion Image Editing"],"prefix":"10.1007","author":[{"given":"Enzhen","family":"Gu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jinpei","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yingjie","family":"Shi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,5,1]]},"reference":[{"key":"7_CR1","unstructured":"Bai, S., Cai, Y., Chen, R., et\u00a0al.: Qwen3-VL technical report (2025). https:\/\/arxiv.org\/abs\/2511.21631"},{"key":"7_CR2","doi-asserted-by":"crossref","unstructured":"Baldrati, A., Morelli, D., Cartella, G., Cornia, M., Bertini, M., Cucchiara, R.: Multimodal garment designer: human-centric latent diffusion models for fashion image editing. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 23393\u201323402 (2023)","DOI":"10.1109\/ICCV51070.2023.02138"},{"key":"7_CR3","unstructured":"Batifol, S., et\u00a0al.: FLUX. 1 Kontext: flow matching for in-context image generation and editing in latent space. arXiv e-prints, arXiv\u20132506 (2025)"},{"key":"7_CR4","doi-asserted-by":"crossref","unstructured":"Bian, S., et al.: ChatGarment: garment estimation, generation and editing via large language models. In: Proceedings of the Computer Vision and Pattern Recognition Conference, pp. 2924\u20132934 (2025)","DOI":"10.1109\/CVPR52734.2025.00278"},{"key":"7_CR5","doi-asserted-by":"crossref","unstructured":"Brooks, T., Holynski, A., Efros, A.A.: InstructPix2Pix: learning to follow image editing instructions. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18392\u201318402 (2023)","DOI":"10.1109\/CVPR52729.2023.01764"},{"key":"7_CR6","unstructured":"Cheng, D., Shi, Y., Sun, S., Zhang, J., Wang, W., Liu, Y.: ControlEdit: a MultiModal local clothing image editing method. arXiv preprint arXiv:2409.14720 (2024)"},{"key":"7_CR7","doi-asserted-by":"crossref","unstructured":"Choi, S., Park, S., Lee, M., Choo, J.: VITON-HD: high-resolution virtual try-on via misalignment-aware normalization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14131\u201314140 (2021)","DOI":"10.1109\/CVPR46437.2021.01391"},{"key":"7_CR8","doi-asserted-by":"crossref","unstructured":"Ge, Y., Zhang, R., Wang, X., Tang, X., Luo, P.: DeepFashion2: a versatile benchmark for detection, pose estimation, segmentation and re-identification of clothing images. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5337\u20135345 (2019)","DOI":"10.1109\/CVPR.2019.00548"},{"key":"7_CR9","doi-asserted-by":"crossref","unstructured":"Girella, F., Talon, D., Liu, Z., Ruan, Z., Wang, Y., Cristani, M.: Lots of fashion! Multi-conditioning for image generation via sketch-text pairing. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 19711\u201319720 (2025)","DOI":"10.1109\/ICCV51701.2025.01833"},{"key":"7_CR10","doi-asserted-by":"crossref","unstructured":"Han, X., et al.: Automatic spatially-aware fashion concept discovery. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 1463\u20131471 (2017)","DOI":"10.1109\/ICCV.2017.163"},{"key":"7_CR11","doi-asserted-by":"crossref","unstructured":"Han, X., Wu, Z., Jiang, Y.G., Davis, L.S.: Learning fashion compatibility with bidirectional LSTMS. In: Proceedings of the 25th ACM International Conference on Multimedia, pp. 1078\u20131086 (2017)","DOI":"10.1145\/3123266.3123394"},{"key":"7_CR12","unstructured":"Ho, J., Jain, A., Abbeel, P.: Denoising diffusion probabilistic models. In: Advances in Neural Information Processing Systems, vol. 33, pp. 6840\u20136851 (2020)"},{"key":"7_CR13","unstructured":"Hu, E.J., et al.: LoRA: low-rank adaptation of large language models. In: ICLR, vol. 1, no. 2, p. 3 (2022)"},{"key":"7_CR14","doi-asserted-by":"crossref","unstructured":"Huang, Y., et al.: Diffusion model-based image editing: a survey. IEEE Trans. Pattern Anal. Mach. Intell. (2025)","DOI":"10.1109\/TPAMI.2025.3541625"},{"key":"7_CR15","doi-asserted-by":"crossref","unstructured":"Huang, Y., et\u00a0al.: SmartEdit: exploring complex instruction-based image editing with multimodal large language models. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8362\u20138371 (2024)","DOI":"10.1109\/CVPR52733.2024.00799"},{"key":"7_CR16","doi-asserted-by":"crossref","unstructured":"Jia, M., et al.: Fashionpedia: ontology, segmentation, and an attribute localization dataset. In: European Conference on Computer Vision, pp. 316\u2013332. Springer (2020)","DOI":"10.1007\/978-3-030-58452-8_19"},{"key":"7_CR17","unstructured":"Jiang, T., et al.: RTMPose: real-time multi-person pose estimation based on MMPose. arXiv preprint arXiv:2303.07399 (2023)"},{"key":"7_CR18","doi-asserted-by":"crossref","unstructured":"Jiang, Y., Yang, S., Qiu, H., Wu, W., Loy, C.C., Liu, Z.: Text2Human: text-driven controllable human image generation. ACM Trans. Graph. (TOG) 41(4), 1\u201311 (2022)","DOI":"10.1145\/3528223.3530104"},{"key":"7_CR19","unstructured":"Khanam, R., Hussain, M.: YOLOv11: an overview of the key architectural enhancements. arXiv preprint arXiv:2410.17725 (2024)"},{"key":"7_CR20","unstructured":"Kim, K., Park, S., Lee, J., Choo, J.: Reference-based image composition with sketch via structure-aware diffusion model. arXiv preprint arXiv:2304.09748 (2023)"},{"key":"7_CR21","doi-asserted-by":"crossref","unstructured":"Korosteleva, M., Sorkine-Hornung, O.: GarmentCode: programming parametric sewing patterns. ACM Trans. Graph. (TOG) 42(6), 1\u201315 (2023)","DOI":"10.1145\/3618351"},{"key":"7_CR22","doi-asserted-by":"crossref","unstructured":"Li, L., Sleem, L., Nichil, G., State, R., et al.: Exploring the impact of temperature on large language models: hot or cold? Procedia Comput. Sci. 264, 242\u2013251 (2025)","DOI":"10.1016\/j.procs.2025.07.135"},{"key":"7_CR23","doi-asserted-by":"crossref","unstructured":"Li, M., et al.: ControlNet++: improving conditional controls with efficient consistency feedback: project page: liming-ai. github. io\/controlnet_plus_plus. In: European Conference on Computer Vision, pp. 129\u2013147. Springer (2024)","DOI":"10.1007\/978-3-031-72667-5_8"},{"key":"7_CR24","unstructured":"Li, S., Singh, H., Grover, A.: InstructAny2Pix: flexible visual editing via multimodal instruction following. arXiv preprint arXiv:2312.06738 (2023)"},{"key":"7_CR25","unstructured":"Lipman, Y., Chen, R.T., Ben-Hamu, H., Nickel, M., Le, M.: Flow matching for generative modeling. arXiv preprint arXiv:2210.02747 (2022)"},{"key":"7_CR26","doi-asserted-by":"crossref","unstructured":"Liu, Z., Luo, P., Qiu, S., Wang, X., Tang, X.: DeepFashion: powering robust clothes recognition and retrieval with rich annotations. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1096\u20131104 (2016)","DOI":"10.1109\/CVPR.2016.124"},{"key":"7_CR27","doi-asserted-by":"crossref","unstructured":"Ma, Y., Jia, J., Zhou, S., Fu, J., Liu, Y., Tong, Z.: Towards better understanding the clothing fashion styles: a multimodal deep learning approach. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a031 (2017)","DOI":"10.1609\/aaai.v31i1.10509"},{"key":"7_CR28","doi-asserted-by":"crossref","unstructured":"Morelli, D., Fincato, M., Cornia, M., Landi, F., Cesari, F., Cucchiara, R.: Dress code: high-resolution multi-category virtual try-on. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2231\u20132235 (2022)","DOI":"10.1007\/978-3-031-20074-8_20"},{"key":"7_CR29","doi-asserted-by":"crossref","unstructured":"Ouyang, Z., Li, Z., Hou, Q.: K-LoRA: unlocking training-free fusion of any subject and style LoRAs. arXiv preprint arXiv:2502.18461 (2025)","DOI":"10.1109\/CVPR52734.2025.01217"},{"key":"7_CR30","doi-asserted-by":"crossref","unstructured":"Peebles, W., Xie, S.: Scalable diffusion models with transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 4195\u20134205 (2023)","DOI":"10.1109\/ICCV51070.2023.00387"},{"key":"7_CR31","unstructured":"Rostamzadeh, N., et al.: Fashion-GEN: the generative fashion dataset and challenge. arXiv preprint arXiv:1806.08317 (2018)"},{"key":"7_CR32","doi-asserted-by":"crossref","unstructured":"Sanguigni, F., Morelli, D., Cornia, M., Cucchiara, R.: Fashion-RAG: multimodal fashion image editing via retrieval-augmented generation. arXiv preprint arXiv:2504.14011 (2025)","DOI":"10.1109\/IJCNN64981.2025.11229186"},{"key":"7_CR33","doi-asserted-by":"crossref","unstructured":"Sengupta, A., Jin, F., Zhang, R., Cao, S.: mm-Pose: real-time human skeletal posture estimation using mmWave radars and CNNs. IEEE Sens. J. 20(17), 10032\u201310044 (2020)","DOI":"10.1109\/JSEN.2020.2991741"},{"key":"7_CR34","doi-asserted-by":"crossref","unstructured":"Shi, W., Wong, W., Zou, X.: Generative ai in fashion: overview. ACM Trans. Intell. Syst. Technol. 16(4), 1\u201373 (2025)","DOI":"10.1145\/3718098"},{"key":"7_CR35","doi-asserted-by":"crossref","unstructured":"Wang, X., Cheng, Z.Q., Wang, J., Peng, X.: DPDEdit: detail-preserved diffusion models for multimodal fashion image editing. arXiv preprint arXiv:2409.01086 (2024)","DOI":"10.1109\/ICME59968.2025.11209148"},{"key":"7_CR36","doi-asserted-by":"crossref","unstructured":"Wu, H., et al.: Fashion IQ: a new dataset towards retrieving images by natural language feedback. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11307\u201311317 (2021)","DOI":"10.1109\/CVPR46437.2021.01115"},{"key":"7_CR37","doi-asserted-by":"crossref","unstructured":"Xie, S., Zhang, Z., Lin, Z., Hinz, T., Zhang, K.: SmartBrush: text and shape guided object inpainting with diffusion model. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 22428\u201322437 (2023)","DOI":"10.1109\/CVPR52729.2023.02148"},{"key":"7_CR38","doi-asserted-by":"crossref","unstructured":"Yang, Z., Zeng, A., Yuan, C., Li, Y.: Effective whole-body pose estimation with two-stages distillation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 4210\u20134220 (2023)","DOI":"10.1109\/ICCVW60793.2023.00455"},{"key":"7_CR39","unstructured":"Ye, H., Zhang, J., Liu, S., Han, X., Yang, W.: IP-adapter: text compatible image prompt adapter for text-to-image diffusion models. arXiv preprint arXiv:2308.06721 (2023)"},{"key":"7_CR40","doi-asserted-by":"crossref","unstructured":"Zhang, L., Rao, A., Agrawala, M.: Adding conditional control to text-to-image diffusion models. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 3836\u20133847 (2023)","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"7_CR41","doi-asserted-by":"crossref","unstructured":"Zou, X., Kong, X., Wong, W., Wang, C., Liu, Y., Cao, Y.: FashionAI: a hierarchical dataset for fashion understanding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshops, pp.\u00a00\u20130 (2019)","DOI":"10.1109\/CVPRW.2019.00039"}],"container-title":["Lecture Notes in Computer Science","Evaluation Science and Engineering"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-95-9694-2_7","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,11]],"date-time":"2026-06-11T14:30:01Z","timestamp":1781188201000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-95-9694-2_7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9789819596935","9789819596942"],"references-count":41,"URL":"https:\/\/doi.org\/10.1007\/978-981-95-9694-2_7","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"1 May 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"Bench","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Symposium on Benchmarking, Measuring and Optimization","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Chengdu","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"3 December 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 December 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"bench2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.benchcouncil.org\/bench2025","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}