{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T13:12:20Z","timestamp":1783775540561,"version":"3.55.0"},"publisher-location":"Singapore","reference-count":35,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819228546","type":"print"},{"value":"9789819228522","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,7,12]],"date-time":"2026-07-12T00:00:00Z","timestamp":1783814400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,12]],"date-time":"2026-07-12T00:00:00Z","timestamp":1783814400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-2852-2_38","type":"book-chapter","created":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T12:31:01Z","timestamp":1783773061000},"page":"509-521","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["TraE-VL: A Cascaded Framework for\u00a0Fine-Grained Perception and\u00a0Controllable Stylization in\u00a0Domain-Specific Copywriting Generation"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-9147-4353","authenticated-orcid":false,"given":"Xiang","family":"Fu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7205-7212","authenticated-orcid":false,"given":"Caiyun","family":"Fan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-8825-9327","authenticated-orcid":false,"given":"Mengmeng","family":"Yu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,12]]},"reference":[{"key":"38_CR1","doi-asserted-by":"crossref","unstructured":"Alayrac, J.B., et al.: Flamingo: a visual language model for few-shot learning. In: Proceedings of the NeurIPS, pp. 23716\u201323736 (2022)","DOI":"10.52202\/068431-1723"},{"key":"38_CR2","unstructured":"Bai, J., et al.: Qwen-VL: A versatile vision-language model for understanding, localization, text reading, and beyond (2023)"},{"key":"38_CR3","doi-asserted-by":"publisher","DOI":"10.1007\/s11432-024-4231-5","volume":"67","author":"Z Chen","year":"2024","unstructured":"Chen, Z., et al.: How far are we to GPT-4V? closing the gap to commercial multimodal models with open-source suites. Sci. China Inf. Sci. 67, 220101 (2024)","journal-title":"Sci. China Inf. Sci."},{"key":"38_CR4","doi-asserted-by":"crossref","unstructured":"Cheng, T., et al.: YOLO-World: Real-time open-vocabulary object detection. In: Proceedings of the CVPR, pp. 16901\u201316911 (2024)","DOI":"10.1109\/CVPR52733.2024.01599"},{"key":"38_CR5","doi-asserted-by":"crossref","unstructured":"Cho, S., Oh, H.: Generalized image captioning for multilingual support. Appl. Sci. 13 (2023)","DOI":"10.3390\/app13042446"},{"key":"38_CR6","doi-asserted-by":"crossref","unstructured":"Dai, W., et al.: InstructBLIP: towards general-purpose vision-language models with instruction tuning. In: Proceedings of the NeurIPS, pp. 49250\u201349267 (2023)","DOI":"10.52202\/075280-2142"},{"key":"38_CR7","unstructured":"Dosovitskiy, A., et al.: An image is worth 16x16 words: transformers for image recognition at scale. In: Proceedings of the ICLR (2021)"},{"key":"38_CR8","unstructured":"Houlsby, N., et al.: Parameter-efficient transfer learning for NLP. In: Proceedings of the ICML, pp. 2790\u20132799 (2019)"},{"key":"38_CR9","unstructured":"Hu, E.J., et al.: LoRA: low-rank adaptation of large language models (2021)"},{"key":"38_CR10","unstructured":"Lewis, P., et al.: Retrieval-augmented generation for knowledge-intensive NLP tasks. In: Proceedings of the NeurIPS, pp. 9459\u20139474 (2020)"},{"key":"38_CR11","unstructured":"Li, B., et\u00a0al.: LLaVA-NeXT: stronger LLMs supercharge multimodal capabilities in the wild (2024). https:\/\/llava-vl.github.io\/blog\/2024-05-10-llava-next-stronger-llms\/"},{"key":"38_CR12","unstructured":"Li, J., et al.: BLIP: Bootstrapping language-image pre-training for unified visionlanguage understanding and generation. In: Proceedings of the ICML, pp. 12888\u201312900 (2022)"},{"key":"38_CR13","unstructured":"Li, J., et al.: BLIP-2: Bootstrapping language-image pre-training with frozen image encoders and large language models. In: Proceedings of the ICML, pp. 19730\u201319742 (2023)"},{"key":"38_CR14","doi-asserted-by":"crossref","unstructured":"Li, L.H., et al.: Grounded language-image pre-training. In: Proceedings of the CVPR, pp. 10965\u201310975 (2022)","DOI":"10.1109\/CVPR52688.2022.01069"},{"key":"38_CR15","doi-asserted-by":"crossref","unstructured":"Li, X.L., Liang, P.: Prefix-tuning: Optimizing continuous prompts for generation. In: Proceedings of the ACL, pp. 4582\u20134597 (2021)","DOI":"10.18653\/v1\/2021.acl-long.353"},{"key":"38_CR16","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"121","DOI":"10.1007\/978-3-030-58577-8_8","volume-title":"Computer Vision \u2013 ECCV 2020","author":"X Li","year":"2020","unstructured":"Li, X., et al.: Oscar: object-semantics aligned pre-training for vision-language tasks. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12375, pp. 121\u2013137. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58577-8_8"},{"key":"38_CR17","doi-asserted-by":"crossref","unstructured":"Li, Y., et al.: Evaluating object hallucination in large vision-language models. In: Proceedings of the EMNLP, pp. 292\u2013305 (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.20"},{"key":"38_CR18","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"740","DOI":"10.1007\/978-3-319-10602-1_48","volume-title":"Computer Vision \u2013 ECCV 2014","author":"T-Y Lin","year":"2014","unstructured":"Lin, T.-Y., et al.: Microsoft COCO: common objects in context. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8693, pp. 740\u2013755. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10602-1_48"},{"key":"38_CR19","doi-asserted-by":"crossref","unstructured":"Liu, H., et al.: Visual instruction tuning. In: Proceedings of the NeurIPS, pp. 34892\u201334916 (2023)","DOI":"10.52202\/075280-1516"},{"key":"38_CR20","doi-asserted-by":"crossref","unstructured":"Liu, S., et al.: Grounding DINO: marrying DINO with grounded pre-training. In: Proceedings of the CVPR, pp. 16786\u201316795 (2024)","DOI":"10.1007\/978-3-031-72970-6_3"},{"key":"38_CR21","doi-asserted-by":"crossref","unstructured":"Mathews, A., et al.: SentiCap: generating image descriptions with sentiments (2015)","DOI":"10.1609\/aaai.v30i1.10475"},{"key":"38_CR22","unstructured":"OpenAI: GPT-4o system card (2024)"},{"key":"38_CR23","doi-asserted-by":"crossref","unstructured":"Papineni, K., et al.: BLEU: a method for automatic evaluation of machine translation. In: Proceedings of the ACL, pp. 311\u2013318 (2002)","DOI":"10.3115\/1073083.1073135"},{"key":"38_CR24","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: Proceedings of the ICML, pp. 8748\u20138763 (2021)"},{"key":"38_CR25","doi-asserted-by":"crossref","unstructured":"Rohrbach, A., et al.: Object hallucination in image captioning. In: Proceedings of the EMNLP (2018)","DOI":"10.18653\/v1\/D18-1437"},{"key":"38_CR26","doi-asserted-by":"crossref","unstructured":"Shao, Z., et al.: Long and diverse text generation with planning-based hierarchical variational model (2019). arXiv:1908.06605","DOI":"10.18653\/v1\/D19-1321"},{"key":"38_CR27","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Proceedings of the NeurIPS (2017)"},{"key":"38_CR28","doi-asserted-by":"crossref","unstructured":"Vedantam, R., et al.: CIDEr: consensus-based image description evaluation. In: Proceedings of the CVPR, pp. 4566\u20134575 (2015)","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"38_CR29","doi-asserted-by":"crossref","unstructured":"Vinyals, O., et al.: Show and tell: a neural image caption generator. In: Proceedings of the CVPR, pp. 3156\u20133164 (2015)","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"38_CR30","unstructured":"Wang, N., et al.: Controllable image captioning via prompting. In: Proceedings of the AAAI (2022)"},{"key":"38_CR31","unstructured":"Wang, P., et al.: Qwen2-VL: enhancing vision-language model\u2019s perception of the world at any resolution. arXiv:2409.12191 (2024)"},{"key":"38_CR32","unstructured":"Xie, C., et al.: ZERO: a large-scale Chinese cross-modal benchmark with a new vision-language framework. In: Proceedings of the NeurIPS, pp. 22263\u201322276 (2022)"},{"key":"38_CR33","doi-asserted-by":"publisher","DOI":"10.1007\/s11432-024-4251-x","volume":"67","author":"S Yin","year":"2024","unstructured":"Yin, S., et al.: Woodpecker: hallucination correction for multimodal large language models. Sci. China Inf. Sci. 67, 220105 (2024)","journal-title":"Sci. China Inf. Sci."},{"key":"38_CR34","doi-asserted-by":"crossref","unstructured":"Yu, T., et al.: RLHF-V: towards trustworthy MLLMs via behavior alignment from fine-grained correctional human feedback. In: Proceedings of the CVPR, pp. 13807\u201313816 (2024)","DOI":"10.1109\/CVPR52733.2024.01310"},{"key":"38_CR35","doi-asserted-by":"crossref","unstructured":"Zheng, L., et al.: Judging LLM-as-a-judge with MT-Bench and chatbot arena. In: Proceedings of the NeurIPS, pp. 46595\u201346623 (2023)","DOI":"10.52202\/075280-2020"}],"container-title":["Lecture Notes in Computer Science","Knowledge Science, Engineering and Management"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-2852-2_38","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T12:31:06Z","timestamp":1783773066000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-2852-2_38"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,12]]},"ISBN":["9789819228546","9789819228522"],"references-count":35,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-2852-2_38","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,12]]},"assertion":[{"value":"12 July 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"KSEM","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Knowledge Science, Engineering and Management","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Beijing","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"19 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"19","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ksem2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ksem2026.rosc.org.cn\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}