{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,30]],"date-time":"2026-05-30T02:11:42Z","timestamp":1780107102286,"version":"3.54.0"},"reference-count":75,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2025,3,18]],"date-time":"2025-03-18T00:00:00Z","timestamp":1742256000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,3,18]],"date-time":"2025-03-18T00:00:00Z","timestamp":1742256000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2025,6]]},"DOI":"10.1007\/s00530-025-01737-1","type":"journal-article","created":{"date-parts":[[2025,3,18]],"date-time":"2025-03-18T01:25:03Z","timestamp":1742261103000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["ACIH-VQT: aesthetic constraints incorporated hierarchical VQ-transformer for text logo synthesis"],"prefix":"10.1007","volume":"31","author":[{"given":"Zhixiong","family":"Liu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Fang","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mohan","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shenglan","family":"Cui","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,3,18]]},"reference":[{"key":"1737_CR1","unstructured":"Hardy, G.: Smashing logo design: The art of creating visual identities 24 (2011)"},{"key":"1737_CR2","unstructured":"Airey, D.: Identity designed: The definitive guide to visual branding (2019)"},{"issue":"3","key":"1737_CR3","first-page":"383","volume":"47","author":"TR Williams","year":"2000","unstructured":"Williams, T.R.: Guidelines for designing and evaluating the display of information on the web. Techn. Commun. 47(3), 383\u2013396 (2000)","journal-title":"Techn. Commun."},{"key":"1737_CR4","unstructured":"Seymour, V.: Principles of composition in art and design (2023). https:\/\/daily.jstor.org\/principles-of-composition-in-art-and-design\/"},{"key":"1737_CR5","doi-asserted-by":"publisher","DOI":"10.1145\/3058982","author":"J Zhang","year":"2017","unstructured":"Zhang, J., Yu, J., Zhang, K., Zheng, X.S., Zhang, J.: Computational aesthetic evaluation of logos. ACM Trans. Appl. Percept. (2017). https:\/\/doi.org\/10.1145\/3058982","journal-title":"ACM Trans. Appl. Percept."},{"issue":"2\u20133","key":"1737_CR6","doi-asserted-by":"publisher","first-page":"237","DOI":"10.1177\/0263276406062680","volume":"23","author":"R Shusterman","year":"2006","unstructured":"Shusterman, R.: The aesthetic. Theory Culture Soc 23(2\u20133), 237\u2013243 (2006)","journal-title":"Theory Culture Soc"},{"issue":"9","key":"1737_CR7","doi-asserted-by":"publisher","first-page":"9364","DOI":"10.1109\/TKDE.2023.3237969","volume":"35","author":"T Zhou","year":"2023","unstructured":"Zhou, T., Cai, Z., Liu, F., Su, J.: In pursuit of beauty: aesthetic-aware and context-adaptive photo selection in crowdsensing. IEEE Trans. Knowl. Data Eng. 35(9), 9364\u20139377 (2023)","journal-title":"IEEE Transactions on Knowledge and Data Engineering"},{"key":"1737_CR8","doi-asserted-by":"crossref","unstructured":"Hsu, H.Y., He, X., Peng, Y., Kong, H., Zhang, Q.: Posterlayout: a new benchmark and approach for content-aware visual-textual presentation layout. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 6018\u20136026 (2023)","DOI":"10.1109\/CVPR52729.2023.00583"},{"key":"1737_CR9","doi-asserted-by":"publisher","unstructured":"Zhou, M., Xu, C., Ma, Y., Ge, T., Jiang, Y., Xu, W.: Composition-aware graphic layout gan for visual-textual presentation designs. In: Proceedings of the Thirty-First International Joint Conference on Artificial Intelligence, IJCAI-22, pp. 4995\u20135001 (2022). https:\/\/doi.org\/10.24963\/ijcai.2022\/692","DOI":"10.24963\/ijcai.2022\/692"},{"key":"1737_CR10","doi-asserted-by":"crossref","unstructured":"Horita, D., Inoue, N., Kikuchi, K., Yamaguchi, K., Aizawa, K.: Retrieval-augmented layout transformer for content-aware layout generation (2023)","DOI":"10.1109\/CVPR52733.2024.00015"},{"key":"1737_CR11","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2024.123136","volume":"245","author":"M Zhang","year":"2024","unstructured":"Zhang, M., Liu, F., Li, B., Liu, Z., Ma, W., Ran, C.: Creposter: Leveraging multi level features for cultural relic poster generation via attention-based framework. Expert Syst. Appl. 245, 123136 (2024)","journal-title":"Expert Systems with Applications"},{"key":"1737_CR12","doi-asserted-by":"publisher","first-page":"361","DOI":"10.1007\/978-3-031-41676-7_21","volume-title":"Document analysis and recognition - ICDAR 2023","author":"L He","year":"2023","unstructured":"He, L., Lu, Y., Corring, J., Florencio, D., Zhang, C.: Diffusion-based document layout generation. In: Fink, G.A., Jain, R., Kise, K., Zanibbi, R. (eds.) Document analysis and recognition - ICDAR 2023, pp. 361\u2013378. Springer, Cham (2023)"},{"key":"1737_CR13","doi-asserted-by":"crossref","unstructured":"Yamaguchi, K.: Canvasvae: learning to generate vector graphic documents. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 5481\u20135489 (2021)","DOI":"10.1109\/ICCV48922.2021.00543"},{"key":"1737_CR14","doi-asserted-by":"crossref","unstructured":"Patil, A.G., Ben-Eliezer, O., Perel, O., Averbuch-Elor, H.: Read: recursive autoencoders for document layout generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshops, pp. 544\u2013545 (2020)","DOI":"10.1109\/CVPRW50498.2020.00280"},{"key":"1737_CR15","doi-asserted-by":"crossref","unstructured":"Jing, Q., Zhou, T., Tsang, Y., Chen, L., Sun, L., Zhen, Y., Du, Y.: Layout generation for various scenarios in mobile shopping applications. In: Proceedings of the 2023 CHI Conference on Human Factors in Computing Systems, pp. 1\u201318 (2023)","DOI":"10.1145\/3544548.3581446"},{"key":"1737_CR16","doi-asserted-by":"crossref","unstructured":"Rahman, S., Sermuga\u00a0Pandian, V.P., Jarke, M.: Ruite: Refining ui layout aesthetics using transformer encoder. In: companion Proceedings of the 26th International Conference on Intelligent User Interfaces, pp. 81\u201383 (2021)","DOI":"10.1145\/3397482.3450716"},{"key":"1737_CR17","unstructured":"Zhan, H., Huang, Y., Liu, D., et al.: Pairwise gui dataset construction between android phones and tablets. Adv. Neur. Inform. Process. Syst. 36 (2024)"},{"key":"1737_CR18","doi-asserted-by":"publisher","unstructured":"Shi, Y., Shang, M., Qi, Z.: Intelligent layout generation based on deep generative models: a comprehensive survey. Inf. Fusion. 100(C) (2023). https:\/\/doi.org\/10.1016\/j.inffus.2023.101940","DOI":"10.1016\/j.inffus.2023.101940"},{"key":"1737_CR19","unstructured":"Li, J., Yang, J., et al.: Layoutgan: Generating graphic layouts with wireframe discriminators. Proceedings of the International Conference on Learning Representations (2019)"},{"key":"1737_CR20","doi-asserted-by":"crossref","unstructured":"Yang, C.-F., Fan, W.-C., Yang, F.-E., Wang, Y.-C.F.: Layouttransformer: Scene layout generation with conceptual and spatial diversity. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3732\u20133741 (2021)","DOI":"10.1109\/CVPR46437.2021.00373"},{"key":"1737_CR21","unstructured":"Chen, J., Huang, Y., Lv, T., Cui, L., Chen, Q., Wei, F.: Textdiffuser: Diffusion models as text painters. Adv. Neur. Inform. Process. Syst. 36 (2024)"},{"key":"1737_CR22","doi-asserted-by":"crossref","unstructured":"Thamizharasan, V., Liu, D., Agarwal, S., Fisher, M., Gharbi, M., Wang, O., Jacobson, A., Kalogerakis, E.: Vecfusion: Vector font generation with diffusion. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7943\u20137952 (2024)","DOI":"10.1109\/CVPR52733.2024.00759"},{"key":"1737_CR23","doi-asserted-by":"crossref","unstructured":"Levi, E., Brosh, E., Mykhailych, M., Perez, M.: Dlt: Conditioned layout generation with joint discrete-continuous diffusion layout transformer. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2106\u20132115 (2023)","DOI":"10.1109\/ICCV51070.2023.00201"},{"key":"1737_CR24","unstructured":"Chen, J., Zhang, R., Zhou, Y., Chen, C.: Towards aligned layout generation via diffusion model with aesthetic constraints. In: The Twelfth International Conference on Learning Representations"},{"key":"1737_CR25","doi-asserted-by":"crossref","unstructured":"Zheng, G., Zhou, X., Li, X., Qi, Z., Shan, Y., Li, X.: Layoutdiffusion: Controllable diffusion model for layout-to-image generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 22490\u201322499 (2023)","DOI":"10.1109\/CVPR52729.2023.02154"},{"key":"1737_CR26","unstructured":"Kingma, D.P., Welling, M.: Auto-encoding variational bayes. arXiv preprint arXiv:1312.6114 (2013)"},{"key":"1737_CR27","doi-asserted-by":"crossref","unstructured":"Jyothi, A.A., Durand, T., He, J., Sigal, L.: Layoutvae: Stochastic scene layout generation from a label set. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV) (2019). IEEE","DOI":"10.1109\/ICCV.2019.00999"},{"key":"1737_CR28","doi-asserted-by":"crossref","unstructured":"Kong, X., Jiang, L., Chang, H., Zhang, H., Hao, Y., Gong, H., Essa, I.: Blt: Bidirectional layout transformer for controllable layout generation (2021). arXiv preprint arXiv:2112.05112","DOI":"10.1007\/978-3-031-19790-1_29"},{"key":"1737_CR29","doi-asserted-by":"crossref","unstructured":"Cao, Y., Ma, Y., Zhou, M., Liu, C., Xie, H., Ge, T., Jiang, Y.: Geometry aligned variational transformer for image-conditioned layout generation. In: Proceedings of the 30th ACM International Conference on Multimedia, pp. 1561\u20131571 (2022)","DOI":"10.1145\/3503161.3548332"},{"key":"1737_CR30","unstructured":"Van Den\u00a0Oord, A., Vinyals, O., et al.: Neural discrete representation learning. Advances in neural information processing systems 30 (2017)"},{"key":"1737_CR31","doi-asserted-by":"crossref","unstructured":"Zhang, L., Chen, X., Wang, Y., Lu, Y., Qiao, Y.: Brush your text: Synthesize any scene text on images via diffusion model. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 38, pp. 7215\u20137223 (2024)","DOI":"10.1609\/aaai.v38i7.28550"},{"key":"1737_CR32","unstructured":"Tuo, Y., Xiang, W., He, J.-Y., Geng, Y., Xie, X.: Anytext: Multilingual visual text generation and editing. In: The Twelfth International Conference on Learning Representations"},{"key":"1737_CR33","unstructured":"Goodfellow, I., Pouget-Abadie, J., Mirza, M., Xu, B., Warde-Farley, D., Ozair, S., Courville, A., Bengio, Y.: Generative adversarial nets. Advances in neural information processing systems 27 (2014)"},{"key":"1737_CR34","doi-asserted-by":"crossref","unstructured":"Kikuchi, K., Simo-Serra, E., Otani, M., Yamaguchi, K.: Constrained graphic layout generation via latent optimization. In: Proceedings of the 29th ACM International Conference on Multimedia, pp. 88\u201396 (2021)","DOI":"10.1145\/3474085.3475497"},{"issue":"10","key":"1737_CR35","doi-asserted-by":"publisher","first-page":"4039","DOI":"10.1109\/TVCG.2020.2999335","volume":"27","author":"J Li","year":"2021","unstructured":"Li, J., Yang, J., Zhang, J., Liu, C., Wang, C., Xu, T.: Attribute-conditioned layout gan for automatic graphic design. IEEE Trans. Visualiz. Comput. Graph. 27(10), 4039\u20134048 (2021). https:\/\/doi.org\/10.1109\/TVCG.2020.2999335","journal-title":"IEEE Trans. Visualiz. Comput. Graph."},{"issue":"1","key":"1737_CR36","doi-asserted-by":"publisher","first-page":"1096","DOI":"10.1609\/aaai.v36i1.19994","volume":"36","author":"Z Jiang","year":"2022","unstructured":"Jiang, Z., Sun, S., Zhu, J., Lou, J.-G., Zhang, D.: Coarse-to-fine generative modeling for graphic layouts. Proceedings of the AAAI Conference on Artificial Intelligence 36(1), 1096\u20131103 (2022). https:\/\/doi.org\/10.1609\/aaai.v36i1.19994","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"1737_CR37","doi-asserted-by":"publisher","unstructured":"Jiang, Z., Guo, J., Sun, S., Deng, H., Wu, Z., Mijovic, V., Yang, Z., Lou, J., Zhang, D.: Layoutformer++: Conditional graphic layout generation via constraint serialization and decoding space restriction. In: 2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 18403\u201318412. IEEE Computer Society, Los Alamitos, CA, USA (2023). https:\/\/doi.org\/10.1109\/CVPR52729.2023.01765","DOI":"10.1109\/CVPR52729.2023.01765"},{"key":"1737_CR38","doi-asserted-by":"crossref","unstructured":"Inoue, N., Kikuchi, K., Simo-Serra, E., Otani, M., Yamaguchi, K.: LayoutDM: Discrete Diffusion Model for Controllable Layout Generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 10167\u201310176 (2023)","DOI":"10.1109\/CVPR52729.2023.00980"},{"key":"1737_CR39","doi-asserted-by":"crossref","unstructured":"Lee, D., Kim, C., Kim, S., Cho, M., Han, W.-S.: Autoregressive image generation using residual quantization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11523\u201311532 (2022)","DOI":"10.1109\/CVPR52688.2022.01123"},{"key":"1737_CR40","unstructured":"Yu, J., Li, X., Koh, J.Y., Zhang, H., Pang, R., Qin, J., Ku, A., Xu, Y., Baldridge, J., Wu, Y.: Vector-quantized image modeling with improved vqgan. arXiv preprint arXiv:2110.04627 (2021)"},{"key":"1737_CR41","unstructured":"Razavi, A., Oord, A., Vinyals, O.: Generating diverse high-fidelity images with vq-vae-2. In: Advances in Neural Information Processing Systems (2019)"},{"key":"1737_CR42","doi-asserted-by":"crossref","unstructured":"Peng, J., Liu, D., Zhu, S., Zhang, R., Metaxas, D.: Generating diverse structure for image inpainting with hierarchical vq-vae. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10753\u201310762 (2021)","DOI":"10.1109\/CVPR46437.2021.01063"},{"key":"1737_CR43","unstructured":"Yan, W., Zhang, Y., Tan, P.A., Pathak, D.: Videogpt: Video generation using vq-vae and transformers. arXiv preprint arXiv:2104.10157 (2021)"},{"key":"1737_CR44","doi-asserted-by":"crossref","unstructured":"Karth, I., Aytemiz, B., Green, M.C., Magerko, B.: Neurosymbolic map generation with vq-vae and wfc. In: Proceedings of the 16th Conference on Artificial Intelligence and Interactive Digital Entertainment (2021)","DOI":"10.1145\/3472538.3472584"},{"key":"1737_CR45","doi-asserted-by":"crossref","unstructured":"Wen, L., Zhu, Y., Ye, L., Chen, G., Yu, B., Liu, J., Xu, C.: Layoutransformer: Generating layout patterns with transformer via sequential pattern modeling. In: Proceedings of the 41st IEEE\/ACM International Conference on Computer-aided Design, pp. 1\u20139 (2022)","DOI":"10.1145\/3508352.3549350"},{"key":"1737_CR46","doi-asserted-by":"crossref","unstructured":"Arroyo, D.M., Postels, J., Tombari, F.: Variational transformer networks for layout generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13642\u201313652 (2021)","DOI":"10.1109\/CVPR46437.2021.01343"},{"key":"1737_CR47","doi-asserted-by":"crossref","unstructured":"Gupta, K., Lazarow, J., Achille, A., Davis, L.S., Mahadevan, V., Shrivastava, A.: Layouttransformer: Layout generation and completion with self-attention. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1004\u20131014 (2021)","DOI":"10.1109\/ICCV48922.2021.00104"},{"issue":"1","key":"1737_CR48","doi-asserted-by":"publisher","first-page":"9","DOI":"10.1007\/s44267-023-00010-1","volume":"1","author":"D Shi","year":"2023","unstructured":"Shi, D., Cui, W., Huang, D., Zhang, H., Cao, N.: Reverse-engineering information presentations: recovering hierarchical grouping from layouts of visual elements. Vis. Intellig. 1(1), 9 (2023)","journal-title":"Vis. Intellig."},{"key":"1737_CR49","doi-asserted-by":"crossref","unstructured":"Chai, S., Zhuang, L., Yan, F.: Layoutdm: Transformer-based diffusion model for layout generation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18349\u201318358 (2023)","DOI":"10.1109\/CVPR52729.2023.01760"},{"key":"1737_CR50","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, \u0141., Polosukhin, I.: Attention is all you need. Advances in neural information processing systems 30 (2017)"},{"key":"1737_CR51","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2024.125974","volume":"267","author":"S Cui","year":"2025","unstructured":"Cui, S., Liu, Z., Liu, F., Ye, Y., Zhang, M.: Integrating conceptual and visual representations with domain expertise for scalable visual plagiarism detection. Expert Syst. Appl. 267, 125974 (2025)","journal-title":"Expert Systems with Applications"},{"key":"1737_CR52","unstructured":"Devlin, J., Chang, M.-W., Lee, K., Toutanova, K.: Bert: Pre-training of deep bidirectional transformers for language understanding (2018). arXiv preprint arXiv:1810.04805"},{"issue":"7","key":"1737_CR53","doi-asserted-by":"publisher","first-page":"3727","DOI":"10.1109\/TNNLS.2021.3114378","volume":"34","author":"S Zhao","year":"2021","unstructured":"Zhao, S., Hu, M., Cai, Z., Zhang, Z., Zhou, T., Liu, F.: Enhancing chinese char acter representation with lattice-aligned attention. IEEE Trans. Neural Netw. Learn. Syst. 34(7), 3727\u20133736 (2021)","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"1737_CR54","first-page":"1877","volume":"33","author":"T Brown","year":"2020","unstructured":"Brown, T., Mann, B., Ryder, N., Subbiah, M., Kaplan, J.D., Dhariwal, P., Neelakantan, A., Shyam, P., Sastry, G., Askell, A., et al.: Language models are few-shot learners. Adv. Neur. Inform. Process. Syst. 33, 1877\u20131901 (2020)","journal-title":"Adv. Neur. Inform. Process. Syst."},{"key":"1737_CR55","unstructured":"Achiam, J., Adler, S., Agarwal, S., Ahmad, L., Akkaya, I., Aleman, F.L., Almeida, D., Altenschmidt, J., Altman, S., Anadkat, S., et al.: Gpt-4 technical report. arXiv preprint arXiv:2303.08774 (2023)"},{"issue":"8","key":"1737_CR56","first-page":"9","volume":"1","author":"A Radford","year":"2019","unstructured":"Radford, A., Wu, J., Child, R., Luan, D., Amodei, D., Sutskever, I., et al.: Language models are unsupervised multitask learners. OpenAI blog 1(8), 9 (2019)","journal-title":"OpenAI blog"},{"key":"1737_CR57","doi-asserted-by":"crossref","unstructured":"Lewis, M., Liu, Y., Goyal, N., Ghazvininejad, M., Mohamed, A., Levy, O., Stoyanov, V., Zettlemoyer, L.: Bart:denoising sequence-to-sequence pre-training for natural language generation, translation, and comprehension. arXiv preprint arXiv:1910.13461 (2019)","DOI":"10.18653\/v1\/2020.acl-main.703"},{"key":"1737_CR58","doi-asserted-by":"crossref","unstructured":"Liu, Y., Wan, Y., He, L., Peng, H., Philip, S.Y.: Kg-bart: Knowledge graph-augmented bart for generative commonsense reasoning. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 35, pp. 6418\u20136425 (2021)","DOI":"10.1609\/aaai.v35i7.16796"},{"key":"1737_CR59","doi-asserted-by":"publisher","unstructured":"Zhu, W., Yan, A., Lu, Y., Xu, W., Wang, X., Eckstein, M., Wang, W.Y.: Visualize before you write: Imagination-guided open-ended text generation. In: Vlachos, A., Augenstein, I. (eds.) Findings of the Association for Computational Linguistics: EACL 2023, pp. 78\u201392. Association for Computational Linguistics, Dubrovnik, Croatia (2023). https:\/\/doi.org\/10.18653\/v1\/2023.findings-eacl.5 . https:\/\/aclanthology.org\/2023.findings-eacl.5","DOI":"10.18653\/v1\/2023.findings-eacl.5"},{"issue":"2","key":"1737_CR60","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3606368","volume":"42","author":"X Chen","year":"2023","unstructured":"Chen, X., Song, X., Jing, L., Li, S., Hu, L., Nie, L.: Multimodal dialog systems with dual knowledge-enhanced generative pretrained language model. ACM Trans. Inform. Syst. 42(2), 1\u201325 (2023)","journal-title":"ACM Trans. Inform. Syst."},{"issue":"3","key":"1737_CR61","doi-asserted-by":"publisher","first-page":"1122","DOI":"10.1109\/TNNLS.2021.3104971","volume":"34","author":"S Zhao","year":"2021","unstructured":"Zhao, S., Hu, M., Cai, Z., Liu, F.: Dynamic modeling cross-modal interactions in two-phase prediction for entity-relation extraction. IEEE Trans. Neural Netwo. Learn. Syst. 34(3), 1122\u20131131 (2021)","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"1737_CR62","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2023.101811","volume":"97","author":"F Liu","year":"2023","unstructured":"Liu, F., Zhang, M., Zheng, B., Cui, S., Ma, W., Liu, Z.: Feature fusion via multi target learning for ancient artwork captioning. Inf. Fusion. 97, 101811 (2023)","journal-title":"Information Fusion"},{"key":"1737_CR63","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recogni tion. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"1737_CR64","doi-asserted-by":"crossref","unstructured":"Wang, Y., Pu, G., Luo, W., Wang, P. Yexin ans\u00a0Xiong, Kang, H., Wang, Z., Lian, Z.: Aesthetic text logo synthesis via content-aware layout inferring. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (2022)","DOI":"10.1109\/CVPR52688.2022.00247"},{"key":"1737_CR65","unstructured":"Mittman, A.S.: Look At This!: An Introduction to Art Appreciation. https:\/\/pressbooks.calstate.edu\/lookatthis\/ (2023)"},{"key":"1737_CR66","doi-asserted-by":"crossref","unstructured":"Xu, X., Zhang, Z., Wang, Z., Price, B., Wang, Z., Shi, H.: Rethinking text segmentation: A novel dataset and a text-specific refinement approach. arXiv preprint arXiv:2011.14021 (2020)","DOI":"10.1109\/CVPR46437.2021.01187"},{"key":"1737_CR67","doi-asserted-by":"publisher","unstructured":"Pennington, J., Socher, R., Manning, C.: GloVe: Global vectors for word representation. In: Moschitti, A., Pang, B., Daelemans, W. (eds.) Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing (EMNLP), pp. 1532\u20131543. Association for Computational Linguistics, Doha, Qatar (2014). https:\/\/doi.org\/10.3115\/v1\/D14-1162 . https:\/\/aclanthology.org\/D14-1162\/","DOI":"10.3115\/v1\/D14-1162"},{"key":"1737_CR68","unstructured":"Heusel, M., Ramsauer, H., Unterthiner, T., Nessler, B., Hochreiter, S.: Gans trained by a two time-scale update rule converge to a local nash equilibrium. Adv. Neur. Inform. Process. Syst. 30 (2017)"},{"key":"1737_CR69","unstructured":"Salimans, T., Goodfellow, I., Zaremba, W., Cheung, V., Radford, A., Chen, X.: Improved techniques for training gans. In: Advances in Neural Information Processing Systems, pp. 2234\u20132242 (2016)"},{"key":"1737_CR70","unstructured":"Paszke, A., Gross, S., Massa, F., Lerer, A., Bradbury, J., Chanan, G., Killeen, T., Lin, Z., Gimelshein, N., Antiga, L., Desmaison, A., K\u00f6pf, A., Yang, E.Z., DeVito, Z., Raison, M., Tejani, A., Chilamkurthy, S., Steiner, B., Fang, L., Bai, J., Chintala, S.: PyTorch: An Imperative Style, High-Performance Deep Learning Library (2019)"},{"key":"1737_CR71","unstructured":"Kingma, D.P., Ba, J.: Adam: A method for stochastic optimization (2014). arXiv preprint arXiv:1412.6980"},{"key":"1737_CR72","doi-asserted-by":"crossref","unstructured":"Gu, S., Chen, D., Bao, J., Wen, F., Zhang, B., Chen, D., Yuan, L., Guo, B.: Vector quantized diffusion model for text-to-image synthesis (2021). arXiv preprint arXiv:2111.14822","DOI":"10.1109\/CVPR52688.2022.01043"},{"key":"1737_CR73","doi-asserted-by":"crossref","unstructured":"Liu, F., Lv, J., Cui, S., Luan, Z., Wu, K., Zhou, T.: Smart\u201d error\u201d! exploring imperfect ai to support creative ideation. Proceedings of the ACM on Human Computer Interaction 8(CSCW1), pp. 1\u201328 (2024)","DOI":"10.1145\/3637398"},{"key":"1737_CR74","doi-asserted-by":"publisher","unstructured":"Xiao, S., Wang, L., Ma, X., Zeng, W.: Typedance: Creating semantic typographic logos from image through personalized generation. In: Proceedings of the 2024 CHI Conference on Human Factors in Computing Systems. CHI \u201924. Association for Computing Machinery, New York, NY, USA (2024). https:\/\/doi.org\/10.1145\/3613904.3642185","DOI":"10.1145\/3613904.3642185"},{"key":"1737_CR75","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2024.102304","volume":"106","author":"L Xiao","year":"2024","unstructured":"Xiao, L., Wu, X., Xu, J., Li, W., Jin, C., He, L.: Atlantis: aesthetic-oriented multiple granularities fusion network for joint multimodal aspect-based sentiment analysis. Inform. Fus. 106, 102304 (2024)","journal-title":"Inform. Fus."}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-025-01737-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-025-01737-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-025-01737-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,4]],"date-time":"2025-09-04T15:01:18Z","timestamp":1756998078000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-025-01737-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,3,18]]},"references-count":75,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2025,6]]}},"alternative-id":["1737"],"URL":"https:\/\/doi.org\/10.1007\/s00530-025-01737-1","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,3,18]]},"assertion":[{"value":"5 September 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 February 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 March 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}],"article-number":"165"}}