{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,2]],"date-time":"2026-06-02T22:01:10Z","timestamp":1780437670089,"version":"3.54.1"},"reference-count":41,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,11]]},"DOI":"10.1016\/j.eswa.2026.132968","type":"journal-article","created":{"date-parts":[[2026,5,23]],"date-time":"2026-05-23T06:50:55Z","timestamp":1779519055000},"page":"132968","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Large language model enhanced image caption for Chinese cultural relics"],"prefix":"10.1016","volume":"329","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6903-8774","authenticated-orcid":false,"given":"Chenggang","family":"Mi","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ying","family":"Dong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shaoliang","family":"Xie","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuan","family":"Ma","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"issue":"1","key":"10.1016\/j.eswa.2026.132968_bib0001","doi-asserted-by":"crossref","first-page":"20","DOI":"10.1186\/s40537-022-00571-w","article-title":"Image captioning model using attention and object features to mimic human image understanding","volume":"9","author":"Al-Malla","year":"2022","journal-title":"Journal of Big Data"},{"key":"10.1016\/j.eswa.2026.132968_bib0002","series-title":"Computer vision\u2013ECCV 2016: 14th European conference, Amsterdam, the Netherlands, October 11\u201314, 2016, proceedings, Part V 14","first-page":"382","article-title":"Spice: Semantic propositional image caption evaluation","author":"Anderson","year":"2016"},{"key":"10.1016\/j.eswa.2026.132968_bib0003","series-title":"Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization","first-page":"65","article-title":"Meteor: An automatic metric for mt evaluation with improved correlation with human judgments","author":"Banerjee","year":"2005"},{"key":"10.1016\/j.eswa.2026.132968_bib0004","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1109\/TETCI.2025.3597289","article-title":"An investigation into value misalignment in LLM-generated texts for cultural heritage","author":"Bu","year":"2025","journal-title":"IEEE Transactions on Emerging Topics in Computational Intelligence"},{"key":"10.1016\/j.eswa.2026.132968_bib0005","series-title":"Proceedings of the 2023 conference on empirical methods in natural language processing (EMNLP)","first-page":"13638","article-title":"CLAIR: Evaluating image captions with large language models","author":"Chan","year":"2023"},{"key":"10.1016\/j.eswa.2026.132968_bib0006","series-title":"Proceedings of the 62nd annual meeting of the association for computational linguistics (volume 2: Short papers)","first-page":"693","article-title":"DUAL-REFLECT: Enhancing large language models for reflective translation through dual learning feedback mechanisms","author":"Chen","year":"2024"},{"issue":"2","key":"10.1016\/j.eswa.2026.132968_bib0007","doi-asserted-by":"crossref","first-page":"111","DOI":"10.3233\/AIC-210172","article-title":"Explaining transformer-based image captioning models: An empirical analysis","volume":"35","author":"Cornia","year":"2022","journal-title":"AI Communications"},{"issue":"2","key":"10.1016\/j.eswa.2026.132968_bib0008","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1007\/s40558-025-00312-8","article-title":"Tell me more: Integrating LLMs in a cultural heritage website for advanced information exploration support","volume":"27","author":"Cossatin","year":"2025","journal-title":"Information Technology & Tourism"},{"key":"10.1016\/j.eswa.2026.132968_bib0009","doi-asserted-by":"crossref","DOI":"10.1016\/j.media.2022.102628","article-title":"CrossMoDA 2021 challenge: Benchmark of cross-modality domain adaptation techniques for vestibular schwannoma and cochlea segmentation","volume":"83","author":"Dorent","year":"2023","journal-title":"Medical Image Analysis"},{"key":"10.1016\/j.eswa.2026.132968_bib0010","first-page":"1","article-title":"The faiss library","author":"Douze","year":"2025","journal-title":"IEEE Transactions on Big Data"},{"key":"10.1016\/j.eswa.2026.132968_bib0011","article-title":"GPT2-Chinese: Tools for training GPT2 model in chinese language","author":"Du","year":"2019","journal-title":"GitHub Repository"},{"issue":"26","key":"10.1016\/j.eswa.2026.132968_bib0012","doi-asserted-by":"crossref","first-page":"19051","DOI":"10.1007\/s00521-023-08744-1","article-title":"Improved arabic image captioning model using feature concatenation with pre-trained word embedding","volume":"35","author":"Elbedwehy","year":"2023","journal-title":"Neural Computing and Applications"},{"issue":"3","key":"10.1016\/j.eswa.2026.132968_bib0013","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3617592","article-title":"Deep learning approaches on image captioning: A review","volume":"56","author":"Ghandi","year":"2023","journal-title":"ACM Computing Surveys"},{"key":"10.1016\/j.eswa.2026.132968_bib0014","series-title":"Findings of the association for computational linguistics ACL 2024","first-page":"14041","article-title":"Unsupervised sign language translation and generation","author":"Guo","year":"2024"},{"key":"10.1016\/j.eswa.2026.132968_bib0015","series-title":"International conference on learning representations","article-title":"LoRA: Low-rank adaptation of large language models","author":"Hu","year":"2022"},{"issue":"1","key":"10.1016\/j.eswa.2026.132968_bib0016","doi-asserted-by":"crossref","first-page":"76","DOI":"10.1038\/s40494-025-01621-1","article-title":"CATS: Cultural-heritage classification using llms and distribute model","volume":"13","author":"Hwang","year":"2025","journal-title":"Npj Heritage Science"},{"key":"10.1016\/j.eswa.2026.132968_bib0017","series-title":"Findings of the association for computational linguistics ACL 2024","first-page":"5472","article-title":"Translatotron-V(ison): An end-to-end model for in-image machine translation","author":"Lan","year":"2024"},{"key":"10.1016\/j.eswa.2026.132968_bib0018","series-title":"Proceedings of the 61st annual meeting of the association for computational linguistics (volume 1: Long papers)","first-page":"3479","article-title":"Exploring better text image translation with multimodal codebook","author":"Lan","year":"2023"},{"issue":"1","key":"10.1016\/j.eswa.2026.132968_bib0019","doi-asserted-by":"crossref","first-page":"159","DOI":"10.2307\/2529310","article-title":"The measurement of observer agreement for categorical data","volume":"33","author":"Landis","year":"1977","journal-title":"Biometrics"},{"key":"10.1016\/j.eswa.2026.132968_bib0020","series-title":"Computer vision \u2013 ECCV 2014","first-page":"740","article-title":"Microsoft COCO: Common objects in context","author":"Lin","year":"2014"},{"key":"10.1016\/j.eswa.2026.132968_bib0021","series-title":"Proceedings of the 60th annual meeting of the association for computational linguistics (volume 1: Long papers)","first-page":"3013","article-title":"Cross-modal discrete representation learning","author":"Liu","year":"2022"},{"key":"10.1016\/j.eswa.2026.132968_bib0022","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"15692","article-title":"COTS: Collaborative two-stream vision-language pre-training model for cross-modal retrieval","author":"Lu","year":"2022"},{"key":"10.1016\/j.eswa.2026.132968_bib0023","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2023.109420","article-title":"Towards local visual modeling for image captioning","volume":"138","author":"Ma","year":"2023","journal-title":"Pattern Recognition"},{"issue":"3","key":"10.1016\/j.eswa.2026.132968_bib0024","doi-asserted-by":"crossref","first-page":"1286","DOI":"10.3390\/s23031286","article-title":"Fashion-oriented image captioning with external knowledge retrieval and fully attentive gates","volume":"23","author":"Moratelli","year":"2023","journal-title":"Sensors"},{"key":"10.1016\/j.eswa.2026.132968_bib0025","unstructured":"Nikiforova, S., Deoskar, T., Paperno, D., & Winter, Y. (2022). Generating image captions with external encyclopedic knowledge. 2210.04806."},{"key":"10.1016\/j.eswa.2026.132968_bib0026","unstructured":"OpenAI et al. (2024). GPT-4 technical report. 2303.08774."},{"key":"10.1016\/j.eswa.2026.132968_bib0027","series-title":"Proceedings of the 40th annual meeting of the association for computational linguistics","first-page":"311","article-title":"BLEU: A method for automatic evaluation of machine translation","author":"Papineni","year":"2002"},{"key":"10.1016\/j.eswa.2026.132968_bib0028","series-title":"Proceedings of the 38th international conference on machine learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume":"vol. 139","author":"Radford","year":"2021"},{"key":"10.1016\/j.eswa.2026.132968_bib0029","series-title":"Proceedings of the 17th conference of the European chapter of the association for computational linguistics","first-page":"3666","article-title":"Retrieval-augmented image captioning","author":"Ramos","year":"2023"},{"key":"10.1016\/j.eswa.2026.132968_bib0030","series-title":"2024\u202fIEEE\/CVF Winter conference on applications of computer vision (WACV)","first-page":"5677","article-title":"FuseCap: Leveraging large language models for enriched fused image captions","author":"Rotstein","year":"2024"},{"key":"10.1016\/j.eswa.2026.132968_bib0031","series-title":"Proceedings of workshop on text summarization of ACL, spain","article-title":"A package for automatic evaluation of summaries","volume":"vol. 5","author":"Rouge","year":"2004"},{"key":"10.1016\/j.eswa.2026.132968_bib0032","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1007\/s00146-025-02238-5","article-title":"Explainable AI, LLM, and digitized archival cultural heritage: A case study of the grand ducal archive of the medici","volume":"40","author":"Toth","year":"2025","journal-title":"AI & SOCIETY"},{"key":"10.1016\/j.eswa.2026.132968_bib0033","series-title":"Proceedings of the 2nd international conference of the ACM greek SIGCHI chapter","first-page":"1","article-title":"Large language models for cultural heritage","author":"Trichopoulos","year":"2023"},{"key":"10.1016\/j.eswa.2026.132968_bib0034","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"4566","article-title":"CIDEr: Consensus-based image description evaluation","author":"Vedantam","year":"2015"},{"key":"10.1016\/j.eswa.2026.132968_bib0035","series-title":"2015\u202fIEEE conference on computer vision and pattern recognition (CVPR)","first-page":"3156","article-title":"Show and tell: A neural image caption generator","author":"Vinyals","year":"2015"},{"key":"10.1016\/j.eswa.2026.132968_bib0036","series-title":"Findings of the association for computational linguistics: EMNLP 2023","first-page":"9129","article-title":"Domain adaptation for conversational query production with the RAG model feedback","author":"Wang","year":"2023"},{"key":"10.1016\/j.eswa.2026.132968_bib0037","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"2608","article-title":"Efficient image captioning for edge devices","volume":"vol. 37","author":"Wang","year":"2023"},{"key":"10.1016\/j.eswa.2026.132968_bib0038","unstructured":"Wang, P. et al. (2024). Qwen2-VL: Enhancing vision-language model\u2019s perception of the world at any resolution. 2409.12191."},{"issue":"6","key":"10.1016\/j.eswa.2026.132968_bib0039","doi-asserted-by":"crossref","first-page":"1367","DOI":"10.1109\/TPAMI.2017.2708709","article-title":"Image captioning and visual question answering based on attributes and external knowledge","volume":"40","author":"Wu","year":"2017","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.132968_bib0040","series-title":"Proceedings of the 32nd ACM international conference on multimedia","first-page":"2719","article-title":"Knowledge-aware artifact image synthesis with llm-enhanced prompting and multi-source supervision","author":"Wu","year":"2024"},{"key":"10.1016\/j.eswa.2026.132968_bib0041","series-title":"Proceedings of the 61st annual meeting of the association for computational linguistics (volume 1: Long papers)","first-page":"13433","article-title":"PEIT: Bridging the modality gap with pre-trained models for end-to-end image translation","author":"Zhu","year":"2023"}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426018804?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426018804?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,2]],"date-time":"2026-06-02T21:06:47Z","timestamp":1780434407000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426018804"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,11]]},"references-count":41,"alternative-id":["S0957417426018804"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132968","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,11]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Large language model enhanced image caption for Chinese cultural relics","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132968","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"132968"}}