{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,1]],"date-time":"2025-03-01T06:12:29Z","timestamp":1740809549917,"version":"3.38.0"},"reference-count":50,"publisher":"Institute of Electronics, Information and Communications Engineers (IEICE)","issue":"3","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEICE Trans. Inf. &amp; Syst."],"published-print":{"date-parts":[[2025,3,1]]},"DOI":"10.1587\/transinf.2024edp7036","type":"journal-article","created":{"date-parts":[[2024,10,1]],"date-time":"2024-10-01T22:15:15Z","timestamp":1727820915000},"page":"266-276","source":"Crossref","is-referenced-by-count":0,"title":["UTStyleCap4K: Generating Image Captions with Sentimental Styles"],"prefix":"10.1587","volume":"E108.D","author":[{"given":"Chi","family":"ZHANG","sequence":"first","affiliation":[{"name":"The University of Tokyo"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Li","family":"TAO","sequence":"additional","affiliation":[{"name":"The University of Tokyo"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Toshihiko","family":"YAMASAKI","sequence":"additional","affiliation":[{"name":"The University of Tokyo"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"532","reference":[{"key":"1","doi-asserted-by":"crossref","unstructured":"[1] O. Vinyals, A. Toshev, S. Bengio, and D. Erhan, \u201cShow and tell: A neural image caption generator,\u201d Proc. IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp.3156-3164, 2015. 10.1109\/cvpr.2015.7298935","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"2","doi-asserted-by":"crossref","unstructured":"[2] L. Huang, W. Wang, J. Chen, and X.-Y. Wei, \u201cAttention on attention for image captioning,\u201d Proc. IEEE\/CVF International Conference on Computer Vision (ICCV), pp.4633-4642, 2019. 10.1109\/iccv.2019.00473","DOI":"10.1109\/ICCV.2019.00473"},{"key":"3","doi-asserted-by":"publisher","unstructured":"[3] X. Xiao, L. Wang, K. Ding, S. Xiang, and C. Pan, \u201cDeep hierarchical encoder-decoder network for image captioning,\u201d IEEE Trans. Multimedia, vol.21, no.11, pp.2942-2956, 2019. 10.1109\/tmm.2019.2915033","DOI":"10.1109\/TMM.2019.2915033"},{"key":"4","doi-asserted-by":"crossref","unstructured":"[4] J. Wang, W. Xu, Q. Wang, and A.B. Chan, \u201cCompare and reweight: Distinctive image captioning using similar images sets,\u201d Proc. European Conference on Computer Vision (ECCV), pp.370-386, 2020. 10.1007\/978-3-030-58452-8_22","DOI":"10.1007\/978-3-030-58452-8_22"},{"key":"5","doi-asserted-by":"publisher","unstructured":"[5] A. Mathews, L. Xie, and X. He, \u201cSenticap: Generating image descriptions with sentiments,\u201d Proc. AAAI Conference on Artificial Intelligence (AAAI), vol.30, no.1, 2016. 10.1609\/aaai.v30i1.10475","DOI":"10.1609\/aaai.v30i1.10475"},{"key":"6","doi-asserted-by":"crossref","unstructured":"[6] C. Gan, Z. Gan, X. He, J. Gao, and L. Deng, \u201cStylenet: Generating attractive visual captions with styles,\u201d Proc. IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp.955-964, 2017. 10.1109\/cvpr.2017.108","DOI":"10.1109\/CVPR.2017.108"},{"key":"7","doi-asserted-by":"crossref","unstructured":"[7] T. Chen, Z. Zhang, Q. You, C. Fang, Z. Wang, H. Jin, and J. Luo, \u201c\u201cfactual\u201d or \u201cemotional\u201d: Stylized image captioning with adaptive learning and attention,\u201d Proc. European Conference on Computer Vision (ECCV), pp.527-543, 2018. 10.1007\/978-3-030-01249-6_32","DOI":"10.1007\/978-3-030-01249-6_32"},{"key":"8","doi-asserted-by":"crossref","unstructured":"[8] L. Guo, J. Liu, P. Yao, J. Li, and H. Lu, \u201cMscap: Multi-style image captioning with unpaired stylized text,\u201d Proc. IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp.4199-4208, 2019. 10.1109\/cvpr.2019.00433","DOI":"10.1109\/CVPR.2019.00433"},{"key":"9","doi-asserted-by":"crossref","unstructured":"[9] O.M. Nezami, M. Dras, S. Wan, C. Paris, and L. Hamey, \u201cTowards generating stylized image captions via adversarial training,\u201d Pacific Rim International Conference on Artificial Intelligence (PRICAI), pp.270-284, 2019. 10.1007\/978-3-030-29908-8_22","DOI":"10.1007\/978-3-030-29908-8_22"},{"key":"10","doi-asserted-by":"crossref","unstructured":"[10] Y. Tan, Z. Lin, H. Liu, and F. Zuo, \u201cImproving stylized image captioning with better use of transformer,\u201d International Conference on Artificial Neural Networks, pp.347-358, 2022. 10.1007\/978-3-031-15934-3_29","DOI":"10.1007\/978-3-031-15934-3_29"},{"key":"11","doi-asserted-by":"crossref","unstructured":"[11] M. Wu, X. Zhang, X. Sun, Y. Zhou, C. Chen, J. Gu, X. Sun, and R. Ji, \u201cDifnet: Boosting visual information flow for image captioning,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp.18020-18029, 2022. 10.1109\/cvpr52688.2022.01749","DOI":"10.1109\/CVPR52688.2022.01749"},{"key":"12","doi-asserted-by":"crossref","unstructured":"[12] T.Y. Lin, M. Maire, S. Belongie, J. Hays, P. Perona, D. Ramanan, P. Doll\u00e1r, and C.L. Zitnick, \u201cMicrosoft coco: Common objects in context,\u201d Proc. European Conference on Computer Vision (ECCV), pp.740-755, 2014. 10.1007\/978-3-319-10602-1_48","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"13","unstructured":"[13] J. Devlin, M.-W. Chang, K. Lee, and K. Toutanova, \u201cBERT: Pre-training of deep bidirectional transformers for language understanding,\u201d Proc. 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, vol.1 (Long and Short Papers), 2019. 10.18653\/v1\/N19-1423"},{"key":"14","doi-asserted-by":"publisher","unstructured":"[14] L. Zhou, H. Palangi, L. Zhang, H. Hu, J. Corso, and J. Gao, \u201cUnified vision-language pre-training for image captioning and vqa,\u201d Proc. AAAI Conference on Artificial Intelligence (AAAI), vol.34, no.07, pp.13041-13049, 2020. 10.1609\/aaai.v34i07.7005","DOI":"10.1609\/aaai.v34i07.7005"},{"key":"15","doi-asserted-by":"crossref","unstructured":"[15] C. Deng, N. Ding, M. Tan, and Q. Wu, \u201cLength-controllable image captioning,\u201d Proc. European Conference on Computer Vision (ECCV), pp.712-729, 2020. 10.1007\/978-3-030-58601-0_42","DOI":"10.1007\/978-3-030-58601-0_42"},{"key":"16","doi-asserted-by":"crossref","unstructured":"[16] R. Vedantam, C. Lawrence Zitnick, and D. Parikh, \u201cCider: Consensus-based image description evaluation,\u201d Proc. IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp.4566-4575, 2015. 10.1109\/cvpr.2015.7299087","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"17","unstructured":"[17] C.-Y. Lin, \u201cROUGE: A Package for Automatic Evaluation of Summaries,\u201d Proc. Workshop on Text Summarization Branches Out, Association for Computational Linguistics, pp.74-81, 2004."},{"key":"18","doi-asserted-by":"publisher","unstructured":"[18] W. Zhao, X. Wu, and X. Zhang, \u201cMemcap: Memorizing style knowledge for image captioning,\u201d Proc. AAAI Conference on Artificial Intelligence, vol.34, no.07, pp.12984-12992, 2020. 10.1609\/aaai.v34i07.6998","DOI":"10.1609\/aaai.v34i07.6998"},{"key":"19","doi-asserted-by":"publisher","unstructured":"[19] X. Wu and T. Li, \u201cSentimental visual captioning using multimodal transformer,\u201d International Journal of Computer Vision, vol.131, no.4, pp.1073-1090, 2023. 10.1007\/s11263-023-01752-7","DOI":"10.1007\/s11263-023-01752-7"},{"key":"20","doi-asserted-by":"crossref","unstructured":"[20] P. Achlioptas, M. Ovsjanikov, L. Guibas, and S. Tulyakov, \u201cAffection: Learning affective explanations for real-world visual data,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.6641-6651, 2023. 10.1109\/cvpr52729.2023.00642","DOI":"10.1109\/CVPR52729.2023.00642"},{"key":"21","doi-asserted-by":"crossref","unstructured":"[21] J. Gu, S. Joty, J. Cai, H. Zhao, X. Yang, and G. Wang, \u201cUnpaired image captioning via scene graph alignments,\u201d Proc. IEEE\/CVF International Conference on Computer Vision (ICCV), pp.10322-10331, 2019. 10.1109\/iccv.2019.01042","DOI":"10.1109\/ICCV.2019.01042"},{"key":"22","doi-asserted-by":"publisher","unstructured":"[22] X. Li and S. Jiang, \u201cKnow more say less: Image captioning based on scene graphs,\u201d IEEE Trans. Multimedia, vol.21, no.8, pp.2117-2130, 2019. 10.1109\/tmm.2019.2896516","DOI":"10.1109\/TMM.2019.2896516"},{"key":"23","doi-asserted-by":"crossref","unstructured":"[23] Y. Zhong, L. Wang, J. Chen, D. Yu, and Y. Li, \u201cComprehensive image captioning via scene graph decomposition,\u201d Proc. European Conference on Computer Vision (ECCV), pp.211-229, 2020. 10.1007\/978-3-030-58568-6_13","DOI":"10.1007\/978-3-030-58568-6_13"},{"key":"24","doi-asserted-by":"publisher","unstructured":"[24] W. Zhao and X. Wu, \u201cBoosting entity-aware image captioning with multi-modal knowledge graph,\u201d IEEE Trans. Multimedia, vol.26, pp.2659-2670, 2024. 10.1109\/tmm.2023.3301279","DOI":"10.1109\/TMM.2023.3301279"},{"key":"25","doi-asserted-by":"crossref","unstructured":"[25] F. Sammani and L. Melas-Kyriazi, \u201cShow, edit and tell: A framework for editing image captions,\u201d Proc. IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp.4807-4815, 2020. 10.1109\/cvpr42600.2020.00486","DOI":"10.1109\/CVPR42600.2020.00486"},{"key":"26","unstructured":"[26] T. Brown, B. Mann, N. Ryder, M. Subbiah, J.D. Kaplan, P. Dhariwal, A. Neelakantan, P. Shyam, G. Sastry, A. Askell, et al., \u201cLanguage models are few-shot learners,\u201d Proc. Advances in neural information processing systems (NeurIPS), vol.33, pp.1877-1901, 2020. abs\/10.5555\/3495724.3495883"},{"key":"27","unstructured":"[27] H. Touvron, L. Martin, K. Stone, P. Albert, A. Almahairi, Y. Babaei, N. Bashlykov, S. Batra, P. Bhargava, S. Bhosale, et al., \u201cLlama 2: Open foundation and fine-tuned chat models,\u201d arXiv preprint arXiv:2307.09288, 2023."},{"key":"28","unstructured":"[28] W.L. Chiang, Z. Li, Z. Lin, Y. Sheng, Z. Wu, H. Zhang, L. Zheng, S. Zhuang, Y. Zhuang, J.E. Gonzalez, I. Stoica, and E.P. Xing, \u201cVicuna: An open-source chatbot impressing gpt-4 with 90%* chatgpt quality,\u201d March 2023."},{"key":"29","unstructured":"[29] H. Liu, C. Li, Q. Wu, and Y.J. Lee, \u201cVisual instruction tuning,\u201d Proc. Advances in neural information processing systems (NeurIPS), vol.36, pp.34892-34916, 2023. abs\/10.5555\/3666122.3667638"},{"key":"30","doi-asserted-by":"crossref","unstructured":"[30] N. Rotstein, D. Bensa\u0131\u0301d, S. Brody, R. Ganz, and R. Kimmel, \u201cFusecap: Leveraging large language models for enriched fused image captions,\u201d Proc. IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV), pp.5677-5688, 2024. 10.1109\/wacv57701.2024.00559","DOI":"10.1109\/WACV57701.2024.00559"},{"key":"31","doi-asserted-by":"crossref","unstructured":"[31] J. Zhang, L. Zheng, D. Guo, and M. Wang, \u201cTraining a small emo- tional vision language model for visual art comprehension,\u201d Proc. European Conference on Computer Vision (ECCV), pp.397-413, 2025.","DOI":"10.1007\/978-3-031-72855-6_23"},{"key":"32","unstructured":"[32] A. Radford, J. Wu, R. Child, D. Luan, D. Amodei, I. Sutskever, et al., \u201cLanguage models are unsupervised multitask learners,\u201d OpenAI blog, vol.1, no.8, p.9, 2019."},{"key":"33","unstructured":"[33] OpenAI et al., \u201cGPT-4 technical report,\u201d arXiv preprint arXiv:2303.08774, 2023."},{"key":"34","doi-asserted-by":"crossref","unstructured":"[34] J. Shi, Y. Li, and S. Wang, \u201cPartial off-policy learning: Balance accuracy and diversity for human-oriented image captioning,\u201d Proc. IEEE\/CVF International Conference on Computer Vision (ICCV), pp.2167-2176, Oct. 2021. 10.1109\/iccv48922.2021.00219","DOI":"10.1109\/ICCV48922.2021.00219"},{"key":"35","unstructured":"[35] V. Padmakumar and H. He, \u201cDoes writing with language models reduce content diversity?,\u201d Proc. International Conference on Learning Representations (ICLR), 2024."},{"key":"36","doi-asserted-by":"publisher","unstructured":"[36] C. Meister, T. Pimentel, G. Wiher, and R. Cotterell, \u201cLocally typical sampling,\u201d Transactions of the Association for Computational Linguistics, vol.11, pp.102-121, 2023. 10.1162\/tacl_a_00536","DOI":"10.1162\/tacl_a_00536"},{"key":"37","unstructured":"[37] J.L. Ba, J.R. Kiros, and G.E. Hinton, \u201cLayer normalization,\u201d arXiv preprint arXiv:1607.06450, 2016."},{"key":"38","doi-asserted-by":"crossref","unstructured":"[38] S.J. Rennie, E. Marcheret, Y. Mroueh, J. Ross, and V. Goel, \u201cSelf-critical sequence training for image captioning,\u201d Proc. IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp.1179-1195, 2017. 10.1109\/cvpr.2017.131","DOI":"10.1109\/CVPR.2017.131"},{"key":"39","unstructured":"[39] A. Vaswani, N. Shazeer, N. Parmar, J. Uszkoreit, L. Jones, A.N. Gomez, L.u. Kaiser, and I. Polosukhin, \u201cAttention is all you need,\u201d Proc. Advances in Neural Information Processing Systems (NeurIPS), pp.6000-6010, 2017. 10.5555\/3295222.3295349"},{"key":"40","doi-asserted-by":"crossref","unstructured":"[40] J. Chen, H. Guo, K. Yi, B. Li, and M. Elhoseiny, \u201cVisualgpt: Data-efficient adaptation of pretrained language models for image captioning,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp.18030-18040, June 2022. 10.1109\/cvpr52688.2022.01750","DOI":"10.1109\/CVPR52688.2022.01750"},{"key":"41","unstructured":"[41] W. Dai, J. Li, D. Li, A. Tiong, J. Zhao, W. Wang, B. Li, P.N. Fung, and S. Hoi, \u201cInstructblip: Towards general-purpose vision-language models with instruction tuning,\u201d Proc. Advances in Neural Information Processing Systems (NeurIPS), pp.49250-49267, 2023. 10.5555\/3666122.3668264"},{"key":"42","unstructured":"[42] K. Xu, J. Ba, R. Kiros, K. Cho, A. Courville, R. Salakhudinov, R. Zemel, and Y. Bengio, \u201cShow, attend and tell: Neural image caption generation with visual attention,\u201d Proc. 32nd International Conference on Machine Learning (ICML), vol.37, pp.2048-2057, 2015. 10.5555\/3045118.3045336"},{"key":"43","doi-asserted-by":"crossref","unstructured":"[43] J. Lu, C. Xiong, D. Parikh, and R. Socher, \u201cKnowing when to look: Adaptive attention via a visual sentinel for image captioning,\u201d Proc. IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp.3242-3250, 2017. 10.1109\/cvpr.2017.345","DOI":"10.1109\/CVPR.2017.345"},{"key":"44","doi-asserted-by":"crossref","unstructured":"[44] K. Papineni, S. Roukos, T. Ward, and W.-J. Zhu, \u201cBleu: a method for automatic evaluation of machine translation,\u201d Proc. 40th Annual Meeting of the Association for Computational Linguistics (ACL), pp.311-318, 2002. 10.3115\/1073083.1073135","DOI":"10.3115\/1073083.1073135"},{"key":"45","unstructured":"[45] S. Banerjee and A. Lavie, \u201cMeteor: An automatic metric for mt evaluation with improved correlation with human judgments,\u201d Proc. ACL Workshop on Intrinsic and Extrinsic Evaluation Measures for Machine Translation and\/or Summarization, pp.65-72, 2005. 10.5555\/1626355.1626389"},{"key":"46","doi-asserted-by":"crossref","unstructured":"[46] Z. Wang, B. Feng, K. Narasimhan, and O. Russakovsky, \u201cTowards unique and informative captioning of images,\u201d Proc. European Conference on Computer Vision (ECCV), pp.629-644, 2020. 10.1007\/978-3-030-58571-6_37","DOI":"10.1007\/978-3-030-58571-6_37"},{"key":"47","unstructured":"[47] S. Ren, K. He, R. Girshick, and J. Sun, \u201cFaster r-cnn: Towards real-time object detection with region proposal networks,\u201d Proc. Advances in Neural Information Processing Systems (NeurIPS), pp.91-99, 2015. 10.5555\/2969239.2969250"},{"key":"48","doi-asserted-by":"publisher","unstructured":"[48] R. Krishna, Y. Zhu, O. Groth, J. Johnson, K. Hata, J. Kravitz, S. Chen, Y. Kalantidis, L.-J. Li, D.A. Shamma, M.S. Bernstein, and L. Fei-Fei, \u201cVisual genome: Connecting language and vision using crowdsourced dense image annotations,\u201d International Journal of Computer Vision, vol.123, no.1, pp.32-73, 2017. 10.1007\/s11263-016-0981-7","DOI":"10.1007\/s11263-016-0981-7"},{"key":"49","doi-asserted-by":"crossref","unstructured":"[49] R. Luo, B. Price, S. Cohen, and G. Shakhnarovich, \u201cDiscriminability objective for training descriptive captions,\u201d arXiv preprint arXiv:1803.04376, 2018.","DOI":"10.1109\/CVPR.2018.00728"},{"key":"50","doi-asserted-by":"publisher","unstructured":"[50] X. Wu, W. Zhao, and J. Luo, \u201cLearning cooperative neural modules for stylized image captioning,\u201d International Journal of Computer Vision, vol.130, no.9, pp.2305-2320, 2022. 10.1007\/s11263-022-01636-2","DOI":"10.1007\/s11263-022-01636-2"}],"container-title":["IEICE Transactions on Information and Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E108.D\/3\/E108.D_2024EDP7036\/_pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,3,1]],"date-time":"2025-03-01T03:33:34Z","timestamp":1740800014000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transinf\/E108.D\/3\/E108.D_2024EDP7036\/_article"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,3,1]]},"references-count":50,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2025]]}},"URL":"https:\/\/doi.org\/10.1587\/transinf.2024edp7036","relation":{},"ISSN":["0916-8532","1745-1361"],"issn-type":[{"type":"print","value":"0916-8532"},{"type":"electronic","value":"1745-1361"}],"subject":[],"published":{"date-parts":[[2025,3,1]]},"article-number":"2024EDP7036"}}