{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T03:50:29Z","timestamp":1783050629496,"version":"3.54.6"},"publisher-location":"Cham","reference-count":70,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031732256","type":"print"},{"value":"9783031732263","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-73226-3_2","type":"book-chapter","created":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T15:02:57Z","timestamp":1730386977000},"page":"18-36","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Affective Visual Dialog: A Large-Scale Benchmark for\u00a0Emotional Reasoning Based on\u00a0Visually Grounded Conversations"],"prefix":"10.1007","author":[{"given":"Kilichbek","family":"Haydarov","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaoqian","family":"Shen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Avinash","family":"Madasu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mahmoud","family":"Salem","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Li-Jia","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Gamaleldin","family":"Elsayed","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mohamed","family":"Elhoseiny","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,11,1]]},"reference":[{"key":"2_CR1","unstructured":"Openai api: Gpt-4 (2023). https:\/\/platform.openai.com\/docs\/models\/gpt-4. Accessed 15 Nov 2023"},{"key":"2_CR2","doi-asserted-by":"crossref","unstructured":"Achlioptas, P., Ovsjanikov, M., Guibas, L., Tulyakov, S.: Affection: learning affective explanations for real-world visual data. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 6641\u20136651, June 2023","DOI":"10.1109\/CVPR52729.2023.00642"},{"key":"2_CR3","doi-asserted-by":"crossref","unstructured":"Achlioptas, P., Ovsjanikov, M., Haydarov, K., Elhoseiny, M., Guibas, L.J.: ArtEmis: affective language for visual art. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11569\u201311579 (2021)","DOI":"10.1109\/CVPR46437.2021.01140"},{"key":"2_CR4","doi-asserted-by":"crossref","unstructured":"Agrawal, A., Batra, D., Parikh, D., Kembhavi, A.: Don\u2019t just assume; look and answer: overcoming priors for visual question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4971\u20134980 (2018)","DOI":"10.1109\/CVPR.2018.00522"},{"key":"2_CR5","doi-asserted-by":"crossref","unstructured":"Antol, S., et al.: VQA: visual question answering. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2425\u20132433 (2015)","DOI":"10.1109\/ICCV.2015.279"},{"key":"2_CR6","unstructured":"Bai, Y., et\u00a0al.: Training a helpful and harmless assistant with reinforcement learning from human feedback. arXiv preprint arXiv:2204.05862 (2022)"},{"key":"2_CR7","unstructured":"Black, K., Janner, M., Du, Y., Kostrikov, I., Levine, S.: Training diffusion models with reinforcement learning (2023)"},{"key":"#cr-split#-2_CR8.1","unstructured":"Bogdanov, D., Porter, A., Tovstogan, P., Won, M.: MediaEval 2019: emotion and theme recognition in music using Jamendo. In: Larson, M., (eds.) MediaEval 2019, Multimedia Benchmark Workshop"},{"key":"#cr-split#-2_CR8.2","unstructured":"27-30 October 2019, Sophia Antipolis, France. CEUR, Aachen. CEUR Workshop Proceedings (2019)"},{"key":"2_CR9","doi-asserted-by":"crossref","unstructured":"Brooks, T., Holynski, A., Efros, A.A.: InstructPix2Pix: learning to follow image editing instructions. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 18392\u201318402, June 2023","DOI":"10.1109\/CVPR52729.2023.01764"},{"key":"2_CR10","unstructured":"Buechel, S., Hahn, U.: EmoBank: studying the impact of annotation perspective and representation format on dimensional emotion analysis. arXiv preprint arXiv:2205.01996 (2022)"},{"key":"2_CR11","doi-asserted-by":"crossref","unstructured":"Chen, C., et al.: UTC: a unified transformer with inter-task contrastive learning for visual dialog. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18103\u201318112 (2022)","DOI":"10.1109\/CVPR52688.2022.01757"},{"key":"2_CR12","unstructured":"Chen, J., Zhu, D., Haydarov, K., Li, X., Elhoseiny, M.: Video ChatCaptioner: towards the enriched spatiotemporal descriptions. arXiv preprint arXiv:2304.04227 (2023)"},{"key":"2_CR13","unstructured":"Chen, J., et al.: MiniGPT-V2: large language model as a unified interface for vision-language multi-task learning (2023). https:\/\/arxiv.org\/abs\/2310.09478"},{"key":"2_CR14","unstructured":"Chen, S.Y., Hsu, C.C., Kuo, C.C., Ku, L.W., et\u00a0al.: EmotionLines: an emotion corpus of multi-party conversations. arXiv preprint arXiv:1802.08379 (2018)"},{"key":"2_CR15","unstructured":"Community: Wiki art (2020). https:\/\/www.wikiart.org\/. Accessed 6 Nov 2020"},{"issue":"4","key":"2_CR16","doi-asserted-by":"publisher","first-page":"1924","DOI":"10.1073\/pnas.1910704117","volume":"117","author":"AS Cowen","year":"2020","unstructured":"Cowen, A.S., Fang, X., Sauter, D., Keltner, D.: What music makes us feel: at least 13 dimensions organize subjective experiences associated with music across different cultures. Proc. Nat. Acad. Sci. 117(4), 1924\u20131934 (2020)","journal-title":"Proc. Nat. Acad. Sci."},{"key":"2_CR17","doi-asserted-by":"crossref","unstructured":"Das, A., et al.: Visual dialog. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 326\u2013335 (2017)","DOI":"10.1109\/CVPR.2017.121"},{"key":"2_CR18","doi-asserted-by":"crossref","unstructured":"Demszky, D., Movshovitz-Attias, D., Ko, J., Cowen, A., Nemade, G., Ravi, S.: GoEmotions: a dataset of fine-grained emotions. arXiv preprint arXiv:2005.00547 (2020)","DOI":"10.18653\/v1\/2020.acl-main.372"},{"key":"2_CR19","doi-asserted-by":"crossref","unstructured":"Diener, E., Scollon, C.N., Lucas, R.E.: The evolving concept of subjective well-being: the multifaceted nature of happiness (2009)","DOI":"10.1007\/978-90-481-2354-4_4"},{"issue":"3\u20134","key":"2_CR20","doi-asserted-by":"publisher","first-page":"169","DOI":"10.1080\/02699939208411068","volume":"6","author":"P Ekman","year":"1992","unstructured":"Ekman, P.: An argument for basic emotions. Cogn. Emotion 6(3\u20134), 169\u2013200 (1992)","journal-title":"Cogn. Emotion"},{"key":"2_CR21","doi-asserted-by":"crossref","unstructured":"Fan, J., Thorogood, M., Pasquier, P.: Emo-soundscapes: a dataset for soundscape emotion recognition. In: 2017 Seventh International Conference on Affective Computing and Intelligent Interaction (ACII), pp. 196\u2013201. IEEE (2017)","DOI":"10.1109\/ACII.2017.8273600"},{"key":"2_CR22","doi-asserted-by":"crossref","unstructured":"Goyal, R., et\u00a0al.: The \u201csomething something\u201d video database for learning and evaluating visual common sense. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 5842\u20135850 (2017)","DOI":"10.1109\/ICCV.2017.622"},{"key":"2_CR23","doi-asserted-by":"crossref","unstructured":"Goyal, Y., Khot, T., Summers-Stay, D., Batra, D., Parikh, D.: Making the V in VQA matter: elevating the role of image understanding in visual question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6904\u20136913 (2017)","DOI":"10.1109\/CVPR.2017.670"},{"key":"2_CR24","doi-asserted-by":"crossref","unstructured":"Gunes, H., Piccardi, M.: A bimodal face and body gesture database for automatic analysis of human nonverbal affective behavior. In: 18th International Conference on Pattern Recognition (ICPR 2006), vol.\u00a01, pp. 1148\u20131153. IEEE (2006)","DOI":"10.1109\/ICPR.2006.39"},{"key":"2_CR25","unstructured":"Hung, H.T., Ching, J., Doh, S., Kim, N., Nam, J., Yang, Y.H.: EMOPIA: a multi-modal pop piano dataset for emotion recognition and emotion-based music generation. arXiv preprint arXiv:2108.01374 (2021)"},{"key":"2_CR26","unstructured":"Kottur, S., Moura, J.M., Parikh, D., Batra, D., Rohrbach, M.: CLEVR-Dialog: a diagnostic dataset for multi-round reasoning in visual dialog. arXiv preprint arXiv:1903.03166 (2019)"},{"key":"2_CR27","unstructured":"Lee, K., et al.: Aligning text-to-image models using human feedback. arXiv preprint arXiv:2302.12192 (2023)"},{"key":"2_CR28","doi-asserted-by":"crossref","unstructured":"Lewis, M., et al.: BART: denoising sequence-to-sequence pre-training for natural language generation, translation, and comprehension. In: Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics, pp. 7871\u20137880 (2020)","DOI":"10.18653\/v1\/2020.acl-main.703"},{"key":"2_CR29","doi-asserted-by":"crossref","unstructured":"Li, D., Wu, H., Zhang, J., Huang, K.: A2-RL: aesthetics aware reinforcement learning for image cropping. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 8193\u20138201 (2018)","DOI":"10.1109\/CVPR.2018.00855"},{"key":"2_CR30","unstructured":"Li, H., Zhu, S.C., Zheng, Z.: DiPlomat: a dialogue dataset for situated pragmatic reasoning. In: Advances in Neural Information Processing Systems, vol. 36 (2024)"},{"key":"2_CR31","unstructured":"Li, J., Li, D., Xiong, C., Hoi, S.: BLIP: bootstrapping language-image pre-training for unified vision-language understanding and generation. In: ICML (2022)"},{"key":"2_CR32","unstructured":"Li, Y., Su, H., Shen, X., Li, W., Cao, Z., Niu, S.: DailyDialog: a manually labelled multi-turn dialogue dataset. arXiv preprint arXiv:1710.03957 (2017)"},{"key":"2_CR33","doi-asserted-by":"publisher","unstructured":"Lin, T.Y., et al.: Microsoft COCO: common objects in context. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) Computer Vision \u2013 ECCV 2014. ECCV 2014. LNCS, vol. 8693, pp. 740\u2013755. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10602-1_48","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"2_CR34","doi-asserted-by":"crossref","unstructured":"Liu, X., Shi, H., Chen, H., Yu, Z., Li, X., Zhao, G.: iMiGUE: an identity-free video dataset for micro-gesture understanding and emotion analysis. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10631\u201310642 (2021)","DOI":"10.1109\/CVPR46437.2021.01049"},{"key":"2_CR35","doi-asserted-by":"publisher","unstructured":"Liu, Y., Iter, D., Xu, Y., Wang, S., Xu, R., Zhu, C.: G-Eval: NLG evaluation using GPT-4 with better human alignment. In: Bouamor, H., Pino, J., Bali, K. (eds.) Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, pp. 2511\u20132522. Association for Computational Linguistics, Singapore, December 2023). https:\/\/doi.org\/10.18653\/v1\/2023.emnlp-main.153, https:\/\/aclanthology.org\/2023.emnlp-main.153","DOI":"10.18653\/v1\/2023.emnlp-main.153"},{"key":"2_CR36","unstructured":"Liu, Y., et al.: RoBERTa: a robustly optimized BERT pretraining approach. arXiv preprint arXiv:1907.11692 (2019)"},{"key":"2_CR37","unstructured":"Lu, J., Batra, D., Parikh, D., Lee, S.: ViLBERT: pretraining task-agnostic visiolinguistic representations for vision-and-language tasks. In: Advances in Neural Information Processing Systems, vol. 32 (2019)"},{"key":"2_CR38","doi-asserted-by":"crossref","unstructured":"Machajdik, J., Hanbury, A.: Affective image classification using features inspired by psychology and art theory. In: Proceedings of the 18th ACM International Conference on Multimedia, pp. 83\u201392 (2010)","DOI":"10.1145\/1873951.1873965"},{"key":"2_CR39","doi-asserted-by":"crossref","unstructured":"Mohamed, Y., et al.: ArtELingo: a million emotion annotations of WikiArt with emphasis on diversity over language and culture. In: Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing (EMNLP) (2022)","DOI":"10.18653\/v1\/2022.emnlp-main.600"},{"key":"2_CR40","doi-asserted-by":"crossref","unstructured":"Mohamed, Y., Khan, F.F., Haydarov, K., Elhoseiny, M.: It is okay to not be okay: overcoming emotional bias in affective image captioning by contrastive data collection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 21263\u201321272 (2022)","DOI":"10.1109\/CVPR52688.2022.02058"},{"issue":"1","key":"2_CR41","doi-asserted-by":"publisher","first-page":"18","DOI":"10.1109\/TAFFC.2017.2740923","volume":"10","author":"A Mollahosseini","year":"2017","unstructured":"Mollahosseini, A., Hasani, B., Mahoor, M.H.: AffectNet: a database for facial expression, valence, and arousal computing in the wild. IEEE Trans. Affect. Comput. 10(1), 18\u201331 (2017)","journal-title":"IEEE Trans. Affect. Comput."},{"key":"2_CR42","doi-asserted-by":"publisher","unstructured":"Murahari, V., Batra, D., Parikh, D., Das, A.: Large-scale pretraining for visual dialog: a simple state-of-the-art baseline. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.M. (eds.) European Conference on Computer Vision, pp. 336\u2013352. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58523-5_20","DOI":"10.1007\/978-3-030-58523-5_20"},{"key":"2_CR43","doi-asserted-by":"publisher","unstructured":"Murahari, V., Chattopadhyay, P., Batra, D., Parikh, D., Das, A.: Improving generative visual dialog by answering diverse questions. In: Inui, K., Jiang, J., Ng, V., Wan, X. (eds.) Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP), pp. 1449\u20131454. Association for Computational Linguistics, Hong Kong, China, November 2019. https:\/\/doi.org\/10.18653\/v1\/D19-1152, https:\/\/aclanthology.org\/D19-1152","DOI":"10.18653\/v1\/D19-1152"},{"key":"2_CR44","doi-asserted-by":"crossref","unstructured":"Murahari, V., Chattopadhyay, P., Batra, D., Parikh, D., Das, A.: Improving generative visual dialog by answering diverse questions. arXiv preprint arXiv:1909.10470 (2019)","DOI":"10.18653\/v1\/D19-1152"},{"key":"2_CR45","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"223","DOI":"10.1007\/978-3-030-58586-0_14","volume-title":"Computer Vision \u2013 ECCV 2020","author":"V-Q Nguyen","year":"2020","unstructured":"Nguyen, V.-Q., Suganuma, M., Okatani, T.: Efficient attention mechanism for visual dialog that can handle all the interactions between multiple inputs. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12369, pp. 223\u2013240. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58586-0_14"},{"key":"2_CR46","doi-asserted-by":"crossref","unstructured":"Papineni, K., Roukos, S., Ward, T., Zhu, W.J.: BLEU: a method for automatic evaluation of machine translation. In: Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics, pp. 311\u2013318 (2002)","DOI":"10.3115\/1073083.1073135"},{"key":"2_CR47","doi-asserted-by":"crossref","unstructured":"Pennington, J., Socher, R., Manning, C.D.: GloVe: global vectors for word representation. In: Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing (EMNLP), pp. 1532\u20131543 (2014)","DOI":"10.3115\/v1\/D14-1162"},{"issue":"140","key":"2_CR48","first-page":"1","volume":"21","author":"C Raffel","year":"2020","unstructured":"Raffel, C., et al.: Exploring the limits of transfer learning with a unified text-to-text transformer. J. Mach. Learn. Res. 21(140), 1\u201367 (2020)","journal-title":"J. Mach. Learn. Res."},{"key":"2_CR49","doi-asserted-by":"crossref","unstructured":"Ranganathan, H., Chakraborty, S., Panchanathan, S.: Multimodal emotion recognition using deep learning architectures. In: 2016 IEEE Winter Conference on Applications of Computer Vision (WACV), pp.\u00a01\u20139. IEEE (2016)","DOI":"10.1109\/WACV.2016.7477679"},{"key":"2_CR50","doi-asserted-by":"crossref","unstructured":"Rashkin, H., Smith, E.M., Li, M., Boureau, Y.L.: Towards empathetic open-domain conversation models: a new benchmark and dataset. In: Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics, pp. 5370\u20135381 (2019)","DOI":"10.18653\/v1\/P19-1534"},{"issue":"6","key":"2_CR51","doi-asserted-by":"publisher","first-page":"1161","DOI":"10.1037\/h0077714","volume":"39","author":"JA Russell","year":"1980","unstructured":"Russell, J.A.: A circumplex model of affect. J. Pers. Soc. Psychol. 39(6), 1161 (1980)","journal-title":"J. Pers. Soc. Psychol."},{"key":"2_CR52","unstructured":"Russell, S.: Human Compatible: Artificial Intelligence and the Problem of Control. Penguin (2019)"},{"key":"2_CR53","doi-asserted-by":"crossref","unstructured":"Sammani, F., Mukherjee, T., Deligiannis, N.: NLX-GPT: a model for natural language explanations in vision and vision-language tasks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8322\u20138332 (2022)","DOI":"10.1109\/CVPR52688.2022.00814"},{"key":"2_CR54","doi-asserted-by":"crossref","unstructured":"Sap, M., Rashkin, H., Chen, D., Le\u00a0Bras, R., Choi, Y.: Social IQA: commonsense reasoning about social interactions. In: Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP), pp. 4463\u20134473 (2019)","DOI":"10.18653\/v1\/D19-1454"},{"key":"2_CR55","unstructured":"Seo, P.H., Lehrmann, A., Han, B., Sigal, L.: Visual reference resolution using attention memory for visual dialog. In: Advances in Neural Information Processing Systems, vol. 30 (2017)"},{"key":"2_CR56","doi-asserted-by":"crossref","unstructured":"Sharma, P., Ding, N., Goodman, S., Soricut, R.: Conceptual captions: a cleaned, hypernymed, image alt-text dataset for automatic image captioning. In: Proceedings of ACL (2018)","DOI":"10.18653\/v1\/P18-1238"},{"key":"2_CR57","unstructured":"Simonyan, K., Zisserman, A.: Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556 (2014)"},{"key":"2_CR58","doi-asserted-by":"crossref","unstructured":"Strapparava, C., Mihalcea, R.: SemEval-2007 task 14: affective text. In: Proceedings of the Fourth International Workshop on Semantic Evaluations (SemEval-2007), pp. 70\u201374 (2007)","DOI":"10.3115\/1621474.1621487"},{"key":"2_CR59","unstructured":"Touvron, H., et\u00a0al.: Llama 2: open foundation and fine-tuned chat models. arXiv preprint arXiv:2307.09288 (2023)"},{"key":"2_CR60","doi-asserted-by":"publisher","unstructured":"Urbanek, J., Ringshia, P.: Mephisto: a framework for portable, reproducible, and iterative crowdsourcing (2023). https:\/\/doi.org\/10.48550\/ARXIV.2301.05154, https:\/\/arxiv.org\/abs\/2301.05154","DOI":"10.48550\/ARXIV.2301.05154"},{"key":"2_CR61","doi-asserted-by":"crossref","unstructured":"de\u00a0Vries, H., Strub, F., Chandar, S., Pietquin, O., Larochelle, H., Courville, A.: GuessWhat?! visual object discovery through multi-modal dialogue. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), July 2017","DOI":"10.1109\/CVPR.2017.475"},{"key":"2_CR62","doi-asserted-by":"crossref","unstructured":"Weng, S., Zhang, P., Chang, Z., Wang, X., Li, S., Shi, B.: Affective image filter: reflecting emotions from text to images. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 10810\u201310819, October 2023","DOI":"10.1109\/ICCV51070.2023.00992"},{"key":"2_CR63","doi-asserted-by":"crossref","unstructured":"Yanulevskaya, V., van Gemert, J.C., Roth, K., Herbold, A.K., Sebe, N., Geusebroek, J.M.: Emotional valence categorization using holistic image features. In: 2008 15th IEEE International Conference on Image Processing, pp. 101\u2013104. IEEE (2008)","DOI":"10.1109\/ICIP.2008.4711701"},{"key":"2_CR64","doi-asserted-by":"crossref","unstructured":"You, Q., Luo, J., Jin, H., Yang, J.: Building a large scale dataset for image emotion recognition: the fine print and the benchmark. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a030 (2016)","DOI":"10.1609\/aaai.v30i1.9987"},{"key":"2_CR65","doi-asserted-by":"publisher","first-page":"67","DOI":"10.1162\/tacl_a_00166","volume":"2","author":"P Young","year":"2014","unstructured":"Young, P., Lai, A., Hodosh, M., Hockenmaier, J.: From image descriptions to visual denotations: new similarity metrics for semantic inference over event descriptions. Trans. Assoc. Comput. Linguist. 2, 67\u201378 (2014)","journal-title":"Trans. Assoc. Comput. Linguist."},{"key":"2_CR66","first-page":"27263","volume":"34","author":"W Yuan","year":"2021","unstructured":"Yuan, W., Neubig, G., Liu, P.: BARTScore: evaluating generated text as text generation. Adv. Neural. Inf. Process. Syst. 34, 27263\u201327277 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"2_CR67","unstructured":"Zhang, T., Kishore, V., Wu, F., Weinberger, K.Q., Artzi, Y.: BERTScore: evaluating text generation with BERT. In: International Conference on Learning Representations (2019)"},{"key":"2_CR68","unstructured":"Zhou, X., et\u00a0al.: SOTOPIA: interactive evaluation for social intelligence in language agents. arXiv preprint arXiv:2310.11667 (2023)"},{"key":"2_CR69","unstructured":"Zhu, D., Chen, J., Haydarov, K., Shen, X., Zhang, W., Elhoseiny, M.: ChatGPT asks, BLIP-2 answers: automatic questioning towards enriched visual descriptions. arXiv preprint arXiv:2303.06594 (2023)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-73226-3_2","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,31]],"date-time":"2024-10-31T15:13:03Z","timestamp":1730387583000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-73226-3_2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,1]]},"ISBN":["9783031732256","9783031732263"],"references-count":70,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-73226-3_2","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,1]]},"assertion":[{"value":"1 November 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}