{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,6]],"date-time":"2026-07-06T17:15:42Z","timestamp":1783358142523,"version":"3.54.6"},"reference-count":66,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100015749","name":"Communication University of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100015749","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2027,1]]},"DOI":"10.1016\/j.eswa.2026.133457","type":"journal-article","created":{"date-parts":[[2026,6,29]],"date-time":"2026-06-29T16:13:40Z","timestamp":1782749620000},"page":"133457","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PA","title":["MMVUF: Improving video-based human value understanding in multimodal large language models"],"prefix":"10.1016","volume":"332","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-0778-2183","authenticated-orcid":false,"given":"Yuchen","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"He","family":"Chang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hao","family":"Qiu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Libiao","family":"Jin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhulin","family":"Tao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.133457_bib0001","unstructured":"Albalak, A., Elazar, Y., Xie, S. M., Longpre, S., Lambert, N., Wang, X., Muennighoff, N., Hou, B., Pan, L., Jeong, H. et al. (2024). A survey on data selection for language models. arXiv preprint arXiv: 2402.16827."},{"key":"10.1016\/j.eswa.2026.133457_bib0002","unstructured":"Askell, A., Bai, Y., Chen, A., Drain, D., Ganguli, D., Henighan, T., Jones, A., Joseph, N., Mann, B., DasSarma, N. et al. (2021). A general language assistant as a laboratory for alignment. arXiv preprint arXiv: 2112.00861."},{"key":"10.1016\/j.eswa.2026.133457_bib0003","unstructured":"Bai, S., Chen, K., Liu, X., Wang, J., Ge, W., Song, S., Dang, K., Wang, P., Wang, S., Tang, J. et al. (2025). Qwen2. 5-vl technical report. arXiv preprint arXiv: 2502.13923."},{"key":"10.1016\/j.eswa.2026.133457_bib0004","series-title":"ICML","first-page":"4","article-title":"Is space-time attention all you need for video understanding?","volume":"vol. 2","author":"Bertasius","year":"2021"},{"key":"10.1016\/j.eswa.2026.133457_bib0005","unstructured":"Biedma, P., Yi, X., Huang, L., Sun, M., & Xie, X. (2024). Beyond human norms: Unveiling unique values of large language models through interdisciplinary approaches. arXiv preprint arXiv: 2404.12744."},{"key":"10.1016\/j.eswa.2026.133457_bib0006","unstructured":"Cahyawijaya, S., Chen, D., Bang, Y., Khalatbari, L., Wilie, B., Ji, Z., Ishii, E., & Fung, P. (2024). High-dimension human value representation in large language models. arXiv preprint arXiv: 2404.07900."},{"key":"10.1016\/j.eswa.2026.133457_bib0007","unstructured":"Cai, Z., Cao, M., Chen, H., Chen, K., Chen, K., Chen, X., Chen, X., Chen, Z., Chen, Z., Chu, P. et al. (2024). InternLM2 technical report. arXiv preprint arXiv: 2403.17297."},{"key":"10.1016\/j.eswa.2026.133457_bib0008","doi-asserted-by":"crossref","first-page":"19472","DOI":"10.52202\/079017-0614","article-title":"ShareGPT4Video: Improving video understanding and generation with better captions","volume":"37","author":"Chen","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.133457_bib0009","unstructured":"Chen, Z., Wang, W., Cao, Y., Liu, Y., Gao, Z., Cui, E., Zhu, J., Ye, S., Tian, H., Liu, Z. et al. (2024b). Expanding performance boundaries of open-source multimodal models with model, data, and test-time scaling. arXiv preprint arXiv: 2412.05271."},{"key":"10.1016\/j.eswa.2026.133457_bib0010","series-title":"Proceedings of the 2019 conference of the north american chapter of the association for computational linguistics: human language technologies, volume 1 (long and short papers)","first-page":"4171","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2019"},{"key":"10.1016\/j.eswa.2026.133457_bib0011","doi-asserted-by":"crossref","unstructured":"Ding, B., Liu, L., Bing, L., Kruengkrai, C., Nguyen, T. H., Joty, S., Si, L., & Miao, C. (2020). DAGA: Data augmentation with a generation approach for low-resource tagging tasks. arXiv preprint arXiv: 2011.01549.","DOI":"10.18653\/v1\/2020.emnlp-main.488"},{"key":"10.1016\/j.eswa.2026.133457_bib0012","doi-asserted-by":"crossref","unstructured":"Edunov, S., Ott, M., Auli, M., & Grangier, D. (2018). Understanding back-translation at scale. arXiv preprint arXiv: 1808.09381.","DOI":"10.18653\/v1\/D18-1045"},{"key":"10.1016\/j.eswa.2026.133457_bib0013","unstructured":"Engstrom, L., Feldmann, A., & Madry, A. (2024). DsDm: Model-aware dataset selection with datamodels. arXiv preprint arXiv: 2401.12926."},{"key":"10.1016\/j.eswa.2026.133457_bib0014","unstructured":"Hong, W., Wang, W., Ding, M., Yu, W., Lv, Q., Wang, Y., Cheng, Y., Huang, S., Ji, J., Xue, Z. et al. (2024). CogVLM2: Visual language models for image and video understanding. arXiv preprint arXiv: 2408.16500."},{"issue":"2","key":"10.1016\/j.eswa.2026.133457_bib0015","first-page":"3","article-title":"LoRA: Low-rank adaptation of large language models","volume":"1","author":"Hu","year":"2022","journal-title":"ICLR"},{"key":"10.1016\/j.eswa.2026.133457_bib0016","unstructured":"Hurst, A., Lerer, A., Goucher, A. P., Perelman, A., Ramesh, A., Clark, A., Ostrow, A. J., Welihinda, A., Hayes, A., Radford, A. et al. (2024). GPT-4o system card. arXiv preprint arXiv: 2410.21276."},{"issue":"1","key":"10.1016\/j.eswa.2026.133457_bib0017","doi-asserted-by":"crossref","first-page":"27","DOI":"10.1007\/s44267-025-00099-6","article-title":"Efficient multimodal large language models: A survey","volume":"3","author":"Jin","year":"2025","journal-title":"Visual Intelligence"},{"key":"10.1016\/j.eswa.2026.133457_bib0018","series-title":"Proceedings of the 2019 conference of the north american chapter of the association for computational linguistics: Human language technologies, volume 1 (long and short papers)","first-page":"3609","article-title":"Submodular optimization-based diverse paraphrasing and its effectiveness in data augmentation","author":"Kumar","year":"2019"},{"key":"10.1016\/j.eswa.2026.133457_bib0019","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"9972","article-title":"Hierarchical conditional relation networks for video question answering","author":"Le","year":"2020"},{"key":"10.1016\/j.eswa.2026.133457_bib0020","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2025.111801","article-title":"U3M: Unbiased multiscale modal fusion model for multimodal semantic segmentation","volume":"168","author":"Li","year":"2025","journal-title":"Pattern Recognition"},{"key":"10.1016\/j.eswa.2026.133457_bib0021","unstructured":"Li, B., Zhang, Y., Guo, D., Zhang, R., Li, F., Zhang, H., Zhang, K., Zhang, P., Li, Y., Liu, Z. et al. (2024a). LLaVA-onevision: Easy visual task transfer. arXiv preprint arXiv: 2408.03326."},{"key":"10.1016\/j.eswa.2026.133457_bib0022","unstructured":"Li, L., Liu, Y., Yao, L., Zhang, P., An, C., Wang, L., Sun, X., Kong, L., & Liu, Q. (2024b). Temporal reasoning transfer from text to video. arXiv preprint arXiv: 2410.06166."},{"key":"10.1016\/j.eswa.2026.133457_bib0023","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"7083","article-title":"TSM: Temporal shift module for efficient video understanding","author":"Lin","year":"2019"},{"key":"10.1016\/j.eswa.2026.133457_bib0024","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"26689","article-title":"VILA: On pre-training for visual language models","author":"Lin","year":"2024"},{"key":"10.1016\/j.eswa.2026.133457_bib0025","unstructured":"Liu, Y. (a). A survey on visual understanding multimodal large language models. In Computer science undergradaute conference 2025@ XJTU."},{"key":"10.1016\/j.eswa.2026.133457_bib0026","unstructured":"Liu, Y., Ott, M., Goyal, N., Du, J., Joshi, M., Chen, D., Levy, O., Lewis, M., Zettlemoyer, L., & Stoyanov, V. (2019). RoBERTa: A robustly optimized BERT pretraining approach. arXiv preprint arXiv: 1907.11692."},{"key":"10.1016\/j.eswa.2026.133457_bib0027","doi-asserted-by":"crossref","first-page":"10117","DOI":"10.52202\/079017-0325","article-title":"TSDS: Data selection for task-specific model finetuning","volume":"37","author":"Liu","year":"2024","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.133457_bib0028","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"3202","article-title":"Video swin transformer","author":"Liu","year":"2022"},{"key":"10.1016\/j.eswa.2026.133457_bib0029","series-title":"Proceedings of the 62nd annual meeting of the association for computational linguistics (volume 1: Long papers)","first-page":"12585","article-title":"Video-chatGPT: Towards detailed video understanding via large vision and language models","author":"Maaz","year":"2024"},{"key":"10.1016\/j.eswa.2026.133457_bib0030","series-title":"European conference on computer vision","first-page":"304","article-title":"MM1: Methods, analysis and insights from multimodal llm pre-training","author":"McKinzie","year":"2024"},{"key":"10.1016\/j.eswa.2026.133457_bib0031","unstructured":"Moore, J., Deshpande, T., & Yang, D. (2024). Are large language models consistent over value-laden questions?arXiv preprint arXiv: 2407.02996."},{"issue":"3","key":"10.1016\/j.eswa.2026.133457_bib0032","doi-asserted-by":"crossref","first-page":"1065","DOI":"10.1214\/aoms\/1177704472","article-title":"On estimation of a probability density function and mode","volume":"33","author":"Parzen","year":"1962","journal-title":"The Annals of Mathematical Statistics"},{"key":"10.1016\/j.eswa.2026.133457_bib0033","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"11183","article-title":"ValueNet: A new dataset for human value driven dialogue system","volume":"vol. 36","author":"Qiu","year":"2022"},{"key":"10.1016\/j.eswa.2026.133457_sbref0034","series-title":"Proceedings of the 2019 conference on empirical methods in natural language processing","article-title":"Sentence-BERT: Sentence embeddings using siamese BERT-networks","author":"Reimers","year":"2019"},{"key":"10.1016\/j.eswa.2026.133457_bib0035","doi-asserted-by":"crossref","unstructured":"Schwartz, S. H. (b). An overview of the schwartz theory of basic values. Online Readings in Psychology and Culture, 2(1), 11.","DOI":"10.9707\/2307-0919.1116"},{"key":"10.1016\/j.eswa.2026.133457_bib0036","doi-asserted-by":"crossref","unstructured":"Shen, H., Clark, N., & Mitra, T. (2025). Mind the value-action gap: Do LLMs act in alignment with their values?arXiv preprint arXiv: 2501.15463.","DOI":"10.18653\/v1\/2025.emnlp-main.154"},{"key":"10.1016\/j.eswa.2026.133457_bib0037","unstructured":"Shi, Z., Wang, Z., Fan, H., Zhang, Z., Li, L., Zhang, Y., Yin, Z., Sheng, L., Qiao, Y., & Shao, J. (2024). Assessment of multimodal large language models in alignment with human values. arXiv preprint arXiv: 2403.17830."},{"issue":"9","key":"10.1016\/j.eswa.2026.133457_bib0038","doi-asserted-by":"crossref","first-page":"5311","DOI":"10.1109\/TKDE.2025.3527978","article-title":"How to bridge the gap between modalities: Survey on multimodal large language model","volume":"37","author":"Song","year":"2025","journal-title":"IEEE Transactions on Knowledge and Data Engineering"},{"key":"10.1016\/j.eswa.2026.133457_bib0039","doi-asserted-by":"crossref","unstructured":"Tan, H., Sun, F., Liu, S., Su, D., Cao, Q., Chen, X., Wang, J., Cai, X., Wang, Y., Shen, H. et al. (2025a). Too consistent to detect: A study of self-consistent errors in LLMs. arXiv preprint arXiv: 2505.17656.","DOI":"10.18653\/v1\/2025.emnlp-main.238"},{"key":"10.1016\/j.eswa.2026.133457_bib0040","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"7202","article-title":"Beyond human data: Aligning multimodal large language models by iterative self-evolution","volume":"vol. 39","author":"Tan","year":"2025"},{"key":"10.1016\/j.eswa.2026.133457_bib0041","unstructured":"Q. Team (2024a). Qwen2 technical report. arXiv preprint arXiv: 2407.10671."},{"key":"10.1016\/j.eswa.2026.133457_bib0042","unstructured":"Q. Team (2024b). Qwen2.5: A party of foundation models. https:\/\/qwenlm.github.io\/blog\/qwen2.5\/."},{"key":"10.1016\/j.eswa.2026.133457_bib0043","series-title":"Proceedings of the IEEE international conference on computer vision","first-page":"4489","article-title":"Learning spatiotemporal features with 3d convolutional networks","author":"Tran","year":"2015"},{"key":"10.1016\/j.eswa.2026.133457_bib0044","first-page":"5998","article-title":"Attention is all you need","volume":"30","author":"Vaswani","year":"2017","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.133457_bib0045","series-title":"Proceedings of the 33rd ACM international conference on multimedia","first-page":"9608","article-title":"SVGen: Interpretable vector graphics generation with large language models","author":"Wang","year":"2025"},{"key":"10.1016\/j.eswa.2026.133457_bib0046","series-title":"European conference on computer vision","first-page":"20","article-title":"Temporal segment networks: Towards good practices for deep action recognition","author":"Wang","year":"2016"},{"key":"10.1016\/j.eswa.2026.133457_bib0047","series-title":"Proceedings of the 7th ACM international conference on multimedia in Asia","article-title":"Can multimodal large language models understand human values in videos?","author":"Wang","year":"2025"},{"key":"10.1016\/j.eswa.2026.133457_bib0048","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2025.130170","article-title":"Multimodal understanding of human values in videos: A benchmark dataset and PLM-based method","volume":"638","author":"Wang","year":"2025","journal-title":"Neurocomputing"},{"key":"10.1016\/j.eswa.2026.133457_bib0049","unstructured":"Xia, M., Malladi, S., Gururangan, S., Arora, S., & Chen, D. (2024). LESS: Selecting influential data for targeted instruction tuning. arXiv preprint arXiv: 2402.04333."},{"key":"10.1016\/j.eswa.2026.133457_bib0050","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"13204","article-title":"Can i trust your answer? Visually grounded video question answering","author":"Xiao","year":"2024"},{"issue":"11","key":"10.1016\/j.eswa.2026.133457_bib0051","doi-asserted-by":"crossref","first-page":"13265","DOI":"10.1109\/TPAMI.2023.3292266","article-title":"Contrastive video question answering via video graph transformer","volume":"45","author":"Xiao","year":"2023","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.133457_bib0052","unstructured":"Yao, J., Yi, X., Wang, X., Gong, Y., & Xie, X. (2023a). Value FULCRA: Mapping large language models to the multidimensional spectrum of basic human values. arXiv preprint arXiv: 2311.10766."},{"key":"10.1016\/j.eswa.2026.133457_bib0053","unstructured":"Yao, J., Yi, X., Wang, X., Wang, J., & Xie, X. (2023b). From instructions to intrinsic human values\u2013a survey of alignment goals for big models. arXiv preprint arXiv: 2308.12014."},{"key":"10.1016\/j.eswa.2026.133457_bib0054","unstructured":"Yao, Y., Yu, T., Zhang, A., Wang, C., Cui, J., Zhu, H., Cai, T., Li, H., Zhao, W., He, Z. et al. (2024). MiniCPM-V: A GPT-4V level MLLM on your phone. arXiv preprint arXiv: 2408.01800."},{"key":"10.1016\/j.eswa.2026.133457_bib0055","unstructured":"Yi, X., Yao, J., Wang, X., & Xie, X. (2023). Unpacking the ethical value alignment in big models. arXiv preprint arXiv: 2310.17551."},{"issue":"12","key":"10.1016\/j.eswa.2026.133457_bib0056","doi-asserted-by":"crossref","DOI":"10.1093\/nsr\/nwae403","article-title":"A survey on multimodal large language models","volume":"11","author":"Yin","year":"2024","journal-title":"National Science Review"},{"issue":"1","key":"10.1016\/j.eswa.2026.133457_bib0057","doi-asserted-by":"crossref","first-page":"18","DOI":"10.1007\/s11263-025-02613-1","article-title":"SafeBench: A safety evaluation framework for multimodal large language models","volume":"134","author":"Ying","year":"2026","journal-title":"International Journal of Computer Vision"},{"key":"10.1016\/j.eswa.2026.133457_bib0058","unstructured":"Zhang, B., Li, K., Cheng, Z., Hu, Z., Yuan, Y., Chen, G., Leng, S., Jiang, Y., Zhang, H., Li, X. et al. (2025a). VideoLLaMA 3: Frontier multimodal foundation models for image and video understanding. arXiv preprint arXiv: 2501.13106."},{"key":"10.1016\/j.eswa.2026.133457_bib0059","series-title":"Proceedings of the 33rd ACM international conference on multimedia","first-page":"3212","article-title":"KAID: Knowledge-aware interactive distillation for vision-language models","author":"Zhang","year":"2025"},{"key":"10.1016\/j.eswa.2026.133457_bib0060","unstructured":"Zhang, Y., Wu, J., Li, W., Li, B., Ma, Z., Liu, Z., & Li, C. (2024a). LLaVA-video: Video instruction tuning with synthetic data. arXiv preprint arXiv: 2410.02713."},{"key":"10.1016\/j.eswa.2026.133457_bib0061","unstructured":"Zhang, Y., Wu, J., Li, W., Li, B., Ma, Z., Liu, Z., & Li, C. (2024b). Video instruction tuning with synthetic data. arXiv preprint arXiv: 2410.02713."},{"key":"10.1016\/j.eswa.2026.133457_bib0062","doi-asserted-by":"crossref","first-page":"46595","DOI":"10.52202\/075280-2020","article-title":"Judging LLM-as-a-judge with MT-bench and chatbot arena","volume":"36","author":"Zheng","year":"2023","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.133457_bib0063","unstructured":"Chang, H., Ye, C., Tao, Z., Wu, J., Yang, Z., Ma, Y., Huang, X., & Chua, T.-S. (2024). A comprehensive evaluation of large language models on temporal event forecasting. arXiv preprint arXiv: 2407.11638."},{"key":"10.1016\/j.eswa.2026.133457_bib0064","doi-asserted-by":"crossref","DOI":"10.1109\/TMM.2025.3604914","article-title":"Deep Frequency-Separable Temporal Network for Efficient Video Denoising","volume":"27","author":"Tao","year":"2025","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.eswa.2026.133457_bib0065","article-title":"Egoblind: Towards egocentric visual assistance for the blind","volume":"38","author":"Xiao","year":"2026","journal-title":"Advances in Neural Information Processing Systems"},{"issue":"7","key":"10.1016\/j.eswa.2026.133457_bib0066","doi-asserted-by":"crossref","first-page":"3970","DOI":"10.1007\/s11263-025-02385-8","article-title":"VideoQA in the Era of LLMs: An Empirical Study","volume":"133","author":"Xiao","year":"2025","journal-title":"International Journal of Computer Vision"}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426023663?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426023663?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,6]],"date-time":"2026-07-06T16:24:07Z","timestamp":1783355047000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426023663"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2027,1]]},"references-count":66,"alternative-id":["S0957417426023663"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.133457","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2027,1]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"MMVUF: Improving video-based human value understanding in multimodal large language models","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.133457","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"133457"}}