{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,19]],"date-time":"2025-09-19T18:06:50Z","timestamp":1758305210604,"version":"3.44.0"},"reference-count":52,"publisher":"Springer Science and Business Media LLC","issue":"10","license":[{"start":{"date-parts":[[2025,5,2]],"date-time":"2025-05-02T00:00:00Z","timestamp":1746144000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,5,2]],"date-time":"2025-05-02T00:00:00Z","timestamp":1746144000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2025,7]]},"DOI":"10.1007\/s10489-025-06547-6","type":"journal-article","created":{"date-parts":[[2025,5,2]],"date-time":"2025-05-02T08:59:33Z","timestamp":1746176373000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["VFE: A large-scale video future event description dataset for evaluating video temporal prediction"],"prefix":"10.1007","volume":"55","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9181-9334","authenticated-orcid":false,"given":"Chenghang","family":"Lai","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Haibo","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,5,2]]},"reference":[{"key":"6547_CR1","unstructured":"Fu T-J, Li L, Gan Z, Lin K, Wang WY, Wang L, Liu Z (2021) Violet: End-to-end video-language transformers with masked visual-token modeling. arXiv:2111.12681"},{"key":"6547_CR2","doi-asserted-by":"crossref","unstructured":"Fu T-J, Li L, Gan Z, Lin K, Wang WY, Wang L, Liu Z (2023) An empirical study of end-to-end video-language transformers with masked visual modeling. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 22898\u201322909","DOI":"10.1109\/CVPR52729.2023.02193"},{"key":"6547_CR3","doi-asserted-by":"publisher","unstructured":"Lei J, Yu L, Bansal M, Berg T (2018) Tvqa: Localized, compositional video question answering. In: Proceedings of the 2018 conference on empirical methods in natural language processing. https:\/\/doi.org\/10.18653\/v1\/d18-1167","DOI":"10.18653\/v1\/d18-1167"},{"key":"6547_CR4","doi-asserted-by":"crossref","unstructured":"Xu J, Mei T, Yao T, Rui Y (2016) Msr-vtt: A large video description dataset for bridging video and language. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 5288\u20135296","DOI":"10.1109\/CVPR.2016.571"},{"issue":"13","key":"6547_CR5","doi-asserted-by":"publisher","first-page":"7510","DOI":"10.1073\/pnas.1917777117","volume":"117","author":"T Blom","year":"2020","unstructured":"Blom T, Feuerriegel D, Johnson P, Bode S, Hogendoorn H (2020) Predictions drive neural representations of visual events ahead of incoming sensory information. Proceed National Academy Sci 117(13):7510\u20137515","journal-title":"Proceed National Academy Sci"},{"issue":"1","key":"6547_CR6","doi-asserted-by":"publisher","first-page":"15276","DOI":"10.1038\/ncomms15276","volume":"8","author":"M Ekman","year":"2017","unstructured":"Ekman M, Kok P, Lange FP (2017) Time-compressed preplay of anticipated events in human primary visual cortex. Nature Commun 8(1):15276","journal-title":"Nature Commun"},{"issue":"1","key":"6547_CR7","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1162\/NECO_a_00912","volume":"29","author":"K Friston","year":"2017","unstructured":"Friston K, FitzGerald T, Rigoli F, Schwartenbeck P, Pezzulo G (2017) Active inference: a process theory. Neural Comput 29(1):1\u201349","journal-title":"Neural Comput"},{"key":"6547_CR8","doi-asserted-by":"crossref","unstructured":"Zhang C, Gao F, Jia B, Zhu Y, Zhu S-C (2019) Raven: A dataset for relational and analogical visual reasoning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 5317\u20135327","DOI":"10.1109\/CVPR.2019.00546"},{"key":"6547_CR9","doi-asserted-by":"crossref","unstructured":"Caba\u00a0Heilbron F, Escorcia V, Ghanem B, Carlos\u00a0Niebles J (2015) Activitynet: A large-scale video benchmark for human activity understanding. In: Proceedings of the Ieee conference on computer vision and pattern recognition, pp 961\u2013970","DOI":"10.1109\/CVPR.2015.7298698"},{"key":"6547_CR10","doi-asserted-by":"crossref","unstructured":"Wang X, Wu J, Chen J, Li L, Wang Y-F, Wang WY (2019) Vatex: A large-scale, high-quality multilingual dataset for video-and-language research. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 4581\u20134591","DOI":"10.1109\/ICCV.2019.00468"},{"key":"6547_CR11","doi-asserted-by":"crossref","unstructured":"Liu J, Chen W, Cheng Y, Gan Z, Yu L, Yang Y, Liu J (2020) Violin: A large-scale dataset for video-and-language inference. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 10900\u201310910","DOI":"10.1109\/CVPR42600.2020.01091"},{"key":"6547_CR12","doi-asserted-by":"crossref","unstructured":"Zellers R, Bisk Y, Farhadi A, Choi Y (2019) From recognition to cognition: Visual commonsense reasoning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 6720\u20136731","DOI":"10.1109\/CVPR.2019.00688"},{"issue":"2","key":"6547_CR13","doi-asserted-by":"publisher","first-page":"502","DOI":"10.1109\/TPAMI.2019.2901464","volume":"42","author":"M Monfort","year":"2019","unstructured":"Monfort M, Andonian A, Zhou B, Ramakrishnan K, Bargal SA, Yan T, Brown L, Fan Q, Gutfreund D, Vondrick C et al (2019) Moments in time dataset: one million videos for event understanding. IEEE Trans Pattern Anal Mach Intell 42(2):502\u2013508","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"6547_CR14","doi-asserted-by":"crossref","unstructured":"Lei J, Yu L, Berg TL, Bansal M (2020) What is more likely to happen next? video-and-language future event prediction. arXiv:2010.07999","DOI":"10.18653\/v1\/2020.emnlp-main.706"},{"key":"6547_CR15","doi-asserted-by":"crossref","unstructured":"Zellers R, Bisk Y, Schwartz R, Choi Y (2018) Swag: A large-scale adversarial dataset for grounded commonsense inference. arXiv:1808.05326","DOI":"10.18653\/v1\/D18-1009"},{"key":"6547_CR16","doi-asserted-by":"crossref","unstructured":"Miech A, Zhukov D, Alayrac J-B, Tapaswi M, Laptev I, Sivic J (2019) Howto100m: Learning a text-video embedding by watching hundred million narrated video clips. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 2630\u20132640","DOI":"10.1109\/ICCV.2019.00272"},{"key":"6547_CR17","first-page":"23634","volume":"34","author":"R Zellers","year":"2021","unstructured":"Zellers R, Lu X, Hessel J, Yu Y, Park JS, Cao J, Farhadi A, Choi Y (2021) Merlot: Multimodal neural script knowledge models. Adv Neural Inf Process Syst 34:23634\u201323651","journal-title":"Adv Neural Inf Process Syst"},{"key":"6547_CR18","doi-asserted-by":"crossref","unstructured":"Zellers R, Lu J, Lu X, Yu Y, Zhao Y, Salehi M, Kusupati A, Hessel J, Farhadi A, Choi Y (2022) Merlot reserve: Neural script knowledge through vision and language and sound. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 16375\u201316387","DOI":"10.1109\/CVPR52688.2022.01589"},{"key":"6547_CR19","doi-asserted-by":"crossref","unstructured":"Xue H, Hang T, Zeng Y, Sun Y, Liu B, Yang H, Fu J, Guo B (2022) Advancing high-resolution video-language representation with large-scale video transcriptions. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 5036\u20135045","DOI":"10.1109\/CVPR52688.2022.00498"},{"key":"6547_CR20","doi-asserted-by":"crossref","unstructured":"Bain M, Nagrani A, Varol G, Zisserman A (2021) Frozen in time: A joint video and image encoder for end-to-end retrieval. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 1728\u20131738","DOI":"10.1109\/ICCV48922.2021.00175"},{"key":"6547_CR21","doi-asserted-by":"crossref","unstructured":"Li K, Wang Y, Li Y, Wang Y, He Y, Wang L, Qiao Y (2023) Unmasked teacher: Towards training-efficient video foundation models. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 19948\u201319960","DOI":"10.1109\/ICCV51070.2023.01826"},{"key":"6547_CR22","doi-asserted-by":"crossref","unstructured":"Nagrani A, Seo PH, Seybold B, Hauth A, Manen S, Sun C, Schmid C (2022) Learning audio-video modalities from image captions. In: European Conference on Computer Vision, Springer, pp 407\u2013426","DOI":"10.1007\/978-3-031-19781-9_24"},{"key":"6547_CR23","doi-asserted-by":"crossref","unstructured":"Gu C, Zhang C, Kuriyama S (2024) Orientation-aware leg movement learning for action-driven human motion prediction. Pattern Recognition 110317","DOI":"10.1016\/j.patcog.2024.110317"},{"key":"6547_CR24","doi-asserted-by":"crossref","unstructured":"Thakur S, Beyan C, Morerio P, Murino V, Del\u00a0Bue A (2024) Leveraging next-active objects for context-aware anticipation in egocentric videos. In: Proceedings of the IEEE\/CVF winter conference on applications of computer vision, pp 8657\u20138666","DOI":"10.1109\/WACV57701.2024.00846"},{"key":"6547_CR25","doi-asserted-by":"crossref","unstructured":"Liu D, Li Q, Dinh A-D, Jiang T, Shah M, Xu C (2023) Diffusion action segmentation. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 10139\u201310149","DOI":"10.1109\/ICCV51070.2023.00930"},{"key":"6547_CR26","doi-asserted-by":"crossref","unstructured":"Xing Z, Dai Q, Hu H, Chen J, Wu Z, Jiang Y-G (2023) Svformer: Semi-supervised video transformer for action recognition. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 18816\u201318826","DOI":"10.1109\/CVPR52729.2023.01804"},{"key":"6547_CR27","doi-asserted-by":"crossref","unstructured":"Guo W, Du Y, Shen X, Lepetit V, Alameda-Pineda X, Moreno-Noguer F (2023) Back to mlp: A simple baseline for human motion prediction. In: Proceedings of the IEEE\/CVF winter conference on applications of computer vision, pp 4809\u20134819","DOI":"10.1109\/WACV56688.2023.00479"},{"key":"6547_CR28","doi-asserted-by":"crossref","unstructured":"Barquero G, Escalera S, Palmero C (2023) Belfusion: Latent diffusion for behavior-driven human motion prediction. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 2317\u20132327","DOI":"10.1109\/ICCV51070.2023.00220"},{"key":"6547_CR29","unstructured":"Lin J, Zeng A, Lu S, Cai Y, Zhang R, Wang H, Zhang L (2024) Motion-x: A large-scale 3d expressive whole-body human motion dataset. Adv Neural Inf Process Syst 36"},{"key":"6547_CR30","doi-asserted-by":"crossref","unstructured":"Chen L-H, Zhang J, Li Y, Pang Y, Xia X, Liu T (2023) Humanmac: Masked motion completion for human motion prediction. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 9544\u20139555","DOI":"10.1109\/ICCV51070.2023.00875"},{"key":"6547_CR31","doi-asserted-by":"crossref","unstructured":"Ye X, Bilodeau G-A (2023) A unified model for continuous conditional video prediction. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 3603\u20133612","DOI":"10.1109\/CVPRW59228.2023.00368"},{"key":"6547_CR32","doi-asserted-by":"crossref","unstructured":"Straka Z, Svoboda T, Hoffmann M (2023) Precnet: Next-frame video prediction based on predictive coding. IEEE Trans Neural Netw Learn Syst","DOI":"10.1109\/TNNLS.2023.3240857"},{"key":"6547_CR33","doi-asserted-by":"publisher","first-page":"7748","DOI":"10.1609\/aaai.v38i7.28609","volume":"38","author":"J Zhu","year":"2024","unstructured":"Zhu J, Wan Z, Dai Y (2024) Video frame prediction from a single image and events. Proceedings of the AAAI conference on artificial intelligence 38:7748\u20137756","journal-title":"Proceedings of the AAAI conference on artificial intelligence"},{"key":"6547_CR34","unstructured":"Fiquet P-\u00c9, Simoncelli E (2024) A polar prediction model for learning to represent visual transformations. Adv Neural Inf Process Syst 36"},{"key":"6547_CR35","doi-asserted-by":"crossref","unstructured":"Zhou L, Xu C, Corso J (2018) Towards automatic learning of procedures from web instructional videos. In: Proceedings of the AAAI conference on artificial intelligence, vol. 32","DOI":"10.1609\/aaai.v32i1.12342"},{"key":"6547_CR36","unstructured":"OpenAI (2023) GPT-4 Technical Report"},{"key":"6547_CR37","doi-asserted-by":"crossref","unstructured":"Fang Y, Wang W, Xie B, Sun Q, Wu L, Wang X, Huang T, Wang X, Cao Y (2023) Eva: Exploring the limits of masked visual representation learning at scale. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 19358\u201319369","DOI":"10.1109\/CVPR52729.2023.01855"},{"key":"6547_CR38","unstructured":"Chiang W-L, Li Z, Lin Z, Sheng Y, Wu Z, Zhang H, Zheng L, Zhuang S, Zhuang Y, Gonzalez JE et al (2023) Vicuna: An open-source chatbot impressing gpt-4 with 90%* chatgpt quality. See https:\/\/vicuna.lmsys.org (accessed 14 April 2023)"},{"key":"6547_CR39","unstructured":"Vaswani A (2017) Attention is all you need. Adv Neural Inf Process Syst"},{"key":"6547_CR40","unstructured":"Zhu D, Chen J, Shen X, Li X, Elhoseiny M (2023) Minigpt-4: Enhancing vision-language understanding with advanced large language models. arXiv:2304.10592"},{"key":"6547_CR41","doi-asserted-by":"crossref","unstructured":"Zhang H, Li X, Bing L (2023) Video-llama: An instruction-tuned audio-visual language model for video understanding. arXiv:2306.02858","DOI":"10.18653\/v1\/2023.emnlp-demo.49"},{"key":"6547_CR42","unstructured":"Li K, He Y, Wang Y, Li Y, Wang W, Luo P, Wang Y, Wang L, Qiao Y (2023) Videochat: Chat-centric video understanding.arXiv:2305.06355"},{"key":"6547_CR43","unstructured":"Touvron H, Lavril T, Izacard G, Martinet X, Lachaux M-A, Lacroix T, Rozi\u00e8re B, Goyal N, Hambro E, Azhar F, et al (2023) Llama: Open and efficient foundation language models. arXiv:2302.13971"},{"key":"6547_CR44","unstructured":"Zhang R, Han J, Liu C, Gao P, Zhou A, Hu X, Yan S, Lu P, Li H, Qiao Y (2023) Llama-adapter: Efficient fine-tuning of language models with zero-init attention. arXiv:2303.16199"},{"key":"6547_CR45","doi-asserted-by":"crossref","unstructured":"Maaz M, Rasheed H, Khan S, Khan FS (2023) Video-chatgpt: Towards detailed video understanding via large vision and language models. arXiv:2306.05424","DOI":"10.18653\/v1\/2024.acl-long.679"},{"key":"6547_CR46","unstructured":"Luo R, Zhao Z, Yang M, Dong J, Qiu M, Lu P, Wang T, Wei Z (2023) Valley: Video assistant with large language model enhanced ability. arXiv:2306.07207"},{"key":"6547_CR47","doi-asserted-by":"crossref","unstructured":"Song E, Chai W, Wang G, Zhang Y, Zhou H, Wu F, Guo X, Ye T, Lu Y, Hwang J-N et al (2023) Moviechat: From dense token to sparse memory for long video understanding. arXiv:2307.16449","DOI":"10.1109\/CVPR52733.2024.01725"},{"key":"6547_CR48","doi-asserted-by":"crossref","unstructured":"Papineni K, Roukos S, Ward T, Zhu W-J (2002) Bleu: a method for automatic evaluation of machine translation. In: Proceedings of the 40th annual meeting of the association for computational linguistics, pp 311\u2013318","DOI":"10.3115\/1073083.1073135"},{"key":"6547_CR49","doi-asserted-by":"crossref","unstructured":"Denkowski M, Lavie A (2014) Meteor universal: Language specific translation evaluation for any target language. In: Proceedings of the ninth workshop on statistical machine translation, pp 376\u2013380","DOI":"10.3115\/v1\/W14-3348"},{"key":"6547_CR50","doi-asserted-by":"crossref","unstructured":"Lin C-Y, Och FJ (2004) Automatic evaluation of machine translation quality using longest common subsequence and skip-bigram statistics. In: Proceedings of the 42nd annual meeting of the association for computational linguistics (ACL-04), pp 605\u2013612","DOI":"10.3115\/1218955.1219032"},{"key":"6547_CR51","doi-asserted-by":"crossref","unstructured":"Vedantam R, Lawrence\u00a0Zitnick C, Parikh D (2015) Cider: Consensus-based image description evaluation. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 4566\u20134575","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"6547_CR52","unstructured":"Loshchilov I, Hutter F (2017) Decoupled weight decay regularization. arXiv:1711.05101"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-025-06547-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10489-025-06547-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-025-06547-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,19]],"date-time":"2025-09-19T13:58:41Z","timestamp":1758290321000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10489-025-06547-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,5,2]]},"references-count":52,"journal-issue":{"issue":"10","published-print":{"date-parts":[[2025,7]]}},"alternative-id":["6547"],"URL":"https:\/\/doi.org\/10.1007\/s10489-025-06547-6","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"type":"print","value":"0924-669X"},{"type":"electronic","value":"1573-7497"}],"subject":[],"published":{"date-parts":[[2025,5,2]]},"assertion":[{"value":"6 April 2025","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 May 2025","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing Interests"}},{"value":"This paper is neither the entire paper nor any parts of its content has been published or is under consideration for publication elsewhere.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical and informed consent for data used"}}],"article-number":"713"}}