{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T17:03:35Z","timestamp":1784567015547,"version":"3.55.0"},"reference-count":112,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,6,1]],"date-time":"2026-06-01T00:00:00Z","timestamp":1780272000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62372339"],"award-info":[{"award-number":["62372339"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62371350"],"award-info":[{"award-number":["62371350"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62372336"],"award-info":[{"award-number":["62372336"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Ministry of Education Industry-University Cooperative Education Project","award":["240700006245501"],"award-info":[{"award-number":["240700006245501"]}]},{"name":"Key Science and Technology Research Project of Xinjiang Production and Construction Corps","award":["2025AB029"],"award-info":[{"award-number":["2025AB029"]}]},{"name":"Hubei Provincial Science and Technology Plan Project","award":["2025BAB020"],"award-info":[{"award-number":["2025BAB020"]}]},{"name":"Hubei Provincial Science and Technology Plan Project","award":["2025CSA057"],"award-info":[{"award-number":["2025CSA057"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2026,6]]},"DOI":"10.1007\/s11263-026-02898-w","type":"journal-article","created":{"date-parts":[[2026,6,4]],"date-time":"2026-06-04T09:32:04Z","timestamp":1780565524000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["StoryVideoQA: Scaling Deep Video Understanding with a Large-Scale, Multi-Genre and Auto-Generated Dataset"],"prefix":"10.1007","volume":"134","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-4258-9581","authenticated-orcid":false,"given":"Zhengqian","family":"Wu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-9209-3876","authenticated-orcid":false,"given":"Zhixian","family":"Liu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-0802-1599","authenticated-orcid":false,"given":"Aodong","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-1494-0135","authenticated-orcid":false,"given":"Jingyang","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-4951-1805","authenticated-orcid":false,"given":"Ruizhe","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-8976-5194","authenticated-orcid":false,"given":"Hanlin","family":"Ge","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9796-488X","authenticated-orcid":false,"given":"Zhongyuan","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4526-6297","authenticated-orcid":false,"given":"Chunxia","family":"Xiao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8287-8655","authenticated-orcid":false,"given":"Chao","family":"Liang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,6,4]]},"reference":[{"key":"2898_CR1","doi-asserted-by":"publisher","first-page":"23716","DOI":"10.52202\/068431-1723","volume":"35","author":"J-B Alayrac","year":"2022","unstructured":"Alayrac, J.-B., Donahue, J., Luc, P., Miech, A., Barr, I., Hasson, Y., Lenc, K., Mensch, A., Millican, K., Reynolds, M., & Ring, R. (2022). Flamingo: a visual language model for few-shot learning. Advances in neural information processing systems, 35, 23716\u201323736.","journal-title":"Advances in neural information processing systems"},{"key":"2898_CR2","doi-asserted-by":"crossref","unstructured":"Argaw, D. M., Yoon, S., Heilbron, F. C., Deilamsalehy, H., Bui, T., Wang, Z., Dernoncourt, F., & Chung, J. S. (2024). Scaling up video summarization pretraining with large language models. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (pp. 8332\u20138341).","DOI":"10.1109\/CVPR52733.2024.00796"},{"key":"2898_CR3","unstructured":"Bai, S., Cai, Y., Chen, R., Chen, K., Chen, X., Cheng, Z., Deng, L., Ding, W., Gao, C., Ge, C., Ge, W., Guo, Z., Huang, Q., Huang, J., Huang, F., Hui, B., Jiang, S., Li, Z., Li, M., Li, M., Li, K., Lin, Z., Lin, J., Liu, X., Liu, J., Liu, C., Liu, Y., Liu, D., Liu, S., Lu, D., Luo, R., Lv, C., Men, R., Meng, L., Ren, X., Ren, X., Song, S., Sun, Y., Tang, J., Tu, J., Wan, J., Wang, P., Wang, P., Wang, Q., Wang, Y., Xie, T., Xu, Y., Xu, H., Xu, J., Yang, Z., Yang, M., Yang, J., Yang, A., Yu, B., Zhang, F., Zhang, H., Zhang, X., Zheng, B., Zhong, H., Zhou, J., Zhou, F., Zhou, J., Zhu, Y., & Zhu, K. (2025). Qwen3-VL Technical Report."},{"key":"2898_CR4","unstructured":"Bishop, C. M. Pattern Recognition and Machine Learning (Vol. 4). Springer."},{"key":"2898_CR5","unstructured":"Chen, G., Liu, Y., Huang, Y., Pei, B., Xu, J., He, Y., Lu, T., Wang, Y., & Wang, L. Cg-bench: Clue-grounded question answering benchmark for long video understanding. In: The Thirteenth International Conference on Learning Representations."},{"issue":"2","key":"2898_CR6","doi-asserted-by":"publisher","first-page":"2159","DOI":"10.1609\/aaai.v39i2.32214","volume":"39","author":"Q Chen","year":"2025","unstructured":"Chen, Q., Di, S., & Xie, W. (2025). Grounded multi-hop videoqa in long-form egocentric videos. Proceedings of the AAAI Conference on Artificial Intelligence, 39(2), 2159\u20132167.","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"2898_CR7","unstructured":"Cheng, C., Guan, J., Wu, W., & Yan, R. (2025). Scaling video-language models to 10k frames via hierarchical differential distillation arXiv:2504.02438"},{"key":"2898_CR8","unstructured":"Cheng, Z., Leng, S., Zhang, H., Xin, Y., Li, X., Chen, G., Zhu, Y., Zhang, W., Luo, Z., Zhao, D., & Bing, L. (2024). Videollama 2: Advancing spatial-temporal modeling and audio understanding in video-llms. arXiv:2406.07476"},{"key":"2898_CR9","unstructured":"Chiang, W.-L., Li, Z., Lin, Z., Sheng, Y., Wu, Z., Zhang, H., Zheng, L., Zhuang, S., Zhuang, Y., Gonzalez, J.E., & Stoica, I. (2023). Vicuna: An open-source chatbot impressing gpt-4 with 90%* chatgpt quality. See URL: https:\/\/vicuna lmsys. org (accessed 14 April 2023)"},{"key":"2898_CR10","doi-asserted-by":"crossref","unstructured":"Choi, J., Lee, S., Chu, J., Choi, M., & Kim, H.J. (2024). vid-tldr: Training free token merging for light-weight video transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), (pp. 18771\u201318781).","DOI":"10.1109\/CVPR52733.2024.01776"},{"key":"2898_CR11","doi-asserted-by":"crossref","unstructured":"Choi, S., On, K.-W., Heo, Y.-J., Seo, A., Jang, Y., Lee, M., & Zhang, B.-T. (2021). Dramaqa: Character-centered video story understanding with hierarchical qa. Proceedings of the AAAI Conference on Artificial Intelligence (Vol. 35, pp. 1166\u20131174).","DOI":"10.1609\/aaai.v35i2.16203"},{"issue":"70","key":"2898_CR12","first-page":"1","volume":"25","author":"HW Chung","year":"2024","unstructured":"Chung, H. W., Hou, L., Longpre, S., Zoph, B., Tay, Y., Fedus, W., Li, Y., Wang, X., Dehghani, M., Brahma, S., Webson, A., Gu, S. S., Dai, Z., Suzgun, M., Chen, X., Chowdhery, A., Castro-Ros, A., Pellat, M., Robinson, K., \u2026 Wei, J. (2024). Scaling instruction-finetuned language models. Journal of Machine Learning Research, 25(70), 1\u201353.","journal-title":"Journal of Machine Learning Research"},{"key":"2898_CR13","doi-asserted-by":"crossref","unstructured":"Curtis, K., Awad, G., Godil, A., & Soboroff, I. (2023). The acm multimedia 2023 deep video understanding grand challenge. Proceedings of the 31st ACM International Conference on Multimedia. MM \u201923 (pp. 9606\u20139609). New York, NY, USA: Association for Computing Machinery.","DOI":"10.1145\/3581783.3612829"},{"key":"2898_CR14","doi-asserted-by":"crossref","unstructured":"Curtis, K., Awad, G., Rajput, S., & Soboroff, I. (2020). Hlvu: A new challenge to test deep understanding of movies the way humans do. Proceedings of the 2020 International Conference on Multimedia Retrieval. ICMR \u201920 (pp. 355\u2013361). New York, NY, USA: Association for Computing Machinery.","DOI":"10.1145\/3372278.3390742"},{"key":"2898_CR15","doi-asserted-by":"crossref","unstructured":"Curtis, K., Awad, G., Rajput, S., & Soboroff, I. (2022). The acm multimedia 2022 deep video understanding grand challenge. Proceedings of the 30th ACM International Conference on Multimedia. MM \u201922 (pp. 7075\u20137078). New York, NY, USA: Association for Computing Machinery.","DOI":"10.1145\/3503161.3551582"},{"key":"2898_CR16","doi-asserted-by":"crossref","unstructured":"Fu, C., Dai, Y., Luo, Y., Li, L., Ren, S., Zhang, R., Wang, Z., Zhou, C., Shen, Y., Zhang, M., Chen, P., Li, Y., Lin, S., Zhao, S., Li, K., Xu, T., Zheng, X., Chen, E., Shan, C., He, R., & Sun, X. (2025). Video-mme: The first-ever comprehensive evaluation benchmark of multi-modal llms in video analysis. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), (pp. 24108\u201324118).","DOI":"10.1109\/CVPR52734.2025.02245"},{"key":"2898_CR17","doi-asserted-by":"crossref","unstructured":"Fu, T.-J., Li, L., Gan, Z., Lin, K., Wang, W.Y., Wang, L., & Liu, Z. (2023). An empirical study of end-to-end video-language transformers with masked visual modeling. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), (pp. 22898\u201322909).","DOI":"10.1109\/CVPR52729.2023.02193"},{"key":"2898_CR18","doi-asserted-by":"crossref","unstructured":"Fung, Y., Wang, H., Wang, T., Kebarighotbi, A., Bansal, M., Ji, H., & Natarajan, P. (2023). Deepmaven: Deep question answering on long-distance movie\/tv show videos with multimedia knowledge extraction and synthesis. In: Proceedings of the 17th Conference of the European Chapter of the Association for Computational Linguistics, (pp. 3041\u20133051).","DOI":"10.18653\/v1\/2023.eacl-main.221"},{"key":"2898_CR19","doi-asserted-by":"crossref","unstructured":"Galanopoulos, D., Goulas, A., Leventakis, A., Patras, I., & Mezaris, V. (2025). An llm framework for long-form video retrieval and audio-visual question answering using qwen2\/2.5. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) Workshops (pp. 3739\u20133748).","DOI":"10.1109\/CVPRW67362.2025.00358"},{"key":"2898_CR20","unstructured":"Grauman, K., Westbury, A., Byrne, E., Chavis, Z., Furnari, A., Girdhar, R., Hamburger, J., Jiang, H., Liu, M., Liu, X., & Martin, M. (2022). Ego4d: Around the world in 3,000 hours of egocentric video. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, (pp. 18995\u201319012)."},{"key":"2898_CR21","doi-asserted-by":"crossref","unstructured":"Guo, J., Liang, C., & Wang, Z. (2023). Who, what and where: Composite-semantic instance search for story videos. 2023 IEEE International Conference on Multimedia and Expo (ICME) (pp. 858\u2013863). IEEE.","DOI":"10.1109\/ICME55011.2023.00152"},{"key":"2898_CR22","doi-asserted-by":"publisher","first-page":"1412","DOI":"10.1109\/TIP.2025.3542272","volume":"34","author":"J Guo","year":"2025","unstructured":"Guo, J., Lu, A., Wu, Z., Wang, Z., & Liang, C. (2025). Who, what, and where: Composite-semantics instance search for story videos. IEEE Transactions on Image Processing, 34, 1412\u20131426.","journal-title":"IEEE Transactions on Image Processing"},{"key":"2898_CR23","doi-asserted-by":"crossref","unstructured":"Gu, G., Wu, Z., He, J., Song, L., Wang, Z., & Liang, C. (2024). Talksee: Interactive video retrieval engine using large language model. MultiMedia Modeling (pp. 387\u2013393). Cham: Springer.","DOI":"10.1007\/978-3-031-53302-0_36"},{"key":"2898_CR24","doi-asserted-by":"crossref","unstructured":"Han, T., Bain, M., Nagrani, A., Varol, G., Xie, W., & Zisserman, A. (2023). Autoad ii: The sequel - who, when, and what in movie audio description. Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV) (pp. 13645\u201313655).","DOI":"10.1109\/ICCV51070.2023.01255"},{"key":"2898_CR25","doi-asserted-by":"crossref","unstructured":"Han, T., Bain, M., Nagrani, A., Varol, G., Xie, W., & Zisserman, A. (2023). Autoad: Movie description in context. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (pp. 18930\u201318940).","DOI":"10.1109\/CVPR52729.2023.01815"},{"key":"2898_CR26","doi-asserted-by":"crossref","unstructured":"Han, T., Bain, M., Nagrani, A., Varol, G., Xie, W., & Zisserman, A. (2024). Autoad iii: The prequel - back to the pixels. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (pp. 18164\u201318174).","DOI":"10.1109\/CVPR52733.2024.01720"},{"key":"2898_CR27","unstructured":"He, X., & Zhu, W. (2020). Visual question answering from theory to application. Image."},{"key":"2898_CR28","doi-asserted-by":"crossref","unstructured":"He, B., Li, H., Jang, Y.K., Jia, M., Cao, X., Shah, A., Shrivastava, A., & Lim, S.-N. (2024). Ma-lmm: Memory-augmented large multimodal model for long-term video understanding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, (pp. 13504\u201313514).","DOI":"10.1109\/CVPR52733.2024.01282"},{"key":"2898_CR29","doi-asserted-by":"crossref","unstructured":"He, B., Wang, J., Qiu, J., Bui, T., Shrivastava, A., & Wang, Z. (2023). Align and attend: Multimodal summarization with dual contrastive losses. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, (pp. 14867\u201314878).","DOI":"10.1109\/CVPR52729.2023.01428"},{"key":"2898_CR30","doi-asserted-by":"crossref","unstructured":"Hu, K., Wu, P., Pu, F., Xiao, W., Zhang, Y., Yue, X., Li, B., & Liu, Z. (2025). Video-mmmu: Evaluating knowledge acquisition from multi-discipline professional videos. arXiv:2501.13826","DOI":"10.18653\/v1\/2026.acl-long.1281"},{"key":"2898_CR31","doi-asserted-by":"crossref","unstructured":"Huang, D.-A., Liao, S., Radhakrishnan, S., Yin, H., Molchanov, P., Yu, Z., & Kautz, J. (2025). Lita: Language instructed temporal-localization assistant. In A. Leonardis, E. Ricci, S. Roth, O. Russakovsky, T. Sattler, & G. Varol (Eds.), Computer Vision - ECCV 2024 (pp. 202\u2013218). Cham: Springer.","DOI":"10.1007\/978-3-031-73039-9_12"},{"key":"2898_CR32","doi-asserted-by":"crossref","unstructured":"Jin, P., Takanobu, R., Zhang, W., Cao, X., & Yuan, L. (2024). Chat-univi: Unified visual representation empowers large language models with image and video understanding. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (pp. 13700\u201313710).","DOI":"10.1109\/CVPR52733.2024.01300"},{"key":"2898_CR33","doi-asserted-by":"crossref","unstructured":"Kukleva, A., Tapaswi, M., & Laptev, I. (2020). Learning interactions and relationships between movie characters. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (pp. 9849\u20139858).","DOI":"10.1109\/CVPR42600.2020.00987"},{"key":"2898_CR34","doi-asserted-by":"crossref","unstructured":"Lee, M. J., Gong, D., & Cho, M. (2025). Video summarization with large language models. Proceedings of the Computer Vision and Pattern Recognition Conference (pp. 18981\u201318991).","DOI":"10.1109\/CVPR52734.2025.01768"},{"key":"2898_CR35","doi-asserted-by":"crossref","unstructured":"Lei, J., Berg, T., & Bansal, M. (2023). Revealing single frame bias for video-and-language learning. In: Rogers, A., Boyd-Graber, J., & Okazaki, N. (eds.) Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), (pp. 487\u2013507). Association for Computational Linguistics, Toronto, Canada.","DOI":"10.18653\/v1\/2023.acl-long.29"},{"key":"2898_CR36","doi-asserted-by":"crossref","unstructured":"Lei, J., Yu, L., Bansal, M., & Berg, T. (2018). TVQA: Localized, compositional video question answering. In: Riloff, E., Chiang, D., Hockenmaier, J., & Tsujii, J. (eds.) Proceedings of the 2018 Conference on Empirical Methods in Natural Language Processing, (pp. 1369\u20131379.) Association for Computational Linguistics, Brussels, Belgium","DOI":"10.18653\/v1\/D18-1167"},{"key":"2898_CR37","doi-asserted-by":"crossref","unstructured":"Lei, J., Yu, L., Berg, T., & Bansal, M. (2020). TVQA+: Spatio-temporal grounding for video question answering. In: Jurafsky, D., Chai, J., Schluter, N., & Tetreault, J. (eds.) Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics, (pp. 8211\u20138225). Association for Computational Linguistics, Online.","DOI":"10.18653\/v1\/2020.acl-main.730"},{"key":"2898_CR38","doi-asserted-by":"crossref","unstructured":"Li, L., Chen, Y.-C., Cheng, Y., Gan, Z., Yu, L., & Liu, J. (2020). HERO: Hierarchical encoder for Video+Language omni-representation pre-training. In: Webber, B., Cohn, T., He, Y., & Liu, Y. (eds.) Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP), (pp. 2046\u20132065). Association for Computational Linguistics, Online.","DOI":"10.18653\/v1\/2020.emnlp-main.161"},{"key":"2898_CR39","unstructured":"Li, Y., Chen, X., Hu, B., Wang, L., Shi, H., & Zhang, M. (2024). Videovista: A versatile benchmark for video understanding and reasoning. arXiv:2406.11303"},{"key":"2898_CR40","doi-asserted-by":"crossref","unstructured":"Li, R., Guo, J., Li, M., Wu, Z., & Liang, C. (2023). A hierarchical deep video understanding method with shot-based instance search and large language model. In: Proceedings of the 31st ACM International Conference on Multimedia, (pp. 9425\u20139429).","DOI":"10.1145\/3581783.3612838"},{"key":"2898_CR41","doi-asserted-by":"crossref","unstructured":"Li, K., He, Y., Wang, Y., Li, Y., Wang, W., Luo, P., Wang, Y., Wang, L., & Qiao, Y. (2024). VideoChat: Chat-Centric Video Understanding.","DOI":"10.1007\/s11432-024-4321-9"},{"key":"2898_CR42","doi-asserted-by":"crossref","unstructured":"Li, D., Li, J., Li, H., Niebles, J.C., & Hoi, S.C. (2022). Align and prompt: Video-and-language pre-training with entity prompts. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, (pp. 4953\u20134963).","DOI":"10.1109\/CVPR52688.2022.00490"},{"key":"2898_CR43","unstructured":"Li, J., Li, D., Savarese, S., & Hoi, S. (2023). BLIP-2: Bootstrapping language-image pre-training with frozen image encoders and large language models. In: Krause, A., Brunskill, E., Cho, K., Engelhardt, B., Sabato, S., & Scarlett, J. (eds.) Proceedings of the 40th International Conference on Machine Learning. Proceedings of Machine Learning Research, (vol. 202, pp. 19730\u201319742)."},{"key":"2898_CR44","doi-asserted-by":"crossref","unstructured":"Li, K., Wang, Y., He, Y., Li, Y., Wang, Y., Liu, Y., Wang, Z., Xu, J., Chen, G., Luo, P., Wang, L., & Qiao, Y. (2024). Mvbench: A comprehensive multi-modal video understanding benchmark. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), (pp. 22195\u201322206).","DOI":"10.1109\/CVPR52733.2024.02095"},{"key":"2898_CR45","unstructured":"Li, X., Wang, Y., Yu, J., Zeng, X., Zhu, Y., Huang, H., Gao, J., Li, K., He, Y., Wang, C., & Qiao, Y. (2024). Videochat-flash: Hierarchical compression for long-context video modeling. arXiv:2501.00574"},{"key":"2898_CR46","doi-asserted-by":"crossref","unstructured":"Liang, C., Xu, C., Cheng, J., & Lu, H. (2011). Tvparser: An automatic tv video parsing method. In CVPR 2011 (pp. 3377\u20133384).","DOI":"10.1109\/CVPR.2011.5995681"},{"key":"2898_CR47","doi-asserted-by":"crossref","unstructured":"Liang, C., Xu, C., Cheng, J., Min, W., & Lu, H. (2012). Script-to-movie: a computational framework for story movie composition. IEEE transactions on multimedia, 15(2), 401\u2013414.","DOI":"10.1109\/TMM.2012.2229972"},{"key":"2898_CR48","unstructured":"Lin, K., Ahmed, F., Li, L., Lin, C.-C., Azarnasab, E., Yang, Z., Wang, J., Liang, L., Liu, Z., Lu, Y., & Liu, C. (2023). Mm-vid: Advancing video understanding with gpt-4v (ision). arXiv:2310.19773"},{"key":"2898_CR49","doi-asserted-by":"crossref","unstructured":"Lin, B., Ye, Y., Zhu, B., Cui, J., Ning, M., Jin, P., & Yuan, L. (2024). Video-LLaVA: Learning united visual representation by alignment before projection. In Y. Al-Onaizan, M. Bansal, & Y.-N. Chen (Eds.), Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing (pp. 5971\u20135984). Miami, Florida, USA: Association for Computational Linguistics.","DOI":"10.18653\/v1\/2024.emnlp-main.342"},{"key":"2898_CR50","doi-asserted-by":"crossref","unstructured":"Liu, H., Jiang, S., Duan, F., Lyu, Y., Wang, X., Ge, H., & Liang, C. (2025). Cadencerag: Context-aware and dependency-enhanced retrieval augmented generation for holistic video understanding. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) Workshops (pp. 3688\u20133697).","DOI":"10.1109\/CVPRW67362.2025.00353"},{"key":"2898_CR51","doi-asserted-by":"crossref","unstructured":"Liu, Y., Li, S., Liu, Y., Wang, Y., Ren, S., Li, L., Chen, S., Sun, X., & Hou, L. (2024). Tempcompass: Do video llms really understand videos? In: Findings of the Association for Computational Linguistics ACL 2024, (pp. 8731\u20138772).","DOI":"10.18653\/v1\/2024.findings-acl.517"},{"key":"2898_CR52","doi-asserted-by":"crossref","unstructured":"Liu, H., Li, C., Wu, Q., Lee, Y. J., Oh, A., Naumann, T., Globerson, A., Saenko, K., & Hardt, M. (2023). Visual instruction tuning. In S. Levine (Ed.), Advances in Neural Information Processing Systems (Vol. 36, pp. 34892\u201334916).","DOI":"10.52202\/075280-1516"},{"key":"2898_CR53","doi-asserted-by":"crossref","unstructured":"Ma, Z., Gou, C., Shi, H., Sun, B., Li, S., Rezatofighi, H., & Cai, J. (2025). Drvideo: Document retrieval based long video understanding. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (pp. 18936\u201318946).","DOI":"10.1109\/CVPR52734.2025.01764"},{"key":"2898_CR54","doi-asserted-by":"crossref","unstructured":"Maaz, M., Rasheed, H., Khan, S., & Khan, F. (2024). Video-ChatGPT: Towards detailed video understanding via large vision and language models. In: Ku, L.-W., Martins, A., & Srikumar, V. (eds.) Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), (pp. 12585\u201312602). Association for Computational Linguistics, Bangkok, Thailand.","DOI":"10.18653\/v1\/2024.acl-long.679"},{"key":"2898_CR55","doi-asserted-by":"crossref","unstructured":"Man, Y., Huang, Y., Zhang, C., Li, B., Niu, W., & Yin, M. (2025). Adacm$$^{2}$$: On understanding extremely long-term video with adaptive cross-modality memory reduction. Proceedings of the Computer Vision and Pattern Recognition Conference (pp. 8534\u20138544).","DOI":"10.1109\/CVPR52734.2025.00798"},{"key":"2898_CR56","doi-asserted-by":"publisher","first-page":"46212","DOI":"10.52202\/075280-2004","volume":"36","author":"K Mangalam","year":"2023","unstructured":"Mangalam, K., Akshulakov, R., & Malik, J. (2023). Egoschema: A diagnostic benchmark for very long-form video language understanding. Advances in Neural Information Processing Systems, 36, 46212\u201346244.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"2898_CR57","doi-asserted-by":"crossref","unstructured":"Mu, F., Mo, S., & Li, Y. (2024). Snag: Scalable and accurate video grounding. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (pp. 18930\u201318940).","DOI":"10.1109\/CVPR52733.2024.01791"},{"key":"2898_CR58","doi-asserted-by":"publisher","first-page":"3636","DOI":"10.18653\/v1\/2024.findings-acl.217","volume-title":"Findings of the Association for Computational Linguistics: ACL 2024","author":"T Nguyen","year":"2024","unstructured":"Nguyen, T., Bin, Y., Xiao, J., Qu, L., Li, Y., Wu, J. Z., Nguyen, C.-D., Ng, S.-K., & Luu, A. T. (2024). Video-language understanding: A survey from model architecture, model training, and data perspectives. In L.-W. Ku, A. Martins, & V. Srikumar (Eds.), Findings of the Association for Computational Linguistics: ACL 2024 (pp. 3636\u20133657). Bangkok, Thailand: Association for Computational Linguistics."},{"issue":"2","key":"2898_CR59","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3617892","volume":"42","author":"Y Niu","year":"2023","unstructured":"Niu, Y., Liang, C., Lu, A., Huang, B., Wang, Z., & Guo, J. (2023). Person-action instance search in story videos: An experimental study. ACM Transactions on Information Systems, 42(2), 1\u201334.","journal-title":"ACM Transactions on Information Systems"},{"key":"2898_CR60","unstructured":"OpenAI (2023). Chatgpt: A language model for conversational ai openai. Technical report."},{"key":"2898_CR61","unstructured":"Qian, L., Li, J., Wu, Y., Ye, Y., Fei, H., Chua, T.-S., Zhuang, Y., & Tang, S. (2024). Momentor: advancing video large language model with fine-grained temporal reasoning. Proceedings of the 41st International Conference on Machine Learning (pp. 41340\u201341356)."},{"key":"2898_CR62","unstructured":"Rawal, R., Saifullah, K., Basri, R., Jacobs, D., Somepalli, G., & Goldstein, T. (2024). Cinepile: A long video question answering dataset and benchmark. arXiv:2405.08813"},{"key":"2898_CR63","unstructured":"Reid, M., Savinov, N., Teplyashin, D., Lepikhin, D., Lillicrap, T., Alayrac, J.-B., Soricut, R., Lazaridou, A., Firat, O., & Schrittwieser, J. (2024). Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context. arXiv:2403.05530"},{"key":"2898_CR64","doi-asserted-by":"crossref","unstructured":"Ren, S., Yao, L., Li, S., Sun, X., & Hou, L. (2024). Timechat: A time-sensitive multimodal large language model for long video understanding. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (pp. 14313\u201314323).","DOI":"10.1109\/CVPR52733.2024.01357"},{"key":"2898_CR65","unstructured":"R\u00e9nyi, A. (1961). On measures of entropy and information. In: Proceedings of the Fourth Berkeley Symposium on Mathematical Statistics and Probability, Volume 1: Contributions to the Theory of Statistics, (vol. 4, pp. 547\u2013562). University of California Press."},{"key":"2898_CR66","doi-asserted-by":"crossref","unstructured":"Rohrbach, A., Rohrbach, M., Tandon, N., & Schiele, B. (2015). A dataset for movie description. In: 2015 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), (pp. 3202\u20133212).","DOI":"10.1109\/CVPR.2015.7298940"},{"issue":"1","key":"2898_CR67","doi-asserted-by":"publisher","first-page":"94","DOI":"10.1007\/s11263-016-0987-1","volume":"123","author":"A Rohrbach","year":"2017","unstructured":"Rohrbach, A., Torabi, A., Rohrbach, M., Tandon, N., Pal, C., Larochelle, H., Courville, A., & Schiele, B. (2017). Movie description. Int. J. Comput. Vision, 123(1), 94\u2013120.","journal-title":"Int. J. Comput. Vision"},{"issue":"3","key":"2898_CR68","doi-asserted-by":"publisher","first-page":"379","DOI":"10.1002\/j.1538-7305.1948.tb01338.x","volume":"27","author":"CE Shannon","year":"1948","unstructured":"Shannon, C. E. (1948). A mathematical theory of communication. The Bell System Technical Journal, 27(3), 379\u2013423.","journal-title":"The Bell System Technical Journal"},{"key":"2898_CR69","unstructured":"Shen, Y., Fu, C., Dong, S., Wang, X., Zhang, Y.-F., Chen, P., Zhang, M., Cao, H., Li, K., Lin, S., & Zheng, X. (2025). Long-vita: Scaling large multi-modal models to 1 million tokens with leading short-context accuracy. arXiv e-prints, 2502."},{"key":"2898_CR70","unstructured":"Shen, X., Xiong, Y., Zhao, C., Wu, L., Chen, J., Zhu, C., Liu, Z., Xiao, F., Varadarajan, B., Bordes, F., Liu, Z., Xu, H., J.\u00a0Kim, H., Soran, B., Krishnamoorthi, R., Elhoseiny, M., & Chandra, V. (2024). Longvu: Spatiotemporal adaptive compression for long video-language understanding. arXiv:2410.17434"},{"key":"2898_CR71","doi-asserted-by":"crossref","unstructured":"Shu, Y., Liu, Z., Zhang, P., Qin, M., Zhou, J., Liang, Z., Huang, T., & Zhao, B. (2025). Video-xl: Extra-long vision language model for hour-scale video understanding. Proceedings of the Computer Vision and Pattern Recognition Conference (pp. 26160\u201326169).","DOI":"10.1109\/CVPR52734.2025.02436"},{"key":"2898_CR72","doi-asserted-by":"crossref","unstructured":"Song, E., Chai, W., Wang, G., Zhang, Y., Zhou, H., Wu, F., Chi, H., Guo, X., Ye, T., Zhang, Y., & Lu, Y. (2024). Moviechat: From dense token to sparse memory for long video understanding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, (pp. 18221\u201318232).","DOI":"10.1109\/CVPR52733.2024.01725"},{"key":"2898_CR73","doi-asserted-by":"crossref","unstructured":"Tang, Y., Bi, J., Xu, S., Song, L., Liang, S., Wang, T., Zhang, D., An, J., Lin, J., Zhu, R., & Vosoughi, A. (2025). Video understanding with large language models: A survey. IEEE Transactions on Circuits and Systems for Video Technology.","DOI":"10.1109\/TCSVT.2025.3566695"},{"key":"2898_CR74","doi-asserted-by":"crossref","unstructured":"Tapaswi, M., Zhu, Y., Stiefelhagen, R., Torralba, A., Urtasun, R., & Fidler, S. (2016). Movieqa: Understanding stories in movies through question-answering. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (pp. 4631\u20134640).","DOI":"10.1109\/CVPR.2016.501"},{"key":"2898_CR75","unstructured":"Team, G., Anil, R., Borgeaud, S., Wu, Y., Alayrac, J.-B., Yu, J., Soricut, R., Schalkwyk, J., Dai, A.M., Hauth, A., Millican, K., & Silver, D. (2023). Gemini: a family of highly capable multimodal models. arXiv:2312.11805"},{"key":"2898_CR76","unstructured":"Touvron, H., Lavril, T., Izacard, G., Martinet, X., Lachaux, M.-A., Lacroix, T., Rozi\u00e8re, B., Goyal, N., Hambro, E., Azhar, F., Rodriguez, A., & Llama. (2023). Open and efficient foundation language models. arXiv:2302.13971"},{"key":"2898_CR77","unstructured":"Touvron, H., Martin, L., Stone, K., Albert, P., Almahairi, A., Babaei, Y., Bashlykov, N., Batra, S., Bhargava, P., Bhosale, S., & Bikel, D. (2023). Llama 2: Open foundation and fine-tuned chat models. arXiv:2307.09288"},{"key":"2898_CR78","doi-asserted-by":"crossref","unstructured":"Wang, W., He, Z., Hong, W., Cheng, Y., Zhang, X., Qi, J., Ding, M., Gu, X., Huang, S., Xu, B., Dong, Y., & Tang, J. (2025). LVBench: An Extreme Long Video Understanding Benchmark. Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV) (pp. 22958\u201322967).","DOI":"10.1109\/ICCV51701.2025.02131"},{"key":"2898_CR79","unstructured":"Wang, L., Yang, N., Huang, X., Yang, L., Majumder, R., & Wei, F. (2024). Multilingual e5 text embeddings: A technical report. arXiv:2402.05672"},{"key":"2898_CR80","doi-asserted-by":"crossref","unstructured":"Wang, Z., Yu, S., Stengel-Eskin, E., Yoon, J., Cheng, F., Bertasius, G., & Bansal, M. (2025). Videotree: Adaptive tree-based video representation for llm reasoning on long videos. Proceedings of the Computer Vision and Pattern Recognition Conference (CVPR) (pp. 3272\u20133283).","DOI":"10.1109\/CVPR52734.2025.00311"},{"key":"2898_CR81","doi-asserted-by":"crossref","unstructured":"Wang, Z., Yu, S., Stengel-Eskin, E., Yoon, J., Cheng, F., Bertasius, G., & Bansal, M. (2025). Videotree: Adaptive tree-based video representation for llm reasoning on long videos. Proceedings of the Computer Vision and Pattern Recognition Conference (pp. 3272\u20133283).","DOI":"10.1109\/CVPR52734.2025.00311"},{"key":"2898_CR82","doi-asserted-by":"crossref","unstructured":"Wang, X., Zhang, Y., Zohar, O., & Yeung-Levy, S. (2024). Videoagent: Long-form video understanding with large language model as agent. European Conference on Computer Vision (pp. 58\u201376). Springer.","DOI":"10.1007\/978-3-031-72989-8_4"},{"issue":"2","key":"2898_CR83","doi-asserted-by":"publisher","first-page":"221","DOI":"10.1103\/RevModPhys.50.221","volume":"50","author":"A Wehrl","year":"1978","unstructured":"Wehrl, A. (1978). General properties of entropy. Reviews of Modern Physics, 50(2), 221.","journal-title":"Reviews of Modern Physics"},{"key":"2898_CR84","doi-asserted-by":"crossref","unstructured":"Wu, H., Li, D., Chen, B., & Li, J. (2024). Longvideobench: A benchmark for long-context interleaved video-language understanding. In: Globerson, A., Mackey, L., Belgrave, D., Fan, A., Paquet, U., Tomczak, J., & Zhang, C. (eds.) Advances in Neural Information Processing Systems, (vol. 37, pp. 28828\u201328857).","DOI":"10.52202\/079017-0907"},{"key":"2898_CR85","unstructured":"Wu, B., Yu, S., Chen, Z., Tenenbaum, J. B., & Gan, C. (2024). Star: A benchmark for situated reasoning in real-world videos. arXiv:2405.09711"},{"issue":"8","key":"2898_CR86","doi-asserted-by":"publisher","first-page":"8523","DOI":"10.1609\/aaai.v39i8.32920","volume":"39","author":"Z Wu","year":"2025","unstructured":"Wu, Z., Li, R., Xu, Z., Wang, Z., Xiao, C., & Liang, C. (2025). Friendsqa: A new large-scale deep video understanding dataset with fine-grained topic categorization for story videos. Proceedings of the AAAI Conference on Artificial Intelligence, 39(8), 8523\u20138531.","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"2898_CR87","doi-asserted-by":"crossref","unstructured":"Wu, Z., Wang, X., Chang, H., Chen, H., Sun, L., & Zhu, W. (2025). Aligning large multimodal model with sequential recommendation via content-behavior guidance. Proceedings of the 2025 International Conference on Multimedia Retrieval. ICMR \u201925 (pp. 1507\u20131516). New York, NY, USA: Association for Computing Machinery.","DOI":"10.1145\/3731715.3733273"},{"key":"2898_CR88","doi-asserted-by":"crossref","unstructured":"Xiao, J., Huang, N., Qin, H., Li, D., Li, Y., Zhu, F., Tao, Z., Yu, J., Lin, L., Chua, T.-S., & Yao A. (2025). Videoqa in the era of llms: An empirical study. International Journal of Computer Vision, 1\u201324.","DOI":"10.1007\/s11263-025-02385-8"},{"key":"2898_CR89","doi-asserted-by":"crossref","unstructured":"Xiao, J., Shang, X., Yao, A., & Chua, T.-S. (2021). Next-qa: Next phase of question-answering to explaining temporal actions. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (pp. 9777\u20139786).","DOI":"10.1109\/CVPR46437.2021.00965"},{"key":"2898_CR90","doi-asserted-by":"crossref","unstructured":"Xie, J., Han, T., Bain, M., Nagrani, A., Varol, G., Xie, W., & Zisserman, A. (2024). Autoad-zero: A training-free framework for zero-shot audio description. In: Proceedings of the Asian Conference on Computer Vision (ACCV), (pp. 2265\u20132281).","DOI":"10.1007\/978-981-96-0908-6_5"},{"key":"2898_CR91","doi-asserted-by":"crossref","unstructured":"Xu, Z., Guo, J., Zhang, C., Wang, Z., Xiao, C., & Liang, C. (2025). Quantum interference-inspired who-what-where composite-semantics instance search for story videos. In Proceedings of the 33rd ACM International Conference on Multimedia (pp. 4166-4174).","DOI":"10.1145\/3746027.3755325"},{"key":"2898_CR92","unstructured":"Xu, H., Ye, Q., Yan, M., Shi, Y., Ye, J., Xu, Y., Li, C., Bi, B., Qian, Q., Wang, W., & Xu, G. (2023). mplug-2: A modularized multi-modal foundation model across text, image and video. In: International Conference on Machine Learning, (pp. 38728\u201338748). PMLR."},{"key":"2898_CR93","doi-asserted-by":"crossref","unstructured":"Xu, D., Zhao, Z., Xiao, J., Wu, F., Zhang, H., He, X., & Zhuang, Y. (2017). Video question answering via gradually refined attention over appearance and motion. Proceedings of the 25th ACM International Conference on Multimedia. MM \u201917 (pp. 1645\u20131653). New York, NY, USA: Association for Computing Machinery.","DOI":"10.1145\/3123266.3123427"},{"key":"2898_CR94","unstructured":"Yang, Z., Chen, G., Li, X., Wang, W., & Yang, Y. (2024). Doraemongpt: toward understanding dynamic scenes with large language models (exemplified as a video agent). In: Proceedings of the 41st International Conference on Machine Learning, (pp. 55976\u201355997)."},{"key":"2898_CR95","unstructured":"Yang, Z., Li, L., Lin, K., Wang, J., Lin, C.-C., Liu, Z., & Wang, L. (2023). The dawn of lmms: Preliminary explorations with gpt-4v (ision). 9(1) arXiv:2309.17421"},{"key":"2898_CR96","unstructured":"Yang, A., Yang, B., Hui, B., Zheng, B., Yu, B., Zhou, C., Li, C., Li, C., Liu, D., Huang, F., Dong, G., Wei, H., Lin, H., Tang, J., Wang, J., Yang, J., Tu, J., Zhang, J., Ma, J., Xu, J., Zhou, J., Bai, J., He, J., Lin, J., Dang, K., Lu, K., Chen, K., Yang, K., Li, M., Xue, M., Ni, N., Zhang, P., Wang, P., Peng, R., Men, R., Gao, R., Lin, R., Wang, S., Bai, S., Tan, S., Zhu, T., Li, T., Liu, T., Ge, W., Deng, X., Zhou, X., Ren, X., Zhang, X., Wei, X., Ren, X., Fan, Y., Yao, Y., Zhang, Y., Wan, Y., Chu, Y., Liu, Y., Cui, Z., Zhang, Z., & Fan, Z. (2024). Qwen2 technical report. arXiv:2407.10671"},{"key":"2898_CR97","unstructured":"Yang, A., Yang, B., Zhang, B., Hui, B., Zheng, B., Yu, B., Li, C., Liu, D., Huang, F., Wei, H., Lin, H., Yang, J., Tu, J., Zhang, J., Yang, J., Yang, J., Zhou, J., Lin, J., Dang, K., Lu, K., Bao, K., Yang, K., Yu, L., Li, M., Xue, M., Zhang, P., Zhu, Q., Men, R., Lin, R., Li, T., Xia, T., Ren, X., Ren, X., Fan, Y., Su, Y., Zhang, Y., Wan, Y., Liu, Y., Cui, Z., Zhang, Z., & Qiu, Z. (2024). Qwen2.5 technical report. arXiv:2412.15115"},{"key":"2898_CR98","doi-asserted-by":"crossref","unstructured":"Yin, S., Fu, C., Zhao, S., Li, K., Sun, X., Xu, T., & Chen, E. (2023). A survey on multimodal large language models. arXiv:2306.13549","DOI":"10.1093\/nsr\/nwae403"},{"key":"2898_CR99","doi-asserted-by":"crossref","unstructured":"Yu, S., Cho, J., Yadav, P., Bansal, M., Oh, A., Naumann, T., Globerson, A., Saenko, K., & Hardt, M. (2023). Self-chained image-language model for video localization and question answering. In S. Levine (Ed.), Advances in Neural Information Processing Systems (Vol. 36, pp. 76749\u201376771).","DOI":"10.52202\/075280-3354"},{"key":"2898_CR100","doi-asserted-by":"crossref","unstructured":"Yu, J., Wu, Y., Chu, M., Ren, Z., Huang, Z., Chu, P., Zhang, R., He, Y., Li, Q., Li, S., & Li, Z. (2025). Vrbench: A benchmark for multi-step reasoning in long narrative videos. arXiv:2506.10857","DOI":"10.1109\/ICCV51701.2025.02011"},{"key":"2898_CR101","doi-asserted-by":"publisher","first-page":"487","DOI":"10.1007\/978-3-030-01234-2_29","volume-title":"Computer Vision - ECCV 2018","author":"Y Yu","year":"2018","unstructured":"Yu, Y., Kim, J., & Kim, G. (2018). A joint sequence fusion model for video question answering and retrieval. In V. Ferrari, M. Hebert, C. Sminchisescu, & Y. Weiss (Eds.), Computer Vision - ECCV 2018 (pp. 487\u2013503). Cham: Springer."},{"key":"2898_CR102","doi-asserted-by":"publisher","first-page":"9127","DOI":"10.1609\/aaai.v33i01.33019127","volume":"33","author":"Z Yu","year":"2019","unstructured":"Yu, Z., Xu, D., Yu, J., Yu, T., Zhao, Z., Zhuang, Y., & Tao, D. (2019). Activitynet-qa: A dataset for understanding complex web videos via question answering. Proceedings of the AAAI Conference on Artificial Intelligence, 33, 9127\u20139134.","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"2898_CR103","unstructured":"Zhang, T., Kishore, V., Wu, F., Weinberger, K. Q., & Artzi, Y. (2020). Bertscore: Evaluating text generation with bert. International Conference on Learning Representations."},{"key":"2898_CR104","doi-asserted-by":"crossref","unstructured":"Zhang, H., Li, X., & Bing, L. (2023). Video-llama: An instruction-tuned audio-visual language model for video understanding. arXiv:2306.02858","DOI":"10.18653\/v1\/2023.emnlp-demo.49"},{"key":"2898_CR105","unstructured":"Zhang, B., Li, K., Cheng, Z., Hu, Z., Yuan, Y., Chen, G., Leng, S., Jiang, Y., Zhang, H., Li, X., & Jin, P. (2025). Videollama 3: Frontier multimodal foundation models for image and video understanding. arXiv:2501.13106"},{"key":"2898_CR106","unstructured":"Zhang, Y., Li, M., Long, D., Zhang, X., Lin, H., Yang, B., Xie, P., Yang, A., Liu, D., Lin, J., Huang, F., & Zhou, J. (2025). Qwen3 embedding: Advancing text embedding and reranking through foundation models. arXiv:2506.05176"},{"key":"2898_CR107","doi-asserted-by":"crossref","unstructured":"Zhang, C., Lu, T., Islam, M.M., Wang, Z., Yu, S., Bansal, M., & Bertasius, G. (2024). A simple LLM framework for long-range video question-answering. In: Al-Onaizan, Y., Bansal, M., & Chen, Y.-N. (eds.) Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing, (pp. 21715\u201321737). Association for Computational Linguistics, Miami, Florida, USA.","DOI":"10.18653\/v1\/2024.emnlp-main.1209"},{"key":"2898_CR108","unstructured":"Zhang, P., Zhang, K., Li, B., Zeng, G., Yang, J., Zhang, Y., Wang, Z., Tan, H., Li, C., & Liu, Z. (2024). Long context transfer from language to vision.arXiv:2406.16852"},{"key":"2898_CR109","doi-asserted-by":"crossref","unstructured":"Zhang, L., Zhao, T., Ying, H., Ma, Y., & Lee, K. (2024). Omagent: A multi-modal agent framework for complex video understanding with task divide-and-conquer. Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing (pp. 10031\u201310045).","DOI":"10.18653\/v1\/2024.emnlp-main.559"},{"issue":"11","key":"2898_CR110","doi-asserted-by":"publisher","first-page":"7726","DOI":"10.1007\/s11263-025-02555-8","volume":"133","author":"H Zhang","year":"2025","unstructured":"Zhang, H., Dong, L., Liu, Y., Huang, Y., Wang, Y., Wang, L., & Qiao, Y. (2025). Lvbench: A benchmark for long-form video understanding with versatile multi-modal question answering. International Journal of Computer Vision, 133(11), 7726\u20137747.","journal-title":"International Journal of Computer Vision"},{"key":"2898_CR111","doi-asserted-by":"crossref","unstructured":"Zhao, Y., Misra, I., Kr\u00e4henb\u00fchl, P., & Girdhar, R. (2023). Learning video representations from large language models. Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (pp. 6586\u20136597).","DOI":"10.1109\/CVPR52729.2023.00637"},{"key":"2898_CR112","doi-asserted-by":"crossref","unstructured":"Zhong, Y., Ji, W., Xiao, J., Li, Y., Deng, W., & Chua, T.-S. (2022). Video question answering: Datasets, algorithms and challenges. In Y. Goldberg, Z. Kozareva, & Y. Zhang (Eds.), Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing (pp. 6439\u20136455). Abu Dhabi, United Arab Emirates: Association for Computational Linguistics.","DOI":"10.18653\/v1\/2022.emnlp-main.432"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-026-02898-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-026-02898-w","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-026-02898-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,20]],"date-time":"2026-07-20T16:15:01Z","timestamp":1784564101000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-026-02898-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6]]},"references-count":112,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2026,6]]}},"alternative-id":["2898"],"URL":"https:\/\/doi.org\/10.1007\/s11263-026-02898-w","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,6]]},"assertion":[{"value":"7 November 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 May 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"4 June 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"All authors certify that they have no affiliations with or involvement in any organization or entity with any financial interest or non-financial interest in the subject matter or materials discussed in this manuscript.","order":1,"name":"Ethics","label":"Competing Interests","group":{"name":"EthicsHeading","label":"Declarations"}}],"article-number":"308"}}