{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,4]],"date-time":"2026-05-04T10:23:27Z","timestamp":1777890207153,"version":"3.51.4"},"reference-count":134,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,10,19]]},"DOI":"10.1109\/iccv51701.2025.00923","type":"proceedings-article","created":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T19:45:49Z","timestamp":1777491949000},"page":"9900-9912","source":"Crossref","is-referenced-by-count":0,"title":["Learning Streaming Video Representation via Multitask Training"],"prefix":"10.1109","author":[{"given":"Yibin","family":"Yan","sequence":"first","affiliation":[{"name":"School of Artificial Intelligence, Shanghai Jiao Tong University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jilan","family":"Xu","sequence":"additional","affiliation":[{"name":"Fudan University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shangzhe","family":"Di","sequence":"additional","affiliation":[{"name":"School of Artificial Intelligence, Shanghai Jiao Tong University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yikun","family":"Liu","sequence":"additional","affiliation":[{"name":"School of Artificial Intelligence, Shanghai Jiao Tong University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yudi","family":"Shi","sequence":"additional","affiliation":[{"name":"School of Artificial Intelligence, Shanghai Jiao Tong University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qirui","family":"Chen","sequence":"additional","affiliation":[{"name":"School of Artificial Intelligence, Shanghai Jiao Tong University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zeqian","family":"Li","sequence":"additional","affiliation":[{"name":"School of Artificial Intelligence, Shanghai Jiao Tong University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yifei","family":"Huang","sequence":"additional","affiliation":[{"name":"Shanghai AI Laboratory"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Weidi","family":"Xie","sequence":"additional","affiliation":[{"name":"School of Artificial Intelligence, Shanghai Jiao Tong University"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Vatt: Transformers for multimodal self-supervised learning from raw video, audio and text","author":"Akbari","year":"2021","journal-title":"NeurIPS"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.618"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"ref4","article-title":"Qwen technical report","author":"Bai","year":"2023","journal-title":"arXiv preprint arXiv"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02157"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00175"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2102.05095"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.15607\/rss.2023.xix.025"},{"key":"ref9","article-title":"Language models are few-shot learners","author":"Brown","year":"2020","journal-title":"NeurIPS"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.675"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298698"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1023\/A:1007379606734"},{"key":"ref14","article-title":"Collecting highly parallel data for paraphrase evaluation","author":"Chen","year":"2011","journal-title":"ACL"},{"key":"ref15","article-title":"Videollm: Modeling video sequence with large language models","author":"Chen","year":"2023","journal-title":"arXiv preprint arXiv"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01930"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01742"},{"key":"ref18","article-title":"Generative pretraining from pixels","author":"Chen","year":"2020","journal-title":"ICML"},{"key":"ref19","article-title":"Pix2seq: A language modeling framework for object detection","author":"Chen","year":"2022","journal-title":"ICLR"},{"key":"ref20","article-title":"Vision transformer adapter for dense predictions","author":"Chen","year":"2023","journal-title":"ICLR"},{"key":"ref21","article-title":"Videollama 2: Advancing spatialtemporal modeling and audio understanding in video-llms","author":"Cheng","year":"2024","journal-title":"arXiv preprint arXiv"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2022.3200245"},{"key":"ref23","article-title":"Palm: Scaling language modeling with pathways","author":"Chowdhery","year":"2023","journal-title":"Journal of Machine Learning Research"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46454-1_17"},{"key":"ref25","article-title":"Streaming video question-answering with incontext video kv-cache retrieval","author":"Di","year":"2025","journal-title":"ICLR"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00254"},{"key":"ref27","article-title":"An image is worth 16 \u00d7 16 words: Transformers for image recognition at scale","author":"Dosovitskiy","year":"2021","journal-title":"ICLR"},{"key":"ref28","volume-title":"Palm-e: An embodied multimodal language model","author":"Driess","year":"2023"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.304"},{"key":"ref30","article-title":"Scalable pre-training of large autoregressive image models","author":"El-Nouby","year":"2024","journal-title":"ICML"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/icassp55912.2026.11463959"},{"key":"ref32","volume-title":"Trecvit: A recurrent video transformer","author":"P\u0103tr\u0103ucean","year":"2024"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00675"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.787"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00630"},{"key":"ref36","article-title":"Video-mme: The first-ever comprehensive evaluation benchmark of multi-modal 11 ms in video analysis","author":"Fu","year":"2025","journal-title":"CVPR"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.563"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.622"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01834"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01274"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"ref42","article-title":"Continual transformers: redundancy-free attention for online inference","author":"Hedegaard","year":"2023","journal-title":"ICLR"},{"key":"ref43","article-title":"Lora: Low-rank adaptation of large language models","author":"Hu","year":"2022","journal-title":"ICLR"},{"key":"ref44","article-title":"Minvis: A minimal video instance segmentation framework without video-based training","author":"Huang","year":"2022","journal-title":"NeurIPS"},{"key":"ref45","article-title":"Vinci: A real-time embodied smart assistant based on egocentric vision-language model","author":"Huang","year":"2024","journal-title":"arXiv preprint arXiv"},{"key":"ref46","article-title":"Online video understanding: A comprehensive benchmark and memory-augmented method","author":"Huang","year":"2024","journal-title":"arXiv preprint arXiv"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1502.03167"},{"key":"ref48","volume-title":"THUMOS challenge: Action recognition with a large number of classes","author":"Jiang","year":"2014"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00781"},{"key":"ref50","article-title":"Deep reinforcement learning for autonomous driving: A survey","author":"Kiran","year":"2021","journal-title":"TPAMI"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.83"},{"key":"ref52","article-title":"Detecting moments and highlights in videos via natural language queries","author":"Lei","year":"2021","journal-title":"NeurIPS"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01826"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2023.3282631"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02095"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73010-8_25"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00476"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72952-2_19"},{"key":"ref59","article-title":"Video-llava: Learning united visual representation by alignment before projection","author":"Lin","year":"2023","journal-title":"EMNLP"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19833-5_23"},{"key":"ref62","article-title":"Deepseek-v3 technical report","author":"Liu","year":"2024","journal-title":"arXiv preprint arXiv"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.52202\/075280-1516"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"ref65","volume-title":"Llava-next: Improved reasoning, ocr, and world knowledge","author":"Liu","year":"2024"},{"key":"ref66","article-title":"Streamchat: Chatting with streaming video","author":"Liu","year":"2024","journal-title":"arXiv preprint arXiv"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2022.3217368"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00320"},{"key":"ref69","article-title":"Unified-io: A unified model for vision, language, and multi-modal tasks","author":"Lu","year":"2023","journal-title":"ICLR"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00272"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.433"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414640"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01398"},{"key":"ref74","article-title":"St-adapter: Parameter-efficient image-to-video transfer learning","author":"Pan","year":"2022","journal-title":"NeurIPS"},{"key":"ref75","article-title":"Image transformer","author":"Parmar","year":"2018","journal-title":"ICML"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-022-01629-1"},{"key":"ref77","article-title":"Streaming long video understanding with large language models","author":"Qian","year":"2024","journal-title":"NeurIPS"},{"key":"ref78","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021","journal-title":"ICML"},{"key":"ref79","article-title":"Real-world robot learning with masked visual pre-training","author":"Radosavovic","year":"2023","journal-title":"CoRL"},{"key":"ref80","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","author":"Raffel","year":"2020","journal-title":"JMLR"},{"key":"ref81","article-title":"An empirical study of autoregressive pretraining from videos","author":"Rajasegaran","year":"2025","journal-title":"arXiv preprint arXiv"},{"key":"ref82","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-11752-2_15"},{"key":"ref83","article-title":"An overview of multi-task learning in deep neural networks","author":"Ruder","year":"2017","journal-title":"arXiv preprint arXiv"},{"key":"ref84","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58555-6_13"},{"key":"ref85","article-title":"Autoregressive model beats diffusion: Llama for scalable image generation","author":"Sun","year":"2024","journal-title":"arXiv preprint arXiv"},{"key":"ref86","article-title":"Chameleon: Mixed-modal early-fusion foundation models","author":"Team","year":"2024","journal-title":"arXiv preprint arXiv"},{"key":"ref87","article-title":"Visual autoregressive modeling: Scalable image generation via next-scale prediction","author":"Tian","year":"2024","journal-title":"NeurIPS"},{"key":"ref88","article-title":"Learning language-visual embedding for movie understanding with natural-language","author":"Torabi","year":"2016","journal-title":"arXiv preprint arXiv"},{"key":"ref89","article-title":"Llama: Open and efficient foundation language models","author":"Touvron","year":"2023","journal-title":"arXiv preprint arXiv"},{"key":"ref90","article-title":"Llama 2: Open foundation and fine-tuned chat models","author":"Touvron","year":"2023","journal-title":"arXiv preprint arXiv"},{"key":"ref91","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.510"},{"key":"ref92","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00675"},{"key":"ref93","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00565"},{"key":"ref94","article-title":"Conditional image generation with pixelcnn decoders","author":"Van Den Oord","year":"2016","journal-title":"NeurIPS"},{"key":"ref95","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00375"},{"key":"ref96","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01271"},{"key":"ref97","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46484-8_2"},{"key":"ref98","article-title":"Ofa: Unifying architectures, tasks, and modalities through a simple sequence-to-sequence learning framework","author":"Wang","year":"2022","journal-title":"ICML"},{"key":"ref99","article-title":"Qwen2-vl: Enhancing vision-language model\u2019s perception of the world at any resolution","author":"Wang","year":"2024","journal-title":"arXiv preprint arXiv"},{"key":"ref100","article-title":"Visionllm: Large language model is also an open-ended decoder for vision-centric tasks","author":"Wang","year":"2023","journal-title":"NeurIPS"},{"key":"ref101","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00468"},{"key":"ref102","article-title":"Internvideo: General video foundation models via generative and discriminative learning","author":"Wang","year":"2022","journal-title":"arXiv preprint arXiv"},{"key":"ref103","article-title":"Internvid: A large-scale video-text dataset for multimodal understanding and generation","author":"Wang","year":"2024","journal-title":"ICLR"},{"key":"ref104","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73013-9_23"},{"key":"ref105","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19815-1_34"},{"key":"ref106","article-title":"Visionllm v2: An end-to-end generalist multimodal large language model for hundreds of vision-language tasks","author":"Wu","year":"2024","journal-title":"NeurIPS"},{"key":"ref107","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.571"},{"key":"ref108","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01284"},{"key":"ref109","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00563"},{"key":"ref110","article-title":"Long short-term transformer for online action detection","author":"Xu","year":"2021","journal-title":"NeurIPS"},{"key":"ref111","article-title":"Videococa: Video-text modeling with zero-shot transfer from contrastive captioners","author":"Yan","year":"2022","journal-title":"arXiv preprint arXiv"},{"key":"ref112","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52734.2025.02781"},{"key":"ref113","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00529"},{"key":"ref114","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00316"},{"key":"ref115","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00794"},{"key":"ref116","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00089"},{"key":"ref117","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00391"},{"key":"ref118","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01589"},{"key":"ref119","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01100"},{"key":"ref120","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.521"},{"key":"ref121","article-title":"Flash-vstream: Memorybased real-time understanding for long video streams","author":"Zhang","year":"2024","journal-title":"arXiv preprint arXiv"},{"key":"ref122","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-naacl.51"},{"key":"ref123","doi-asserted-by":"publisher","DOI":"10.1109\/tpami.2025.3552694"},{"key":"ref124","volume-title":"Llava-next: A strong zero-shot video understanding model","author":"Zhang","year":"2024"},{"key":"ref125","article-title":"Video instruction tuning with synthetic data","author":"Zhang","year":"2024","journal-title":"arXiv preprint arXiv"},{"key":"ref126","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00876"},{"key":"ref127","article-title":"Videoprism: A foundational visual encoder for video understanding","author":"Zhao","year":"2024","journal-title":"ICML"},{"key":"ref128","article-title":"Does video-text pretraining help open-vocabulary online action detection?","author":"Zhao","year":"2024","journal-title":"NeurIPS"},{"key":"ref129","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19830-4_28"},{"key":"ref130","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00637"},{"key":"ref131","article-title":"Judging llm-as-a-judge with mt-bench and chatbot arena","author":"Zheng","year":"2023","journal-title":"NeurIPS"},{"key":"ref132","article-title":"Mlvu: A comprehensive benchmark for multitask long video understanding","author":"Zhou","year":"2025","journal-title":"CVPR"},{"key":"ref133","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01727"},{"key":"ref134","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01630"}],"event":{"name":"2025 IEEE\/CVF International Conference on Computer Vision (ICCV)","location":"Honolulu, HI, USA","start":{"date-parts":[[2025,10,19]]},"end":{"date-parts":[[2025,10,25]]}},"container-title":["2025 IEEE\/CVF International Conference on Computer Vision (ICCV)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11443115\/11443287\/11445037.pdf?arnumber=11445037","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T05:30:59Z","timestamp":1777613459000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11445037\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,19]]},"references-count":134,"URL":"https:\/\/doi.org\/10.1109\/iccv51701.2025.00923","relation":{},"subject":[],"published":{"date-parts":[[2025,10,19]]}}}