{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,12]],"date-time":"2026-06-12T19:59:20Z","timestamp":1781294360968,"version":"3.54.1"},"reference-count":89,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"3","license":[{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T00:00:00Z","timestamp":1777593600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"National Key R&#x0026;D Program of China","award":["2024YFE0200800"],"award-info":[{"award-number":["2024YFE0200800"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62321001"],"award-info":[{"award-number":["62321001"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62471055"],"award-info":[{"award-number":["62471055"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U23B2001"],"award-info":[{"award-number":["U23B2001"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62101064"],"award-info":[{"award-number":["62101064"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62201072"],"award-info":[{"award-number":["62201072"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Fundamental and Interdisciplinary Disciplines Breakthrough Plan"},{"DOI":"10.13039\/501100002338","name":"Ministry of Education of the People's Republic of China","doi-asserted-by":"publisher","award":["JYB2025XDXM107"],"award-info":[{"award-number":["JYB2025XDXM107"]}],"id":[{"id":"10.13039\/501100002338","id-type":"DOI","asserted-by":"publisher"}]},{"name":"High-Quality Development Project of the MIIT","award":["2440STCZB2584"],"award-info":[{"award-number":["2440STCZB2584"]}]},{"name":"Ministry of Education and China Mobile Joint Fund","award":["MCM20200202"],"award-info":[{"award-number":["MCM20200202"]}]},{"name":"Ministry of Education and China Mobile Joint Fund","award":["MCM20180101"],"award-info":[{"award-number":["MCM20180101"]}]},{"DOI":"10.13039\/501100002766","name":"Beijing University of Posts and Telecommunications","doi-asserted-by":"publisher","award":["2025YZ005"],"award-info":[{"award-number":["2025YZ005"]}],"id":[{"id":"10.13039\/501100002766","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Serv. Comput."],"published-print":{"date-parts":[[2026,5]]},"DOI":"10.1109\/tsc.2026.3687026","type":"journal-article","created":{"date-parts":[[2026,4,24]],"date-time":"2026-04-24T19:46:43Z","timestamp":1777060003000},"page":"2103-2117","source":"Crossref","is-referenced-by-count":0,"title":["Enhancing MLLMs for Online Understanding in Video Services via Preference Optimization"],"prefix":"10.1109","volume":"19","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0829-4624","authenticated-orcid":false,"given":"Qi","family":"Qi","sequence":"first","affiliation":[{"name":"State Key Laboratory of Networking and Switching Technology, Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-2746-1327","authenticated-orcid":false,"given":"Yixiao","family":"He","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Networking and Switching Technology, Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9713-9938","authenticated-orcid":false,"given":"Menghao","family":"Zhang","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Networking and Switching Technology, Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3072-7422","authenticated-orcid":false,"given":"Haifeng","family":"Sun","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Networking and Switching Technology, Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1691-6457","authenticated-orcid":false,"given":"Pengfei","family":"Ren","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Networking and Switching Technology, Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5733-2060","authenticated-orcid":false,"given":"Huazheng","family":"Wang","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Networking and Switching Technology, Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1486-0573","authenticated-orcid":false,"given":"Jianxin","family":"Liao","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Networking and Switching Technology, Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2182-2228","authenticated-orcid":false,"given":"Jingyu","family":"Wang","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Networking and Switching Technology, Beijing University of Posts and Telecommunications, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/TSC.2025.3592391"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TSC.2023.3349051"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/TSC.2024.3451237"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/TSC.2022.3155500"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/TSC.2022.3150012"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/TSC.2024.3451215"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/TSC.2017.2653116"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/TSC.2024.3407588"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/TSC.2023.3248321"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2024.3404076"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/TMC.2025.3591188"},{"key":"ref12","first-page":"19730","article-title":"BLIP-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Li"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.52202\/075280-1516"},{"key":"ref14","article-title":"Qwen2.5-VL technical report","author":"Team","year":"2025"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01742"},{"key":"ref16","article-title":"Online video understanding: A comprehensive benchmark and memory-augmented method","author":"Huang","year":"2025"},{"key":"ref17","article-title":"Streaming long video understanding with large language models","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Qian"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1145\/3746027.3754839"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.02239"},{"key":"ref20","article-title":"Flash-VStream: Memory-based real-time understanding for long video streams","author":"Zhang","year":"2024"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-demo.49"},{"key":"ref22","article-title":"LLaVA-video: Video instruction tuning with synthetic data","volume":"2025","author":"Zhang","year":"2025","journal-title":"Trans. Mach. Learn. Res."},{"key":"ref23","article-title":"LLaVA-onevision: Easy visual task transfer","volume":"2025","author":"Li","year":"2025","journal-title":"Trans. Mach. Learn. Res."},{"key":"ref24","article-title":"Qwen2-VL: Enhancing vision-language model\u2019s perception of the world at any resolution","author":"Team","year":"2024"},{"key":"ref25","first-page":"38571","article-title":"Unhackable temporal reward for scalable video MLLMs","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Yu"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01281"},{"key":"ref27","article-title":"VideoHallucer: Evaluating intrinsic and extrinsic hallucinations in large video-language models","author":"Wang","year":"2024"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ICME59968.2025.11209127"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.20"},{"key":"ref30","first-page":"4908","article-title":"Analyzing and mitigating object hallucination in large vision-language models","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Zhou"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01316"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73010-8_8"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.460"},{"key":"ref34","article-title":"Direct preference optimization: Your language model is secretly a reward model","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Rafailov"},{"key":"ref35","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Radford"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72983-6_18"},{"key":"ref37","first-page":"68777","article-title":"FG-CLIP: Fine-grained visual and textual alignment","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Xie"},{"key":"ref38","article-title":"FG-CLIP 2: A bilingual fine-grained vision-language alignment model","author":"Xie","year":"2025"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/icassp55912.2026.11463959"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01761"},{"key":"ref41","first-page":"28828","article-title":"Longvideobench: A benchmark for long-context interleaved video-language understanding","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Wu"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01278"},{"key":"ref43","first-page":"12888","article-title":"BLIP: Bootstrapping language-image pre-training for unified vision-language understanding and generation","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Li"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01100"},{"key":"ref45","article-title":"GPT-4O system card","year":"2024"},{"key":"ref46","article-title":"Qwen2 technical report","author":"Team","year":"2024"},{"key":"ref47","article-title":"The Llama 3 herd of models","author":"Dubey","year":"2024"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-main.432"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1145\/3746027.3754999"},{"key":"ref51","article-title":"Do LVLMs truly understand video anomalies? Revealing hallucination via co-occurrence patterns","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Zhang"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1145\/3746027.3754500"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.01954"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01825"},{"key":"ref55","article-title":"Knowledge-based visual question answer with multimodal processing, retrieval and filtering","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Hong"},{"key":"ref56","article-title":"GoT: Unleashing reasoning capability of multimodal large language model for visual generation and editing","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Fang"},{"key":"ref57","article-title":"LlaVA-neXT: Improved reasoning, OCR, and world knowledge","author":"Liu","year":"2024"},{"key":"ref58","article-title":"Expanding performance boundaries of open-source multimodal models with model, data, and test-time scaling","author":"Chen","year":"2024"},{"key":"ref59","first-page":"2252","article-title":"Patch n\u2019 pack: Navit, a vision transformer for any aspect ratio and resolution","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Dehghani"},{"key":"ref60","first-page":"27730","article-title":"Training language models to follow instructions with human feedback","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Ouyang"},{"key":"ref61","article-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.775"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01310"},{"key":"ref64","article-title":"MM-RLHF: The next step forward in multimodal LLM alignment","author":"Zhang","year":"2025"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/ICME59968.2025.11209377"},{"key":"ref66","article-title":"Aligning modalities in vision large language models via preference fine-tuning","author":"Zhou","year":"2024"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01861"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.358"},{"key":"ref69","article-title":"Enhancing the reasoning ability of multimodal large language models via mixed preference optimization","author":"Wang","year":"2024"},{"key":"ref70","article-title":"DAMA: Data- and model-aware alignment of multi-modal LLMs","author":"Lu","year":"2025"},{"key":"ref71","article-title":"Fine-grained preference optimization improves spatial reasoning in VLMs","author":"Shen","year":"2025"},{"key":"ref72","article-title":"Elicit and enhance: Advancing multimodal reasoning in medical scenarios","author":"Mu","year":"2025"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.775"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.naacl-long.349"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-emnlp.336"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.02048"},{"key":"ref77","article-title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","author":"Reid","year":"2024"},{"key":"ref78","article-title":"VideoLLaMA 2: Advancing spatial-temporal modeling and audio understanding in video-LLMs","author":"Cheng","year":"2024"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02520"},{"key":"ref80","article-title":"Video-CCAM: Enhancing video-language understanding with causal cross-attention masks for short and long videos","author":"Fei","year":"2024"},{"key":"ref81","article-title":"Long context transfer from language to vision","volume":"2025","author":"Zhang","year":"2025","journal-title":"Trans. Mach. Learn. Res."},{"key":"ref82","doi-asserted-by":"publisher","DOI":"10.1036\/1097-8542.361800"},{"key":"ref83","article-title":"Qwen2.5 technical report","author":"Team","year":"2024"},{"key":"ref84","article-title":"LongVU: Spatiotemporal adaptive compression for long video-language understanding","author":"Shen","year":"2024"},{"key":"ref85","first-page":"12513","article-title":"LoRA: Low-rank adaptation of large language models","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Hu"},{"key":"ref86","doi-asserted-by":"publisher","DOI":"10.1109\/SC41405.2020.00024"},{"key":"ref87","first-page":"13083","article-title":"CHiP: Cross-modal hierarchical direct preference optimization for multimodal LLMs","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Fu"},{"key":"ref88","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.02245"},{"key":"ref89","article-title":"MIMO-VL technical report","author":"Yue","year":"2025"}],"container-title":["IEEE Transactions on Services Computing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/4629386\/11559165\/11494414.pdf?arnumber=11494414","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,12]],"date-time":"2026-06-12T19:43:55Z","timestamp":1781293435000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11494414\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5]]},"references-count":89,"journal-issue":{"issue":"3"},"URL":"https:\/\/doi.org\/10.1109\/tsc.2026.3687026","relation":{},"ISSN":["1939-1374","2372-0204"],"issn-type":[{"value":"1939-1374","type":"electronic"},{"value":"2372-0204","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,5]]}}}