{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,28]],"date-time":"2026-03-28T06:42:23Z","timestamp":1774680143982,"version":"3.50.1"},"reference-count":24,"publisher":"IEEE","license":[{"start":{"date-parts":[[2026,2,3]],"date-time":"2026-02-03T00:00:00Z","timestamp":1770076800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,2,3]],"date-time":"2026-02-03T00:00:00Z","timestamp":1770076800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026,2,3]]},"DOI":"10.1109\/icce67443.2026.11449786","type":"proceedings-article","created":{"date-parts":[[2026,3,27]],"date-time":"2026-03-27T19:47:50Z","timestamp":1774640870000},"page":"1-6","source":"Crossref","is-referenced-by-count":0,"title":["HiSum: A Hierarchical Framework for Long Video Summarization with Temporal Enhancement"],"prefix":"10.1109","author":[{"given":"Jiaojiao","family":"Lin","sequence":"first","affiliation":[{"name":"Intel Corporation,Intel Client Computing Group,Shanghai,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Cong","family":"Zhang","sequence":"additional","affiliation":[{"name":"Intel Corporation,Intel Client Computing Group,Shanghai,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wenjian","family":"Shao","sequence":"additional","affiliation":[{"name":"Intel Corporation,Intel Client Computing Group,Shanghai,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lingyun","family":"Zhu","sequence":"additional","affiliation":[{"name":"Intel Corporation,Intel Client Computing Group,Shanghai,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yi","family":"Qian","sequence":"additional","affiliation":[{"name":"Intel Corporation,Intel Client Computing Group,Shanghai,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jokul","family":"Li","sequence":"additional","affiliation":[{"name":"Intel Corporation,Intel Client Computing Group,Shanghai,China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Qwen2. 5-vl technical report","author":"Bai","year":"2025"},{"key":"ref2","article-title":"Internvl3: Exploring advanced training and test-time recipes for open-source multimodal models","author":"Zhu","year":"2025"},{"key":"ref3","article-title":"Llava-onevision: Easy visual task transfer","author":"Li","year":"2024"},{"key":"ref4","article-title":"Video instruction tuning with synthetic data","author":"Zhang","year":"2024"},{"key":"ref5","article-title":"Videochat-flash: Hierarchical compression for long-context video modeling","author":"Li","year":"2024"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.679"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.1209"},{"key":"ref8","article-title":"Mlvu: A comprehensive benchmark for multitask long video understanding","author":"Zhou","year":"2024"},{"key":"ref9","article-title":"Infinibench: A comprehensive benchmark for large multimodal models in very long video understanding","author":"Ataallah","year":"2024"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.510"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"key":"ref12","article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","author":"Dosovitskiy","year":"2020"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2102.05095"},{"key":"ref15","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"International conference on machine learning","author":"Radford"},{"key":"ref16","article-title":"Kimi-vl technical report","author":"Du","year":"2025"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01725"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.02436"},{"key":"ref19","article-title":"Videollamb: Long-context video understanding with recurrent memory bridges","author":"Wang","year":"2024"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01723"},{"key":"ref21","article-title":"Longvitu: Instruction tuning for long-form video understanding","author":"Wu","year":"2025"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72989-8_4"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01764"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1080\/01621459.2012.737745"}],"event":{"name":"2026 IEEE International Conference on Consumer Electronics (ICCE)","location":"Dubai, United Arab Emirates","start":{"date-parts":[[2026,2,3]]},"end":{"date-parts":[[2026,2,5]]}},"container-title":["2026 IEEE International Conference on Consumer Electronics (ICCE)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11449575\/11449585\/11449786.pdf?arnumber=11449786","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,28]],"date-time":"2026-03-28T05:16:20Z","timestamp":1774674980000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11449786\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,2,3]]},"references-count":24,"URL":"https:\/\/doi.org\/10.1109\/icce67443.2026.11449786","relation":{},"subject":[],"published":{"date-parts":[[2026,2,3]]}}}