{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,26]],"date-time":"2026-07-26T14:02:29Z","timestamp":1785074549443,"version":"3.55.0"},"reference-count":48,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,9,1]],"date-time":"2026-09-01T00:00:00Z","timestamp":1788220800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100005146","name":"Basic Research Programs of Sichuan Province","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100005146","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100013318","name":"Shanxi Provincial Department of Science and Technology","doi-asserted-by":"publisher","award":["202403021222152"],"award-info":[{"award-number":["202403021222152"]}],"id":[{"id":"10.13039\/501100013318","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100013318","name":"Shanxi Provincial Department of Science and Technology","doi-asserted-by":"publisher","award":["202403021222180"],"award-info":[{"award-number":["202403021222180"]}],"id":[{"id":"10.13039\/501100013318","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Knowledge-Based Systems"],"published-print":{"date-parts":[[2026,9]]},"DOI":"10.1016\/j.knosys.2026.116374","type":"journal-article","created":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T23:18:36Z","timestamp":1781651916000},"page":"116374","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Fine-grained semantics-driven decoupling optimization for joint video moment retrieval and highlight detection"],"prefix":"10.1016","volume":"349","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-9517-5383","authenticated-orcid":false,"given":"Fuwei","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-7774-8043","authenticated-orcid":false,"given":"Peiyuan","family":"Liang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-7793-0957","authenticated-orcid":false,"given":"Yuxin","family":"Feng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-4094-8484","authenticated-orcid":false,"given":"Weiming","family":"Han","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4080-4209","authenticated-orcid":false,"given":"Boying","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-1832-1168","authenticated-orcid":false,"given":"Yuanyuan","family":"Shen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4934-3640","authenticated-orcid":false,"given":"Shuangjiao","family":"Zhai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Alex Jinpeng","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.knosys.2026.116374_b1","first-page":"11846","article-title":"Detecting moments and highlights in videos via natural language queries","author":"Lei","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.knosys.2026.116374_b2","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"3042","article-title":"Umt: Unified multi-modal transformers for joint video moment retrieval and highlight detection","author":"Liu","year":"2022"},{"key":"10.1016\/j.knosys.2026.116374_b3","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"2794","article-title":"Univtg: Towards unified video-language temporal grounding","author":"Lin","year":"2023"},{"key":"10.1016\/j.knosys.2026.116374_b4","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"23023","article-title":"Query-dependent video representation for moment retrieval and highlight detection","author":"Moon","year":"2023"},{"key":"10.1016\/j.knosys.2026.116374_b5","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"4998","article-title":"Tr-detr: Task-reciprocal transformer for joint moment retrieval and highlight detection","author":"Sun","year":"2024"},{"key":"10.1016\/j.knosys.2026.116374_b6","series-title":"Proceedings of the 33rd ACM International Conference on Multimedia","first-page":"4514","article-title":"Ms-detr: Towards effective video moment retrieval and highlight detection by joint motion-semantic learning","author":"Ma","year":"2025"},{"key":"10.1016\/j.knosys.2026.116374_b7","first-page":"1235","article-title":"Online data organizer: micro-video categorization by structure-guided multimodal dictionary learning","author":"Liu","year":"2018","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.knosys.2026.116374_b8","series-title":"Proceedings of the IEEE International Conference on Computer Vision","first-page":"5267","article-title":"Tall: Temporal activity localization via language query","author":"Gao","year":"2017"},{"key":"10.1016\/j.knosys.2026.116374_b9","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"1523","article-title":"Fast video moment retrieval","author":"Gao","year":"2021"},{"key":"10.1016\/j.knosys.2026.116374_b10","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"8978","article-title":"Zero-shot video moment retrieval via off-the-shelf multimodal large language models","author":"Xu","year":"2025"},{"key":"10.1016\/j.knosys.2026.116374_b11","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2025.113071","article-title":"Variational global clue inference for weakly supervised video moment retrieval","author":"Lv","year":"2025","journal-title":"Knowl.-Based Syst."},{"key":"10.1016\/j.knosys.2026.116374_b12","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"5179","article-title":"Tvsum: Summarizing web videos using titles","author":"Song","year":"2015"},{"key":"10.1016\/j.knosys.2026.116374_b13","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"1258","article-title":"Less is more: Learning highlight detection from video duration","author":"Xiong","year":"2019"},{"key":"10.1016\/j.knosys.2026.116374_b14","doi-asserted-by":"crossref","first-page":"40542","DOI":"10.52202\/075280-1764","article-title":"Mr. hisum: A large-scale dataset for video highlight detection and summarization","author":"Sul","year":"2023","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.knosys.2026.116374_b15","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"18709","article-title":"Bridging the gap: A unified video comprehension framework for moment retrieval and highlight detection","author":"Xiao","year":"2024"},{"key":"10.1016\/j.knosys.2026.116374_b16","series-title":"Chinese Conference on Pattern Recognition and Computer Vision","first-page":"254","article-title":"Utilizing text-video relationships: A text-driven multi-modal fusion framework for moment retrieval and highlight detection","author":"Zhou","year":"2024"},{"key":"10.1016\/j.knosys.2026.116374_b17","series-title":"2024 IEEE International Conference on Multimedia and Expo","first-page":"1","article-title":"Frequency-domain enhanced cross-modal interaction mechanism for joint video moment retrieval and highlight detection","author":"Feng","year":"2024"},{"key":"10.1016\/j.knosys.2026.116374_b18","series-title":"International Conference on Machine Learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.knosys.2026.116374_b19","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"6202","article-title":"Slowfast networks for video recognition","author":"Feichtenhofer","year":"2019"},{"key":"10.1016\/j.knosys.2026.116374_b20","series-title":"Very deep convolutional networks for large-scale image recognition","author":"Simonyan","year":"2014"},{"key":"10.1016\/j.knosys.2026.116374_b21","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"6299","article-title":"Quo vadis, action recognition? A new model and the kinetics dataset","author":"Carreira","year":"2017"},{"key":"10.1016\/j.knosys.2026.116374_b22","series-title":"Proceedings of the IEEE International Conference on Computer Vision","first-page":"4489","article-title":"Learning spatiotemporal features with 3d convolutional networks","author":"Tran","year":"2015"},{"key":"10.1016\/j.knosys.2026.116374_b23","series-title":"Mamba: Linear-time sequence modeling with selective state spaces","author":"Gu","year":"2023"},{"key":"10.1016\/j.knosys.2026.116374_b24","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"2625","article-title":"Long-term recurrent convolutional networks for visual recognition and description","author":"Donahue","year":"2015"},{"key":"10.1016\/j.knosys.2026.116374_b25","series-title":"Proceedings of the IEEE International Conference on Computer Vision","first-page":"4507","article-title":"Describing videos by exploiting temporal structure","author":"Yao","year":"2015"},{"key":"10.1016\/j.knosys.2026.116374_b26","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition","first-page":"8739","article-title":"End-to-end dense video captioning with masked transformer","author":"Zhou","year":"2018"},{"key":"10.1016\/j.knosys.2026.116374_b27","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"839","article-title":"Actor-transformers for group activity recognition","author":"Gavrilyuk","year":"2020"},{"key":"10.1016\/j.knosys.2026.116374_b28","series-title":"Icml","first-page":"4","article-title":"Is space-time attention all you need for video understanding?","author":"Bertasius","year":"2021"},{"key":"10.1016\/j.knosys.2026.116374_b29","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"6836","article-title":"Vivit: A video vision transformer","author":"Arnab","year":"2021"},{"key":"10.1016\/j.knosys.2026.116374_b30","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"7464","article-title":"Videobert: A joint model for video and language representation learning","author":"Sun","year":"2019"},{"key":"10.1016\/j.knosys.2026.116374_b31","series-title":"Violet: End-to-end video-language transformers with masked visual-token modeling","author":"Fu","year":"2021"},{"key":"10.1016\/j.knosys.2026.116374_b32","series-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"6545","article-title":"Fine-tuned clip models are efficient video learners","author":"Rasheed","year":"2023"},{"key":"10.1016\/j.knosys.2026.116374_b33","doi-asserted-by":"crossref","first-page":"8896","DOI":"10.1109\/TCSVT.2024.3389024","article-title":"Modality-aware heterogeneous graph for joint video moment retrieval and highlight detection","author":"Wang","year":"2024","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.knosys.2026.116374_b34","series-title":"2024 International Joint Conference on Neural Networks","first-page":"1","article-title":"Mh-detr: Video moment and highlight detection with cross-modal transformer","author":"Xu","year":"2024"},{"key":"10.1016\/j.knosys.2026.116374_b35","series-title":"Correlation-guided query-dependency calibration in video representation learning for temporal grounding","author":"Moon","year":"2023"},{"key":"10.1016\/j.knosys.2026.116374_b36","article-title":"Subtask prior-driven optimized mechanism on joint video moment retrieval and highlight detection","author":"Zhou","year":"2024","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.knosys.2026.116374_b37","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"24286","article-title":"Timeexpert: An expert-guided video llm for video temporal grounding","author":"Yang","year":"2025"},{"key":"10.1016\/j.knosys.2026.116374_b38","doi-asserted-by":"crossref","DOI":"10.1016\/j.engappai.2024.109660","article-title":"Multilingual entity alignment by abductive knowledge reasoning on multiple knowledge graphs","author":"Akhtar","year":"2025","journal-title":"Eng. Appl. Artif. Intell."},{"key":"10.1016\/j.knosys.2026.116374_b39","doi-asserted-by":"crossref","DOI":"10.1016\/j.knosys.2022.109494","article-title":"Entity alignment based on relational semantics augmentation for multilingual knowledge graphs","author":"Akhtar","year":"2022","journal-title":"Knowl.-Based Syst."},{"key":"10.1016\/j.knosys.2026.116374_b40","first-page":"9694","article-title":"Align before fuse: Vision and language representation learning with momentum distillation","author":"Li","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.knosys.2026.116374_b41","series-title":"Dab-detr: Dynamic anchor boxes are better queries for detr","author":"Liu","year":"2022"},{"key":"10.1016\/j.knosys.2026.116374_b42","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"13846","article-title":"Knowing where to focus: Event-aware transformer for video grounding","author":"Jang","year":"2023"},{"key":"10.1016\/j.knosys.2026.116374_b43","series-title":"Proceedings of the 27th ACM International Conference on Multimedia","first-page":"2332","article-title":"Sentence specified dynamic video thumbnail generation","author":"Yuan","year":"2019"},{"key":"10.1016\/j.knosys.2026.116374_b44","series-title":"European Conference on Computer Vision","first-page":"300","article-title":"Learning trailer moments in full-length movies with co-contrastive attention","author":"Wang","year":"2020"},{"key":"10.1016\/j.knosys.2026.116374_b45","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"7970","article-title":"Cross-category video highlight detection via set-based learning","author":"Xu","year":"2021"},{"key":"10.1016\/j.knosys.2026.116374_b46","series-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","first-page":"7950","article-title":"Temporal cue guided video highlight detection with low-rank audio-visual fusion","author":"Ye","year":"2021"},{"key":"10.1016\/j.knosys.2026.116374_b47","series-title":"Proceedings of the AAAI Conference on Artificial Intelligence","first-page":"2986","article-title":"Boundary proposal network for two-stage natural language video localization","author":"Xiao","year":"2021"},{"key":"10.1016\/j.knosys.2026.116374_b48","series-title":"Simvtp: Simple video text pre-training with masked autoencoders","author":"Ma","year":"2022"}],"container-title":["Knowledge-Based Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0950705126011007?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0950705126011007?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,26]],"date-time":"2026-07-26T13:34:46Z","timestamp":1785072886000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0950705126011007"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,9]]},"references-count":48,"alternative-id":["S0950705126011007"],"URL":"https:\/\/doi.org\/10.1016\/j.knosys.2026.116374","relation":{},"ISSN":["0950-7051"],"issn-type":[{"value":"0950-7051","type":"print"}],"subject":[],"published":{"date-parts":[[2026,9]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Fine-grained semantics-driven decoupling optimization for joint video moment retrieval and highlight detection","name":"articletitle","label":"Article Title"},{"value":"Knowledge-Based Systems","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.knosys.2026.116374","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"116374"}}