{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T12:05:36Z","timestamp":1784203536183,"version":"3.55.0"},"reference-count":54,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62576149"],"award-info":[{"award-number":["62576149"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100004032","name":"Jilin University","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100004032","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Neural Networks"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.neunet.2026.109063","type":"journal-article","created":{"date-parts":[[2026,4,30]],"date-time":"2026-04-30T23:08:01Z","timestamp":1777590481000},"page":"109063","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["TDP-DETR: Temporal dynamics perception framework for video moment retrieval and highlight detection"],"prefix":"10.1016","volume":"202","author":[{"given":"Huilin","family":"An","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zefan","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shijie","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kehua","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8060-4725","authenticated-orcid":false,"given":"Tian","family":"Bai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.neunet.2026.109063_bib0001","unstructured":"Ahmadian, M., Guerin, F., & Gilbert, A. (2023). Mofo: Motion focused self-supervision for video understanding. arXiv: 2308.12447."},{"key":"10.1016\/j.neunet.2026.109063_bib0002","doi-asserted-by":"crossref","unstructured":"Apostolidis, E., Adamantidou, E., Metsai, A. I., Mezaris, V., & Patras, I. (2021). Video summarization using deep neural networks: A survey. arXiv: 2101.06072.","DOI":"10.1109\/JPROC.2021.3117472"},{"key":"10.1016\/j.neunet.2026.109063_bib0003","doi-asserted-by":"crossref","first-page":"3799","DOI":"10.1109\/TMM.2023.3316025","article-title":"Cross-modality knowledge calibration network for video corpus moment retrieval","volume":"26","author":"Chen","year":"2024","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.neunet.2026.109063_bib0004","unstructured":"Chung, J., Gulcehre, C., Cho, K., & Bengio, Y. (2014). Empirical evaluation of gated recurrent neural networks on sequence modeling. arXiv: 1412.3555."},{"issue":"2","key":"10.1016\/j.neunet.2026.109063_bib0005","doi-asserted-by":"crossref","first-page":"1735","DOI":"10.1609\/aaai.v38i2.27941","article-title":"Wer steps, better performance: Efficient cross-modal clip trimming for video moment retrieval using language","volume":"38","author":"Fang","year":"2024","journal-title":"Proceedings of the AAAI Conference on Artificial Intelligence"},{"key":"10.1016\/j.neunet.2026.109063_bib0006","doi-asserted-by":"crossref","first-page":"7517","DOI":"10.1109\/TMM.2022.3222965","article-title":"Multi-modal cross-domain alignment network for video moment retrieval","volume":"25","author":"Fang","year":"2023","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.neunet.2026.109063_bib0007","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"6202","article-title":"Slowfast networks for video recognition","author":"Feichtenhofer","year":"2019"},{"key":"10.1016\/j.neunet.2026.109063_bib0008","series-title":"Proceedings of the IEEE international conference on computer vision","first-page":"5267","article-title":"Tall: Temporal activity localization via language query","author":"Gao","year":"2017"},{"key":"10.1016\/j.neunet.2026.109063_bib0009","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"3621","article-title":"Fast convergence of detr with spatially modulated co-attention","author":"Gao","year":"2021"},{"key":"10.1016\/j.neunet.2026.109063_bib0010","series-title":"Proceedings of the IEEE\/CVF winter conference on applications of computer vision","first-page":"907","article-title":"Saliency-guided detr for moment retrieval and highlight detection","author":"Gordeev","year":"2026"},{"key":"10.1016\/j.neunet.2026.109063_bib0011","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"897","article-title":"Ranking info noise contrastive estimation: Boosting contrastive learning via ranked positives","volume":"36","author":"Hoffmann","year":"2022"},{"key":"10.1016\/j.neunet.2026.109063_bib0012","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"13493","article-title":"Mgmae: Motion guided masking for video masked autoencoding","author":"Huang","year":"2023"},{"key":"10.1016\/j.neunet.2026.109063_bib0013","series-title":"Proceedings of the IEEE\/CVF winter conference on applications of computer vision","first-page":"6783","article-title":"Semantic fusion augmentation and semantic boundary detection: A novel approach to multi-target video moment retrieval","author":"Huang","year":"2024"},{"key":"10.1016\/j.neunet.2026.109063_bib0014","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"13846","article-title":"Knowing where to focus: Event-aware transformer for video grounding","author":"Jang","year":"2023"},{"key":"10.1016\/j.neunet.2026.109063_bib0015","doi-asserted-by":"crossref","first-page":"9657","DOI":"10.1109\/TMM.2024.3396272","article-title":"Zero-shot video moment retrieval with angular reconstructive text embeddings","volume":"26","author":"Jiang","year":"2024","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.neunet.2026.109063_bib0016","series-title":"Proceedings of the 32nd ACM international conference on multimedia","first-page":"7249","article-title":"Prior knowledge integration via llm encoding and pseudo event regulation for video moment retrieval","author":"Jiang","year":"2024"},{"key":"10.1016\/j.neunet.2026.109063_bib0017","series-title":"Proceedings of the 61st annual meeting of the association for computational linguistics (volume 1: Long papers)","first-page":"1387","article-title":"Ms-detr: Natural language video localization with sampling moment-moment interaction","author":"Jing","year":"2023"},{"key":"10.1016\/j.neunet.2026.109063_bib0018","doi-asserted-by":"crossref","first-page":"2880","DOI":"10.1109\/TASLP.2020.3030497","article-title":"Panns: Large-scale pretrained audio neural networks for audio pattern recognition","volume":"28","author":"Kong","year":"2020","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"key":"10.1016\/j.neunet.2026.109063_bib0019","series-title":"European conference on computer vision","first-page":"220","article-title":"Bam-detr: Boundary-aligned moment detection transformer for temporal sentence grounding in videos","author":"Lee","year":"2024"},{"key":"10.1016\/j.neunet.2026.109063_bib0020","first-page":"11846","article-title":"Detecting moments and highlights in videos via natural language queries","volume":"34","author":"Lei","year":"2021","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.neunet.2026.109063_bib0021","doi-asserted-by":"crossref","first-page":"65948","DOI":"10.52202\/075280-2880","article-title":"Momentdiff: Generative video moment retrieval from random to real","volume":"36","author":"Li","year":"2023","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.neunet.2026.109063_bib0022","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"2794","article-title":"Univtg: Towards unified video-language temporal grounding","author":"Lin","year":"2023"},{"key":"10.1016\/j.neunet.2026.109063_bib0023","series-title":"Proceedings of the 31st ACM international conference on multimedia","first-page":"4190","article-title":"Filling the information gap between video and query for language-driven moment retrieval","author":"Liu","year":"2023"},{"key":"10.1016\/j.neunet.2026.109063_bib0024","doi-asserted-by":"crossref","first-page":"7155","DOI":"10.1109\/TCSVT.2025.3542081","article-title":"What and where: Semantic grasping and contextual scanning for moment retrieval and highlight detection","volume":"35","author":"Liu","year":"2025","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.neunet.2026.109063_bib0025","series-title":"The 41st international ACM SIGIR conference on research & development in information retrieval","first-page":"15","article-title":"Attentive moment retrieval in videos","author":"Liu","year":"2018"},{"key":"10.1016\/j.neunet.2026.109063_bib0026","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"3042","article-title":"Umt: Unified multi-modal transformers for joint video moment retrieval and highlight detection","author":"Liu","year":"2022"},{"key":"10.1016\/j.neunet.2026.109063_bib0027","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"3855","article-title":"Towards balanced alignment: Modal-enhanced semantic modeling for video moment retrieval","volume":"38","author":"Liu","year":"2024"},{"key":"10.1016\/j.neunet.2026.109063_bib0028","series-title":"Proceedings of the 33rd ACM international conference on multimedia","first-page":"4514","article-title":"Ms-detr: Towards effective video moment retrieval and highlight detection by joint motion-semantic learning","author":"Ma","year":"2025"},{"key":"10.1016\/j.neunet.2026.109063_bib0029","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"2798","article-title":"Llavilo: Boosting video moment retrieval via adapter-based multimodal modeling","author":"Ma","year":"2023"},{"key":"10.1016\/j.neunet.2026.109063_bib0030","series-title":"European conference on computer vision","first-page":"76","article-title":"Ea-vtr: Event-aware video-text retrieval","author":"Ma","year":"2024"},{"key":"10.1016\/j.neunet.2026.109063_bib0031","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"23023","article-title":"Query-dependent video representation for moment retrieval and highlight detection","author":"Moon","year":"2023"},{"key":"10.1016\/j.neunet.2026.109063_bib0032","series-title":"International conference on machine learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.neunet.2026.109063_bib0033","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"6684","article-title":"Cdtr: Semantic alignment for video moment retrieval using concept decomposition transformer","volume":"39","author":"Ran","year":"2025"},{"key":"10.1016\/j.neunet.2026.109063_bib0034","doi-asserted-by":"crossref","first-page":"25","DOI":"10.1162\/tacl_a_00207","article-title":"Grounding action descriptions in videos","volume":"1","author":"Regneri","year":"2013","journal-title":"Transactions of the Association for Computational Linguistics"},{"key":"10.1016\/j.neunet.2026.109063_bib0035","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"658","article-title":"Generalized intersection over union: A metric and a loss for bounding box regression","author":"Rezatofighi","year":"2019"},{"key":"10.1016\/j.neunet.2026.109063_bib0036","series-title":"Proceedings of the 31st ACM international conference on multimedia","first-page":"4109","article-title":"Semantics-enriched cross-modal alignment for complex-query video moment retrieval","author":"Shen","year":"2023"},{"issue":"4","key":"10.1016\/j.neunet.2026.109063_bib0037","doi-asserted-by":"crossref","first-page":"215","DOI":"10.1561\/1500000014","article-title":"Concept-based video retrieval","volume":"2","author":"Snoek","year":"2009","journal-title":"Foundations and Trends\u00ae in Information Retrieval"},{"key":"10.1016\/j.neunet.2026.109063_bib0038","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"3224","article-title":"Vlg-Net: Video-language graph matching network for video grounding","author":"Soldan","year":"2021"},{"key":"10.1016\/j.neunet.2026.109063_bib0039","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"4998","article-title":"Tr-detr: Task-reciprocal transformer for joint moment retrieval and highlight detection","volume":"38","author":"Sun","year":"2024"},{"key":"10.1016\/j.neunet.2026.109063_bib0040","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"7131","article-title":"Diversifying query: Region-guided transformer for temporal sentence grounding","volume":"39","author":"Sun","year":"2025"},{"key":"10.1016\/j.neunet.2026.109063_bib0041","series-title":"Proceedings of the computer vision and pattern recognition conference","first-page":"8438","article-title":"Smile: Infusing spatial and motion semantics in masked video learning","author":"Thoker","year":"2025"},{"key":"10.1016\/j.neunet.2026.109063_bib0042","doi-asserted-by":"crossref","first-page":"11044","DOI":"10.1109\/TMM.2024.3443672","article-title":"Gist, content, target-oriented: A 3-level human-like framework for video moment retrieval","volume":"26","author":"Wang","year":"2024","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.neunet.2026.109063_bib0043","series-title":"Proceedings of the 30th ACM SIGKDD conference on knowledge discovery and data mining","first-page":"3024","article-title":"Routing evidence for unseen actions in video moment retrieval","author":"Wang","year":"2024"},{"key":"10.1016\/j.neunet.2026.109063_bib0044","doi-asserted-by":"crossref","first-page":"3921","DOI":"10.1109\/TMM.2022.3168424","article-title":"Siamese alignment network for weakly supervised video moment retrieval","volume":"25","author":"Wang","year":"2023","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.neunet.2026.109063_bib0045","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"2986","article-title":"Boundary proposal network for two-stage natural language video localization","volume":"35","author":"Xiao","year":"2021"},{"key":"10.1016\/j.neunet.2026.109063_bib0046","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"18709","article-title":"Bridging the gap: A unified video comprehension framework for moment retrieval and highlight detection","author":"Xiao","year":"2024"},{"key":"10.1016\/j.neunet.2026.109063_bib0047","series-title":"2024\u202fIEEE International conference on multimedia and expo (ICME)","first-page":"1","article-title":"Multi-modal fusion and query refinement network for video moment retrieval and highlight detection","author":"Xu","year":"2024"},{"key":"10.1016\/j.neunet.2026.109063_bib0048","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"13623","article-title":"Unloc: A unified framework for video localization tasks","author":"Yan","year":"2023"},{"key":"10.1016\/j.neunet.2026.109063_bib0049","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"18308","article-title":"Task-driven exploration: Decoupling and inter-task feedback for joint moment retrieval and highlight detection","author":"Yang","year":"2024"},{"issue":"2","key":"10.1016\/j.neunet.2026.109063_bib0050","doi-asserted-by":"crossref","first-page":"34","DOI":"10.1007\/s00138-026-01799-9","article-title":"Guided by structure: Boundary-aware modeling for moment retrieval and highlight detection","volume":"37","author":"Yu","year":"2026","journal-title":"Machine Vision and Applications"},{"key":"10.1016\/j.neunet.2026.109063_bib0051","first-page":"1","article-title":"Resilient semantic pseudo-text embedding for zero-shot video moment retrieval","volume":"22","author":"Zhang","year":"2026","journal-title":"ACM Transactions on Multimedia Computing, Communications and Applications"},{"key":"10.1016\/j.neunet.2026.109063_bib0052","doi-asserted-by":"crossref","first-page":"14522","DOI":"10.1109\/TNNLS.2024.3516033","article-title":"DiffusionVMR: Diffusion model for joint video moment retrieval and highlight detection","volume":"36","author":"Zhao","year":"2024","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"10.1016\/j.neunet.2026.109063_bib0053","unstructured":"Zhao, P., He, Z., Zhang, F., Lin, S., & Zhou, F. (2025). Ld-detr: Loop decoder detection transformer for video moment retrieval and highlight detection. arXiv: 2501.10787."},{"key":"10.1016\/j.neunet.2026.109063_bib0054","doi-asserted-by":"crossref","unstructured":"Zhou, X., Wei, F., Duan, L., & Li, W. (2025). The devil is in the spurious correlation: Boosting moment retrieval via temporal dynamic learning. arXiv: 2501.07305.","DOI":"10.1109\/ICCV51701.2025.01950"}],"container-title":["Neural Networks"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S089360802600523X?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S089360802600523X?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T11:21:58Z","timestamp":1784200918000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S089360802600523X"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":54,"alternative-id":["S089360802600523X"],"URL":"https:\/\/doi.org\/10.1016\/j.neunet.2026.109063","relation":{},"ISSN":["0893-6080"],"issn-type":[{"value":"0893-6080","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"TDP-DETR: Temporal dynamics perception framework for video moment retrieval and highlight detection","name":"articletitle","label":"Article Title"},{"value":"Neural Networks","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.neunet.2026.109063","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"109063"}}