{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T21:19:28Z","timestamp":1783113568703,"version":"3.54.6"},"reference-count":52,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62125305"],"award-info":[{"award-number":["62125305"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62503384"],"award-info":[{"award-number":["62503384"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62572384"],"award-info":[{"award-number":["62572384"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U24A20325"],"award-info":[{"award-number":["U24A20325"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100015401","name":"Key Research and Development Projects of Shaanxi Province","doi-asserted-by":"publisher","award":["2024PT-ZCK-80"],"award-info":[{"award-number":["2024PT-ZCK-80"]}],"id":[{"id":"10.13039\/501100015401","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100018537","name":"National Science and Technology Major Project","doi-asserted-by":"publisher","award":["2023ZD0121300"],"award-info":[{"award-number":["2023ZD0121300"]}],"id":[{"id":"10.13039\/501100018537","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition"],"published-print":{"date-parts":[[2026,11]]},"DOI":"10.1016\/j.patcog.2026.113872","type":"journal-article","created":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T00:13:17Z","timestamp":1777421597000},"page":"113872","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PD","title":["PR-DETR: Injecting position and relation prior for dense video captioning"],"prefix":"10.1016","volume":"179","author":[{"given":"Yizhe","family":"Li","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4100-2304","authenticated-orcid":false,"given":"Sanping","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zheng","family":"Qin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Le","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.patcog.2026.113872_b1","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.111138","article-title":"Fully exploring object relation interaction and hidden state attention for video captioning","volume":"159","author":"Yuan","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113872_b2","doi-asserted-by":"crossref","unstructured":"W. Pei, J. Zhang, X. Wang, L. Ke, X. Shen, Y.-W. Tai, Memory-attended recurrent network for video captioning, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2019, pp. 8347\u20138356.","DOI":"10.1109\/CVPR.2019.00854"},{"key":"10.1016\/j.patcog.2026.113872_b3","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2025.112192","article-title":"Dual-hierarchical knowledge distillation for video captioning","volume":"171","author":"Luo","year":"2026","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113872_b4","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2026.113105","article-title":"Ask and focus more: Question-prompt uncertainty allocation for dual-controllable video captioning","volume":"175","author":"Chen","year":"2026","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113872_b5","doi-asserted-by":"crossref","unstructured":"R. Krishna, K. Hata, F. Ren, L. Fei-Fei, J. Carlos Niebles, Dense-captioning events in videos, in: Proceedings of the IEEE International Conference on Computer Vision, 2017, pp. 706\u2013715.","DOI":"10.1109\/ICCV.2017.83"},{"key":"10.1016\/j.patcog.2026.113872_b6","doi-asserted-by":"crossref","unstructured":"Y. Li, T. Yao, Y. Pan, H. Chao, T. Mei, Jointly localizing and describing events for dense video captioning, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2018, pp. 7492\u20137500.","DOI":"10.1109\/CVPR.2018.00782"},{"key":"10.1016\/j.patcog.2026.113872_b7","doi-asserted-by":"crossref","unstructured":"T. Wang, R. Zhang, Z. Lu, F. Zheng, R. Cheng, P. Luo, End-to-End Dense Video Captioning with Parallel Decoding, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 6847\u20136857.","DOI":"10.1109\/ICCV48922.2021.00677"},{"key":"10.1016\/j.patcog.2026.113872_b8","doi-asserted-by":"crossref","unstructured":"A. Yang, A. Nagrani, P.H. Seo, A. Miech, J. Pont-Tuset, I. Laptev, J. Sivic, C. Schmid, Vid2Seq: Large-Scale Pretraining of a Visual Language Model for Dense Video Captioning, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023.","DOI":"10.1109\/CVPR52729.2023.01032"},{"key":"10.1016\/j.patcog.2026.113872_b9","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110754","article-title":"Text-guided distillation learning to diversify video embeddings for text-video retrieval","volume":"156","author":"Lee","year":"2024","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113872_b10","doi-asserted-by":"crossref","DOI":"10.1109\/TIP.2024.3514352","article-title":"Exploiting unlabeled videos for video-text retrieval via pseudo-supervised learning","author":"Lu","year":"2024","journal-title":"IEEE Trans. Image Process."},{"key":"10.1016\/j.patcog.2026.113872_b11","doi-asserted-by":"crossref","unstructured":"M.M. Islam, N. Ho, X. Yang, T. Nagarajan, L. Torresani, G. Bertasius, Video recap: Recursive captioning of hour-long videos, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 18198\u201318208.","DOI":"10.1109\/CVPR52733.2024.01723"},{"key":"10.1016\/j.patcog.2026.113872_b12","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2025.111642","article-title":"Scene-enhanced multi-scale temporal aware network for video moment retrieval","volume":"165","author":"Wang","year":"2025","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113872_b13","doi-asserted-by":"crossref","unstructured":"V. Iashin, E. Rahtu, Multi-modal dense video captioning, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshops, 2020, pp. 958\u2013959.","DOI":"10.1109\/CVPRW50498.2020.00487"},{"key":"10.1016\/j.patcog.2026.113872_b14","series-title":"2018 25th IEEE International Conference on Image Processing","first-page":"1288","article-title":"Hierarchical context encoding for events captioning in videos","author":"Yang","year":"2018"},{"key":"10.1016\/j.patcog.2026.113872_b15","series-title":"Proceedings of the European Conference on Computer Vision","first-page":"213","article-title":"End-to-end object detection with transformers","author":"Carion","year":"2020"},{"key":"10.1016\/j.patcog.2026.113872_b16","doi-asserted-by":"crossref","unstructured":"M. Kim, H.B. Kim, J. Moon, J. Choi, S.T. Kim, Do You Remember? Dense Video Captioning with Cross-Modal Memory Retrieval, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 13894\u201313904.","DOI":"10.1109\/CVPR52733.2024.01318"},{"key":"10.1016\/j.patcog.2026.113872_b17","doi-asserted-by":"crossref","unstructured":"H. Wu, H. Liu, Y. Qiao, X. Sun, DIBS: Enhancing Dense Video Captioning with Unlabeled Videos via Pseudo Boundary Enrichment and Online Refinement, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 18699\u201318708.","DOI":"10.1109\/CVPR52733.2024.01769"},{"key":"10.1016\/j.patcog.2026.113872_b18","article-title":"Towards automatic learning of procedures from web instructional videos","volume":"vol. 32","author":"Zhou","year":"2018"},{"key":"10.1016\/j.patcog.2026.113872_b19","doi-asserted-by":"crossref","unstructured":"J. Wang, W. Jiang, L. Ma, W. Liu, Y. Xu, Bidirectional attentive fusion with context gating for dense video captioning, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2018, pp. 7190\u20137198.","DOI":"10.1109\/CVPR.2018.00751"},{"key":"10.1016\/j.patcog.2026.113872_b20","doi-asserted-by":"crossref","unstructured":"C. Deng, S. Chen, D. Chen, Y. He, Q. Wu, Sketch, Ground, and Refine: Top-Down Dense Video Captioning, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2021, pp. 234\u2013243.","DOI":"10.1109\/CVPR46437.2021.00030"},{"key":"10.1016\/j.patcog.2026.113872_b21","doi-asserted-by":"crossref","unstructured":"S. Chen, Y.-G. Jiang, Towards bridging event captioner and sentence localizer for weakly supervised dense event captioning, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2021, pp. 8425\u20138435.","DOI":"10.1109\/CVPR46437.2021.00832"},{"key":"10.1016\/j.patcog.2026.113872_b22","doi-asserted-by":"crossref","unstructured":"J. Mun, L. Yang, Z. Ren, N. Xu, B. Han, Streamlined dense video captioning, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2019, pp. 6588\u20136597.","DOI":"10.1109\/CVPR.2019.00675"},{"key":"10.1016\/j.patcog.2026.113872_b23","first-page":"9459","article-title":"Retrieval-augmented generation for knowledge-intensive nlp tasks","volume":"33","author":"Lewis","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.113872_b24","doi-asserted-by":"crossref","unstructured":"X. Zhou, A. Arnab, S. Buch, S. Yan, A. Myers, X. Xiong, A. Nagrani, C. Schmid, Streaming dense video captioning, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 18243\u201318252.","DOI":"10.1109\/CVPR52733.2024.01727"},{"key":"10.1016\/j.patcog.2026.113872_b25","doi-asserted-by":"crossref","unstructured":"K. Wu, P. Li, J. Fu, Y. Li, Y. Wu, Y. Liu, J. Wang, S. Zhou, Event-Equalized Dense Video Captioning, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2025, pp. 8417\u20138427.","DOI":"10.1109\/CVPR52734.2025.00788"},{"key":"10.1016\/j.patcog.2026.113872_b26","unstructured":"A. Dosovitskiy, L. Beyer, A. Kolesnikov, D. Weissenborn, X. Zhai, T. Unterthiner, M. Dehghani, M. Minderer, G. Heigold, S. Gelly, et al., An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale, in: International Conference on Learning Representations, 2021."},{"key":"10.1016\/j.patcog.2026.113872_b27","article-title":"Visual-linguistic feature alignment with semantic and kinematic guidance for referring multi-object tracking","author":"Li","year":"2025","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.patcog.2026.113872_b28","doi-asserted-by":"crossref","unstructured":"S. Huang, Z. Lu, X. Cun, Y. Yu, X. Zhou, X. Shen, Deim: Detr with improved matching for fast convergence, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition Conference, 2025, pp. 15162\u201315171.","DOI":"10.1109\/CVPR52734.2025.01412"},{"key":"10.1016\/j.patcog.2026.113872_b29","unstructured":"X. Zhu, W. Su, L. Lu, B. Li, X. Wang, J. Dai, Deformable DETR: Deformable Transformers for End-to-End Object Detection, in: International Conference on Learning Representations, 2021."},{"key":"10.1016\/j.patcog.2026.113872_b30","first-page":"4293","article-title":"HiCM2: Hierarchical compact memory modeling for dense video captioning","volume":"vol. 39","author":"Kim","year":"2025"},{"key":"10.1016\/j.patcog.2026.113872_b31","doi-asserted-by":"crossref","unstructured":"F.C. Heilbron, J.C. Niebles, B. Ghanem, Fast temporal activity proposals for efficient detection of human actions in untrimmed videos, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2016, pp. 1914\u20131923.","DOI":"10.1109\/CVPR.2016.211"},{"key":"10.1016\/j.patcog.2026.113872_b32","doi-asserted-by":"crossref","unstructured":"J. Gao, Z. Yang, K. Chen, C. Sun, R. Nevatia, Turn tap: Temporal unit regression network for temporal action proposals, in: Proceedings of the IEEE International Conference on Computer Vision, 2017, pp. 3628\u20133636.","DOI":"10.1109\/ICCV.2017.392"},{"key":"10.1016\/j.patcog.2026.113872_b33","doi-asserted-by":"crossref","unstructured":"T. Lin, X. Liu, X. Li, E. Ding, S. Wen, Bmn: Boundary-matching network for temporal action proposal generation, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2019, pp. 3889\u20133898.","DOI":"10.1109\/ICCV.2019.00399"},{"key":"10.1016\/j.patcog.2026.113872_b34","doi-asserted-by":"crossref","unstructured":"T. Lin, X. Zhao, H. Su, C. Wang, M. Yang, Bsn: Boundary sensitive network for temporal action proposal generation, in: Proceedings of the European Conference on Computer Vision, 2018, pp. 3\u201319.","DOI":"10.1007\/978-3-030-01225-0_1"},{"key":"10.1016\/j.patcog.2026.113872_b35","doi-asserted-by":"crossref","unstructured":"R. Zeng, H. Xu, W. Huang, P. Chen, M. Tan, C. Gan, Dense regression network for video grounding, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2020, pp. 10287\u201310296.","DOI":"10.1109\/CVPR42600.2020.01030"},{"key":"10.1016\/j.patcog.2026.113872_b36","doi-asserted-by":"crossref","unstructured":"M. Soldan, M. Xu, S. Qu, J. Tegner, B. Ghanem, Vlg-net: Video-language graph matching network for video grounding, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 3224\u20133234.","DOI":"10.1109\/ICCVW54120.2021.00361"},{"key":"10.1016\/j.patcog.2026.113872_b37","doi-asserted-by":"crossref","unstructured":"J. Devlin, M.-W. Chang, K. Lee, K. Toutanova, Bert: Pre-training of deep bidirectional transformers for language understanding, in: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers), 2019, pp. 4171\u20134186.","DOI":"10.18653\/v1\/N19-1423"},{"key":"10.1016\/j.patcog.2026.113872_b38","series-title":"International Conference on Machine Learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.patcog.2026.113872_b39","article-title":"Attention is all you need","volume":"30","author":"Vaswani","year":"2017","journal-title":"Adv. Neural Inf. Process. Syst."},{"issue":"1\u20132","key":"10.1016\/j.patcog.2026.113872_b40","doi-asserted-by":"crossref","first-page":"83","DOI":"10.1002\/nav.3800020109","article-title":"The hungarian method for the assignment problem","volume":"2","author":"Kuhn","year":"1955","journal-title":"Nav. Res. Logist. Q."},{"key":"10.1016\/j.patcog.2026.113872_b41","doi-asserted-by":"crossref","unstructured":"H. Rezatofighi, N. Tsoi, J. Gwak, A. Sadeghian, I. Reid, S. Savarese, Generalized intersection over union: A metric and a loss for bounding box regression, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2019, pp. 658\u2013666.","DOI":"10.1109\/CVPR.2019.00075"},{"key":"10.1016\/j.patcog.2026.113872_b42","unstructured":"T.-Y. Ross, G. Doll\u00e1r, Focal loss for dense object detection, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2017, pp. 2980\u20132988."},{"key":"10.1016\/j.patcog.2026.113872_b43","doi-asserted-by":"crossref","unstructured":"H. Zhang, X. Li, L. Bing, Video-llama: An instruction-tuned audio-visual language model for video understanding, in: Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing: System Demonstrations, 2023, pp. 543\u2013553.","DOI":"10.18653\/v1\/2023.emnlp-demo.49"},{"issue":"10","key":"10.1016\/j.patcog.2026.113872_b44","doi-asserted-by":"crossref","DOI":"10.1007\/s11432-024-4321-9","article-title":"Videochat: Chat-centric video understanding","volume":"68","author":"Li","year":"2025","journal-title":"Sci. China Inf. Sci."},{"key":"10.1016\/j.patcog.2026.113872_b45","doi-asserted-by":"crossref","unstructured":"S. Ren, L. Yao, S. Li, X. Sun, L. Hou, Timechat: A time-sensitive multimodal large language model for long video understanding, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 14313\u201314323.","DOI":"10.1109\/CVPR52733.2024.01357"},{"issue":"5","key":"10.1016\/j.patcog.2026.113872_b46","doi-asserted-by":"crossref","first-page":"1890","DOI":"10.1109\/TCSVT.2020.3014606","article-title":"Event-centric hierarchical representation for dense video captioning","volume":"31","author":"Wang","year":"2020","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"10.1016\/j.patcog.2026.113872_b47","doi-asserted-by":"crossref","unstructured":"L. Zhou, Y. Zhou, J.J. Corso, R. Socher, C. Xiong, End-to-end dense video captioning with masked transformer, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2018, pp. 8739\u20138748.","DOI":"10.1109\/CVPR.2018.00911"},{"key":"10.1016\/j.patcog.2026.113872_b48","doi-asserted-by":"crossref","unstructured":"R. Vedantam, C. Lawrence Zitnick, D. Parikh, Cider: Consensus-based image description evaluation, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2015, pp. 4566\u20134575.","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"10.1016\/j.patcog.2026.113872_b49","doi-asserted-by":"crossref","unstructured":"K. Papineni, S. Roukos, T. Ward, W.-J. Zhu, Bleu: a method for automatic evaluation of machine translation, in: Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics, 2002, pp. 311\u2013318.","DOI":"10.3115\/1073083.1073135"},{"key":"10.1016\/j.patcog.2026.113872_b50","unstructured":"S. Banerjee, A. Lavie, METEOR: An automatic metric for MT evaluation with improved correlation with human judgments, in: Proceedings of the ACL Workshop on Intrinsic and Extrinsic Evaluation Measures for Machine Translation and\/Or Summarization, 2005, pp. 65\u201372."},{"key":"10.1016\/j.patcog.2026.113872_b51","series-title":"Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part VI 16","first-page":"517","article-title":"SODA: Story oriented dense video captioning evaluation framework","author":"Fujita","year":"2020"},{"key":"10.1016\/j.patcog.2026.113872_b52","doi-asserted-by":"crossref","unstructured":"Y. Xiong, B. Dai, D. Lin, Move forward and tell: A progressive generator of video descriptions, in: Proceedings of the European Conference on Computer Vision, 2018, pp. 468\u2013483.","DOI":"10.1007\/978-3-030-01252-6_29"}],"container-title":["Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S003132032600837X?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S003132032600837X?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T20:37:38Z","timestamp":1783111058000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S003132032600837X"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,11]]},"references-count":52,"alternative-id":["S003132032600837X"],"URL":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113872","relation":{},"ISSN":["0031-3203"],"issn-type":[{"value":"0031-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,11]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"PR-DETR: Injecting position and relation prior for dense video captioning","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113872","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Published by Elsevier Ltd.","name":"copyright","label":"Copyright"}],"article-number":"113872"}}