{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,2]],"date-time":"2025-08-02T17:04:34Z","timestamp":1754154274858,"version":"3.41.2"},"reference-count":28,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,12,13]],"date-time":"2024-12-13T00:00:00Z","timestamp":1734048000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,12,13]],"date-time":"2024-12-13T00:00:00Z","timestamp":1734048000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100006190","name":"Research and Development","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100006190","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,12,13]]},"DOI":"10.1109\/hpcc64274.2024.00023","type":"proceedings-article","created":{"date-parts":[[2025,7,23]],"date-time":"2025-07-23T18:33:48Z","timestamp":1753295628000},"page":"94-101","source":"Crossref","is-referenced-by-count":0,"title":["End-to-End Dense Video Captioning Model Based on Multimodal Feature Fusion"],"prefix":"10.1109","author":[{"given":"Shixin","family":"Peng","sequence":"first","affiliation":[{"name":"Central China Normal University,Wuhan,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ting","family":"Xiong","sequence":"additional","affiliation":[{"name":"Central China Normal University,Wuhan,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jingying","family":"Chen","sequence":"additional","affiliation":[{"name":"Central China Normal University,Wuhan,China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"year":"2023","key":"ref1","article-title":"FFmpeg Software Suite"},{"key":"ref2","article-title":"YouTube Data API Video Captions"},{"key":"ref3","article-title":"Neural machine translation by jointly learning to align and translate","author":"Bahdanau","year":"2014","journal-title":"Computer Science"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.675"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"article-title":"Bert: Pre-training of deep bidirectional transformers for language understanding","year":"2018","author":"Devlin","key":"ref6"},{"key":"ref7","article-title":"Cnn architectures for large-scale audio classification","author":"Hershey","year":"2016","journal-title":"IEEE"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/K19-1039"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.83"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW50498.2020.00487"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00782"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00782"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/TNET.2024.3375108"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00675"},{"key":"ref15","article-title":"Algorithms for the assignment and transportation problems","volume":"10","author":"Munkres","year":"1962","journal-title":"SIAM. J"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00900"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00633"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.510"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46484-8_2"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00677"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2019.2934681"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1007\/BF00992696"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01032"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00852"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/IWQoS49365.2020.9213038"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00911"},{"article-title":"Anomalyclip: Object-agnostic prompt learning for zero-shot anomaly detection","year":"2024","author":"Zhou","key":"ref27"},{"article-title":"Deformable detr: Deformable transformers for end-to-end object detection","volume-title":"International Conference on Learning Representations","author":"Zhu","key":"ref28"}],"event":{"name":"2024 IEEE International Conference on High Performance Computing and Communications (HPCC)","start":{"date-parts":[[2024,12,13]]},"location":"Wuhan, China","end":{"date-parts":[[2024,12,15]]}},"container-title":["2024 IEEE International Conference on High Performance Computing and Communications (HPCC)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11083115\/11083125\/11083301.pdf?arnumber=11083301","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,24]],"date-time":"2025-07-24T04:50:11Z","timestamp":1753332611000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11083301\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,13]]},"references-count":28,"URL":"https:\/\/doi.org\/10.1109\/hpcc64274.2024.00023","relation":{},"subject":[],"published":{"date-parts":[[2024,12,13]]}}}