{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T20:20:20Z","timestamp":1776889220707,"version":"3.51.2"},"reference-count":22,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,4,6]],"date-time":"2025-04-06T00:00:00Z","timestamp":1743897600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,4,6]],"date-time":"2025-04-06T00:00:00Z","timestamp":1743897600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,4,6]]},"DOI":"10.1109\/icassp49660.2025.10888725","type":"proceedings-article","created":{"date-parts":[[2025,3,12]],"date-time":"2025-03-12T13:52:43Z","timestamp":1741787563000},"page":"1-5","source":"Crossref","is-referenced-by-count":1,"title":["Spatiotemporal-Aware Visual Captioning using Vision-Language Pre-Training Model"],"prefix":"10.1109","author":[{"given":"Shuai","family":"Wu","sequence":"first","affiliation":[{"name":"Fudan University,School of Computer Science,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Weidong","family":"Yang","sequence":"additional","affiliation":[{"name":"Fudan University,School of Computer Science,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shuyan","family":"Wu","sequence":"additional","affiliation":[{"name":"Xi&#x2019;an Jiaotong University,Faculty of Electronic and Information Engineering,Xi&#x2019;an,China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","article-title":"Minigpt-4: Enhancing vision-language understanding with advanced large language models","volume-title":"ArXiv","volume":"abs\/2304.10592","author":"Zhu","year":"2023"},{"key":"ref2","article-title":"Improved baselines with visual instruction tuning","volume-title":"ArXiv","volume":"abs\/2310.03744","author":"Liu","year":"2023"},{"key":"ref3","first-page":"10 714","article-title":"Vid2seq: Large-scale pretraining of a visual language model for dense video captioning","volume-title":"2023 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Yang"},{"key":"ref4","first-page":"17 938","article-title":"End-to-end generative pretraining for multimodal video captioning","volume-title":"2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Seo"},{"key":"ref5","article-title":"Visual instruction tuning","volume-title":"ArXiv","volume":"abs\/2304.08485","author":"Liu","year":"2023"},{"key":"ref6","doi-asserted-by":"crossref","DOI":"10.18653\/v1\/2024.acl-long.679","article-title":"Video-chatgpt: Towards detailed video understanding via large vision and language models","volume-title":"Annual Meeting of the Association for Computational Linguistics","author":"Maaz"},{"key":"ref7","article-title":"Attention is all you need","volume-title":"Neural Information Processing Systems","author":"Vaswani","year":"2017"},{"key":"ref8","article-title":"Learning transferable visual models from natural language supervision","volume-title":"International Conference on Machine Learning","author":"Radford"},{"key":"ref9","article-title":"Vicuna: An open-source chatbot impressing gpt-4 with 90%* chatgpt quality","author":"Chiang","year":"2023"},{"key":"ref10","article-title":"Microsoft coco captions: Data collection and evaluation server","volume-title":"ArXiv","volume":"abs\/1504.00325","author":"Chen","year":"2015"},{"key":"ref11","doi-asserted-by":"crossref","DOI":"10.18653\/v1\/P18-1238","article-title":"Conceptual captions: A cleaned, hypernymed, image alt-text dataset for automatic image captioning","volume-title":"Annual Meeting of the Association for Computational Linguistics","author":"Sharma"},{"key":"ref12","article-title":"Collecting highly parallel data for paraphrase evaluation","volume-title":"Annual Meeting of the Association for Computational Linguistics","author":"Chen"},{"key":"ref13","first-page":"5288","article-title":"Msr-vtt: A large video description dataset for bridging video and language","volume-title":"2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Xu"},{"key":"ref14","article-title":"Ego-exo4d: Understanding skilled human activity from first- and third-person perspectives","volume-title":"ArXiv","volume":"abs\/2311.18259","author":"Grauman","year":"2023"},{"key":"ref15","first-page":"8917","article-title":"Joint syntax representation learning and visual cue translation for video captioning","volume-title":"2019 IEEE\/CVF International Conference on Computer Vision (ICCV)","author":"Hou"},{"key":"ref16","doi-asserted-by":"crossref","DOI":"10.1007\/978-3-030-58548-8_20","article-title":"Learning modality interaction for temporal sentence localization and event captioning in videos","volume-title":"European Conference on Computer Vision","author":"Chen"},{"key":"ref17","first-page":"13 275","article-title":"Object relational graph with teacher-recommended learning for video captioning","volume-title":"2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Zhang"},{"key":"ref18","first-page":"17 928","article-title":"Swinbert: End-to-end transformers with sparse attention for video captioning","volume-title":"2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Lin"},{"key":"ref19","article-title":"Chat-univi: Unified visual representation empowers large language models with image and video understanding","volume-title":"ArXiv","volume":"abs\/2311.08046","author":"Jin","year":"2023"},{"key":"ref20","article-title":"Delving deeper into the decoder for video captioning","volume-title":"European Conference on Artificial Intelligence","author":"Chen"},{"key":"ref21","doi-asserted-by":"crossref","DOI":"10.18653\/v1\/2021.naacl-main.193","article-title":"Decembert: Learning from noisy instructional videos via dense captions and entropy minimization","volume-title":"North American Chapter of the Association for Computational Linguistics","author":"Tang","year":"2021"},{"key":"ref22","article-title":"Roformer: Enhanced transformer with rotary position embedding","volume-title":"ArXiv","volume":"abs\/2104.09864","author":"Su","year":"2021"}],"event":{"name":"ICASSP 2025 - 2025 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","location":"Hyderabad, India","start":{"date-parts":[[2025,4,6]]},"end":{"date-parts":[[2025,4,11]]}},"container-title":["ICASSP 2025 - 2025 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10887540\/10887541\/10888725.pdf?arnumber=10888725","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,25]],"date-time":"2026-03-25T05:23:31Z","timestamp":1774416211000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10888725\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,4,6]]},"references-count":22,"URL":"https:\/\/doi.org\/10.1109\/icassp49660.2025.10888725","relation":{},"subject":[],"published":{"date-parts":[[2025,4,6]]}}}