{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T06:19:37Z","timestamp":1778048377737,"version":"3.51.4"},"reference-count":47,"publisher":"IEEE","license":[{"start":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T00:00:00Z","timestamp":1772755200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T00:00:00Z","timestamp":1772755200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001321","name":"National Research Foundation","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001321","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026,3,6]]},"DOI":"10.1109\/wacv61042.2026.00738","type":"proceedings-article","created":{"date-parts":[[2026,5,5]],"date-time":"2026-05-05T19:59:32Z","timestamp":1778011172000},"page":"7647-7657","source":"Crossref","is-referenced-by-count":0,"title":["ReFineVQA: Iterative Refinement of Video Description via Feedback Generation for Video Question Answering"],"prefix":"10.1109","author":[{"given":"Jeongwan","family":"Shin","sequence":"first","affiliation":[{"name":"DGIST"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chan","family":"Hur","sequence":"additional","affiliation":[{"name":"Kyungpook National University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Seongmin","family":"Cho","sequence":"additional","affiliation":[{"name":"4Genon"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jaeho","family":"Choi","sequence":"additional","affiliation":[{"name":"DGIST"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hyeyoung","family":"Park","sequence":"additional","affiliation":[{"name":"Kyungpook National University"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.52202\/068431-1723"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2102.05095"},{"key":"ref3","first-page":"1877","article-title":"Language models are few-shot learners","volume-title":"Advances in Neural Information Processing Systems","author":"Brown","year":"2020"},{"key":"ref4","article-title":"Video chatcaptioner: Towards enriched spatiotemporal descriptions","author":"Chen","year":"2023","journal-title":"CoRR"},{"key":"ref5","article-title":"Video-of-thought: Step-by-step video reasoning from perception to cognition","volume-title":"Forty-first International Conference on Machine Learning","author":"Fei"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52734.2025.02245"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-naacl.162"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.26599\/bdma.2024.9020026"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.261"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.naacl-long.528"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/d18-1167"},{"key":"ref12","article-title":"Llava-onevision: Easy visual task transfer","author":"Li","year":"2024"},{"key":"ref13","first-page":"19730","article-title":"BLIP-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","volume-title":"Proceedings of the 40th International Conference on Machine Learning","author":"Li"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/cvprw67362.2025.00024"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.384"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00718"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/icassp55912.2026.11463959"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.52202\/075280-1516"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00310"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.01764"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.679"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2004"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01257"},{"key":"ref25","article-title":"Gpt-4 technical report","year":"2024"},{"key":"ref26","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"Proceedings of the 38th International Conference on Machine Learning","author":"Radford"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51701.2025.01969"},{"key":"ref28","article-title":"Attention is all you need","volume-title":"Advances in Neural Information Processing Systems","author":"Vaswani","year":"2017"},{"key":"ref29","article-title":"Tarsier: Recipes for training and evaluating large video description models","author":"Wang","year":"2024"},{"key":"ref30","first-page":"58","article-title":"Videoagent: Long-form video understanding withnbsp;large language model asnbsp;agent","volume-title":"Computer Vision \u2013 ECCV 2024: 18th European Conference, Milan, Italy, September 29\u2013October 4, 2024, Proceedings, Part LXXX","author":"Wang"},{"key":"ref31","article-title":"Internvideo2.5: Empowering video mllms with long and rich context modeling","author":"Wang","year":"2025"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52734.2025.00311"},{"key":"ref33","article-title":"Chain-of-thought prompting elicits reasoning in large language models","volume-title":"Proceedings of the 36th International Conference on Neural Information Processing Systems","author":"Wei"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00965"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01254"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123427"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00171"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01413"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.52202\/075280-3354"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33019127"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01824"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.1209"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-demo.49"},{"key":"ref44","article-title":"LLaMA-adapter: Efficient fine-tuning of large language models with zero-initialized attention","volume-title":"The Twelfth International Conference on Learning Representations","author":"Zhang"},{"key":"ref45","article-title":"Video instruction tuning with synthetic data","author":"Zhang","year":"2024"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW59228.2023.00090"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01216-8_43"}],"event":{"name":"2026 IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV)","location":"Tucson, AZ, USA","start":{"date-parts":[[2026,3,6]]},"end":{"date-parts":[[2026,3,10]]}},"container-title":["2026 IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11491838\/11491925\/11492401.pdf?arnumber=11492401","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T05:56:24Z","timestamp":1778046984000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11492401\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,6]]},"references-count":47,"URL":"https:\/\/doi.org\/10.1109\/wacv61042.2026.00738","relation":{},"subject":[],"published":{"date-parts":[[2026,3,6]]}}}