{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T12:02:41Z","timestamp":1784894561579,"version":"3.55.0"},"reference-count":58,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T00:00:00Z","timestamp":1782864000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62402526"],"award-info":[{"award-number":["62402526"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Beijing Natural Science Foundation","award":["4244086"],"award-info":[{"award-number":["4244086"]}]},{"DOI":"10.13039\/501100010244","name":"Science Foundation of China University of Petroleum, Beijing","doi-asserted-by":"publisher","award":["2462024BJRC013, 2462023YJRC029"],"award-info":[{"award-number":["2462024BJRC013, 2462023YJRC029"]}],"id":[{"id":"10.13039\/501100010244","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1007\/s00530-026-02436-1","type":"journal-article","created":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T10:22:54Z","timestamp":1783765374000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Efficient video captioning annotation: a semi-automated framework via LLM-driven heuristics"],"prefix":"10.1007","volume":"32","author":[{"given":"Qianwen","family":"Cao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Guixin","family":"Liao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tao","family":"Tang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,11]]},"reference":[{"key":"2436_CR1","doi-asserted-by":"crossref","unstructured":"Xu, J., Mei, T., Yao, T., Rui, Y.: Msr-vtt: a large video description dataset for bridging video and language. In: Conference on Computer Vision and Pattern Recognition (CVPR) (2016)","DOI":"10.1109\/CVPR.2016.571"},{"key":"2436_CR2","doi-asserted-by":"crossref","unstructured":"Krishna, R., Hata, K., Ren, F., Fei-Fei, L., Niebles, J.C.: Dense-captioning events in videos. IEEE (2017)","DOI":"10.1109\/ICCV.2017.83"},{"key":"2436_CR3","doi-asserted-by":"crossref","unstructured":"Wang, X., Wu, J., Chen, J., Li, L., Wang, Y.F., Wang, W.Y.: Vatex: A large-scale, high-quality multilingual dataset for video-and-language research. IEEE (2019)","DOI":"10.1109\/ICCV.2019.00468"},{"key":"2436_CR4","unstructured":"Chen, D.L., Dolan, W.B.: Collecting highly parallel data for paraphrase evaluation. Association for Computational Linguistics (2011)"},{"key":"2436_CR5","doi-asserted-by":"crossref","unstructured":"Zhou, L., Xu, C., Corso, J.J.: Towards automatic learning of procedures from web instructional videos. Preprint at arXiv:1703.09788 (2017)","DOI":"10.1609\/aaai.v32i1.12342"},{"key":"2436_CR6","unstructured":"Sigurdsson, G.A., Gupta, A., Schmid, C., Farhadi, A., Alahari, K.: Charades-ego: a large-scale dataset of paired third and first person videos. Preprint at arXiv:1804.09626 (2018)"},{"key":"2436_CR7","doi-asserted-by":"crossref","unstructured":"Rohrbach, A., Rohrbach, M., Tandon, N., Schiele, B.: A dataset for movie description. IEEE (2015)","DOI":"10.1109\/CVPR.2015.7298940"},{"key":"2436_CR8","doi-asserted-by":"crossref","unstructured":"Li, Y., Song, Y., Cao, L., Tetreault, J., Luo, J.: Tgif: a new dataset and benchmark on animated gif description. IEEE (2016)","DOI":"10.1109\/CVPR.2016.502"},{"key":"2436_CR9","doi-asserted-by":"crossref","unstructured":"Miech, A., Zhukov, D., Alayrac, J.B., Tapaswi, M., Laptev, I., Sivic, J.: Howto100m: Learning a text-video embedding by watching hundred million narrated video clips. IEEE (2019)","DOI":"10.1109\/ICCV.2019.00272"},{"key":"2436_CR10","unstructured":"Pang, B., Rivera, C., Huang, G., Soricut, R., Zhu, Z.: Multimodal pretraining for dense video captioning. Association for Computational Linguistics (2020)"},{"key":"2436_CR11","unstructured":"LLC, G.: YouTube Video Platform. https:\/\/www.youtube.com\/. A publicly accessible video platform with a broad user base and diversified content ecosystems (2025)"},{"key":"2436_CR12","unstructured":"Inc., B.: Bilibili Video Platform. https:\/\/www.bilibili.com\/. A publicly accessible video platform with a broad user base and diversified content ecosystems (2025)"},{"key":"2436_CR13","unstructured":"Baidu, I.: Baidu Video Platform. https:\/\/video.baidu.com\/. A publicly accessible video platform with a broad user base and diversified content ecosystems (2025)"},{"key":"2436_CR14","doi-asserted-by":"crossref","unstructured":"Jiang, Y.G., Ye, G., Chang, S.F., Ellis, D.P.W., Loui, A.C.: Consumer video understanding: A benchmark database and an evaluation of human and machine performance. ACM (2011)","DOI":"10.1145\/1991996.1992025"},{"key":"2436_CR15","doi-asserted-by":"crossref","unstructured":"Das, P., Xu, C., Doell, R.F., Corso, J.J.: A thousand frames in just a few words: Lingual description of videos through latent topics and sparse object stitching. IEEE (2013)","DOI":"10.1109\/CVPR.2013.340"},{"key":"2436_CR16","doi-asserted-by":"crossref","unstructured":"Heilbron, F.C., Escorcia, V., Ghanem, B., Niebles, J.C.: Activitynet: A large-scale video benchmark for human activity understanding. IEEE (2015)","DOI":"10.1109\/CVPR.2015.7298698"},{"issue":"11","key":"2436_CR17","doi-asserted-by":"publisher","first-page":"9468","DOI":"10.1109\/TPAMI.2024.3381075","volume":"47","author":"K Grauman","year":"2024","unstructured":"Grauman, K., Westbury, A., Byrne, E., Cartillier, V., Chavis, Z., Furnari, A., Girdhar, R., Hamburger, J., Jiang, H., Kukreja, D.: Ego4d: Around the world in 3,000 hours of egocentric video. Pattern Anal. Mach. Intell. IEEE Trans. 47(11), 9468\u20139509 (2024)","journal-title":"Pattern Anal. Mach. Intell. IEEE Trans."},{"key":"2436_CR18","doi-asserted-by":"crossref","unstructured":"Chen, L., Wei, X., Li, J., Dong, X., Zhang, P., Zang, Y., Chen, Z., Duan, H., Lin, B., Tang, Z.: Sharegpt4video: improving video understanding and generation with better captions. Preprint at arXiv:2406.04325 (2024)","DOI":"10.52202\/079017-0614"},{"key":"2436_CR19","doi-asserted-by":"crossref","unstructured":"Bain, M., Nagrani, A., Varol, G., Zisserman, A.: Frozen in time: a joint video and image encoder for end-to-end retrieval. Preprint at arXiv:2104.00650 (2021)","DOI":"10.1109\/ICCV48922.2021.00175"},{"key":"2436_CR20","doi-asserted-by":"crossref","unstructured":"Xue, H., Hang, T., Zeng, Y., Sun, Y., Liu, B., Yang, H., Fu, J., Guo, B.: Advancing high-resolution video-language representation with large-scale video transcriptions. Preprint at arXiv:2111.10337 (2021)","DOI":"10.1109\/CVPR52688.2022.00498"},{"issue":"2","key":"2436_CR21","doi-asserted-by":"publisher","first-page":"179","DOI":"10.1207\/s15516709cog1402_1","volume":"14","author":"JL Elman","year":"1990","unstructured":"Elman, J.L.: Finding structure in time. Cogn. Sci. 14(2), 179\u2013211 (1990)","journal-title":"Cogn. Sci."},{"key":"2436_CR22","doi-asserted-by":"crossref","unstructured":"Cho, K., Van\u00a0Merrienboer, B., Gulcehre, C., Bahdanau, D., Bougares, F., Schwenk, H., Bengio, Y.: Learning phrase representations using rnn encoder-decoder for statistical machine translation. Computer Science (2014)","DOI":"10.3115\/v1\/D14-1179"},{"key":"2436_CR23","unstructured":"Lipton, Z.C., Berkowitz, J., Elkan, C.: A critical review of recurrent neural networks for sequence learning. Computer Science (2015)"},{"issue":"8","key":"2436_CR24","doi-asserted-by":"publisher","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter, S., Schmidhuber, J.: Long short-term memory. Neural Comput. 9(8), 1735\u20131780 (1997). https:\/\/doi.org\/10.1162\/neco.1997.9.8.1735","journal-title":"Neural Comput."},{"issue":"11","key":"2436_CR25","doi-asserted-by":"publisher","first-page":"2673","DOI":"10.1109\/78.650093","volume":"45","author":"M Schuster","year":"1997","unstructured":"Schuster, M., Paliwal, K.K.: Bidirectional recurrent neural networks. IEEE Trans. Signal Process. 45(11), 2673\u20132681 (1997). https:\/\/doi.org\/10.1109\/78.650093","journal-title":"IEEE Trans. Signal Process."},{"issue":"11","key":"2436_CR26","doi-asserted-by":"publisher","first-page":"2278","DOI":"10.1109\/5.726791","volume":"86","author":"Y Lecun","year":"1998","unstructured":"Lecun, Y., Bottou, L.: Gradient-based learning applied to document recognition. Proc. IEEE 86(11), 2278\u20132324 (1998)","journal-title":"Proc. IEEE"},{"key":"2436_CR27","unstructured":"Krizhevsky, A., Sutskever, I., Hinton, G.: Imagenet classification with deep convolutional neural networks. In: NIPS (2012)"},{"key":"2436_CR28","unstructured":"Simonyan, K., Zisserman, A.: Very deep convolutional networks for large-scale image recognition. Computer Science (2014)"},{"key":"2436_CR29","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. IEEE (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"2436_CR30","doi-asserted-by":"crossref","unstructured":"Huang, G., Liu, Z., Laurens, V.D.M., Weinberger, K.Q.: Densely connected convolutional networks. IEEE Computer Society (2016)","DOI":"10.1109\/CVPR.2017.243"},{"key":"2436_CR31","doi-asserted-by":"crossref","unstructured":"Venugopalan, S., Rohrbach, M., Donahue, J., Mooney, R., Saenko, K.: Sequence to sequence \u2013 video to text. IEEE (2016)","DOI":"10.1109\/ICCV.2015.515"},{"key":"2436_CR32","doi-asserted-by":"crossref","unstructured":"Giannakopoulos, A., Coriou, M., Hossmann, A., Baeriswyl, M., Musat, C.: Resilient combination of complementary cnn and rnn features for text classification through attention and ensembling. Swisscom AG (Schweiz);Swisscom AG (Schweiz);Swisscom AG (Schweiz);Swisscom AG (Schweiz);Swisscom AG (Schweiz); (2019)","DOI":"10.1109\/SDS.2019.000-7"},{"key":"2436_CR33","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, L., Polosukhin, I.: Attention is all you need. arXiv (2017)"},{"key":"2436_CR34","unstructured":"Radford, A., Narasimhan, K.: Improving language understanding by generative pre-training. OpenAI Research Preprint (2018)"},{"key":"2436_CR35","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: Bert: Pre-training of deep bidirectional transformers for language understanding. Preprint at ArXiv:abs\/1810.04805 (2019)"},{"key":"2436_CR36","unstructured":"Sabour, S., Frosst, N., Hinton, G.E.: Dynamic routing between capsules. ArXiv (2017)"},{"key":"2436_CR37","unstructured":"Hinton, G.E., Sabour, S., Frosst, N.: Matrix capsules with em routing. In: International Conference on Learning Representations (ICLR) (2018)"},{"key":"2436_CR38","doi-asserted-by":"crossref","unstructured":"Edraki, M., Rahnavard, N., Shah, M.: Subspace capsule network. Preprint at arXiv:2002.02924 (2020)","DOI":"10.1609\/aaai.v34i07.6703"},{"key":"2436_CR39","unstructured":"Ha, D., Schmidhuber, J.: Recurrent world models facilitate policy evolution. Adv. Neural Inf. Process. Syst. (2018)"},{"key":"2436_CR40","unstructured":"Hafner, D., Lillicrap, T., Fischer, I., Villegas, R., Ha, D., Lee, H., Davidson, J.: Learning latent dynamics for planning from pixels. In: 36th International Conference on Machine Learning: ICML 2019, Long Beach, California, USA, 9-15 June 2019, Part 7 of 19 (2019)"},{"key":"2436_CR41","unstructured":"Hafner, D., Lillicrap, T., Ba, J., Norouzi, M.: Dream to control: Learning behaviors by latent imagination. Preprint at arXiv:1912.01603 (2019)"},{"key":"2436_CR42","unstructured":"Hafner, D., Lillicrap, T., Norouzi, M., Ba, J.: Mastering atari with discrete world models. Preprint at arXiv:2010.02193 (2020)"},{"issue":"8059","key":"2436_CR43","doi-asserted-by":"publisher","first-page":"647","DOI":"10.1038\/s41586-025-08744-2","volume":"640","author":"D Hafner","year":"2025","unstructured":"Hafner, D., Pasukonis, J., Ba, J., Lillicrap, T.: Mastering diverse control tasks through world models. Nature 640(8059), 647\u2013653 (2025)","journal-title":"Nature"},{"key":"2436_CR44","unstructured":"Wu, P., Escontrela, A., Hafner, D., Goldberg, K., Abbeel, P.: Daydreamer: world models for physical robot learning. Preprint at arXiv:2206.14176 (2022)"},{"key":"2436_CR45","doi-asserted-by":"crossref","unstructured":"Zhao, G., Wang, X., Zhu, Z., Chen, X., Huang, G., Bao, X., Wang, X.: Drivedreamer-2: llm-enhanced world models for diverse driving video generation. Preprint at arXiv:2403.06845 (2024)","DOI":"10.1609\/aaai.v39i10.33130"},{"key":"2436_CR46","unstructured":"Bruce, J., Dennis, M., Edwards, A., Parker-Holder, J., Shi, Y., Hughes, E., Lai, M., Mavalankar, A., Steigerwald, R., Apps, C.: Genie: generative interactive environments. Preprint at arXiv:2402.15391 (2024)"},{"key":"2436_CR47","doi-asserted-by":"publisher","first-page":"409","DOI":"10.1007\/978-3-031-16788-1_25","volume-title":"Pattern Recognition","author":"Z Ghaderi","year":"2022","unstructured":"Ghaderi, Z., Salewski, L., Lensch, H.P.A.: Diverse video captioning by adaptive spatio-temporal attention. In: Andres, B., Bernard, F., Cremers, D., Frintrop, S., Goldl\u00fccke, B., Ihrke, I. (eds.) Pattern Recognition, pp. 409\u2013425. Springer, Cham (2022)"},{"key":"2436_CR48","doi-asserted-by":"publisher","first-page":"302","DOI":"10.1007\/978-981-96-2074-6_36","volume-title":"MultiMedia Modeling","author":"Y-T Cheng","year":"2025","unstructured":"Cheng, Y.-T., Wu, J., Ma, Z., He, J., Wei, X.-Y., Ngo, C.-W.: Interactive video search with multi-modal llm video captioning. In: Ide, I., Kompatsiaris, I., Xu, C., Yanai, K., Chu, W.-T., Nitta, N., Riegler, M., Yamasaki, T. (eds.) MultiMedia Modeling, pp. 302\u2013309. Springer, Singapore (2025)"},{"key":"2436_CR49","doi-asserted-by":"publisher","first-page":"449","DOI":"10.1007\/978-981-96-4139-0_39","volume-title":"ICT: Applications and Social Interfaces","author":"D Bhatt","year":"2025","unstructured":"Bhatt, D., Thakkar, P.: Transformer and llm-based captioning module for dense video captioning. In: Joshi, A., Ragel, R., Mahmud, M., Kartik, S. (eds.) ICT: Applications and Social Interfaces, pp. 449\u2013459. Springer, Singapore (2025)"},{"key":"2436_CR50","doi-asserted-by":"crossref","unstructured":"Bansal, H., Bitton, Y., Szpektor, I., Chang, K.W., Grover, A.: Videocon: robust video-language alignment via contrast captions. IEEE (2023)","DOI":"10.1109\/CVPR52733.2024.01321"},{"key":"2436_CR51","first-page":"11817","volume":"2023","author":"R Wang","year":"2023","unstructured":"Wang, R., Zhou, W., Sachan, M.: Let\u2019s synthesize step by step: iterative dataset synthesis with large language models by extrapolating errors from small models. Find. Assoc. Comput. Linguist. EMNLP 2023, 11817\u201311831 (2023)","journal-title":"Find. Assoc. Comput. Linguist. EMNLP"},{"key":"2436_CR52","doi-asserted-by":"crossref","unstructured":"Choi, J., Yun, J., Jin, K., Kim, Y.B.: Multi-news+: cost-efficient dataset cleansing via llm-based data annotation. Preprint at arXiv:2404.09682 (2024)","DOI":"10.18653\/v1\/2024.emnlp-main.2"},{"key":"2436_CR53","doi-asserted-by":"crossref","unstructured":"Aguda, T., Siddagangappa, S., Kochkina, E., Kaur, S., Wang, D., Smiley, C., Shah, S.: Large language models as financial data annotators: A study on effectiveness and efficiency. Preprint at arXiv:2403.18152 (2024)","DOI":"10.63317\/4oh42vhxfvci"},{"key":"2436_CR54","doi-asserted-by":"crossref","unstructured":"Kholodna, N., Julka, S., Khodadadi, M., Gumus, M.N., Granitzer, M.: LLMs in the loop: leveraging large language model annotations for active learning in low-resource languages. Preprint at arxiv:2404.02261 (2024)","DOI":"10.1007\/978-3-031-70381-2_25"},{"key":"2436_CR55","unstructured":"Gu, J., You, K., Cho, H.-C., Kim, J., Hong, E.K., Roh, B.: CheX-GPT: harnessing Large Language Models for Enhanced Chest X-ray Report Labeling. Preprint at arxiv:2401.11505 (2024)"},{"issue":"6","key":"2436_CR56","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3696461","volume":"16","author":"A Hota","year":"2025","unstructured":"Hota, A., Chatterjee, S., Chakraborty, S.: Evaluating large language models as virtual annotators for time-series physical sensing data. ACM Trans. Intell. Syst. Technol. 16(6), 1\u201325 (2025). https:\/\/doi.org\/10.1145\/3696461","journal-title":"ACM Trans. Intell. Syst. Technol."},{"key":"2436_CR57","doi-asserted-by":"publisher","unstructured":"Wu, A., Han, Y., Yang, Y.: Video interactive captioning with human prompts. In: Proceedings of the Twenty-Eighth International Joint Conference on Artificial Intelligence, IJCAI-19, pp. 961\u2013967. International Joint Conferences on Artificial Intelligence Organization (2019). https:\/\/doi.org\/10.24963\/ijcai.2019\/135","DOI":"10.24963\/ijcai.2019\/135"},{"key":"2436_CR58","doi-asserted-by":"crossref","unstructured":"Madaan, A., Tandon, N., Gupta, P., Hallinan, S., Gao, L., Wiegreffe, S., Alon, U., Dziri, N., Prabhumoye, S., Yang, Y., Gupta, S., Majumder, B.P., Hermann, K., Welleck, S., Yazdanbakhsh, A., Clark, P.: Self-Refine: iterative refinement with self-feedback (2023). https:\/\/arxiv.org\/abs\/2303.17651","DOI":"10.52202\/075280-2019"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-026-02436-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-026-02436-1","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-026-02436-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,24]],"date-time":"2026-07-24T11:31:29Z","timestamp":1784892689000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-026-02436-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7]]},"references-count":58,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2026,7]]}},"alternative-id":["2436"],"URL":"https:\/\/doi.org\/10.1007\/s00530-026-02436-1","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7]]},"assertion":[{"value":"15 December 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 April 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 July 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"The authors declare that they have no Conflict of interest.","order":1,"name":"Ethics","label":"Conflicts of Interest","group":{"name":"EthicsHeading","label":"Declarations"}}],"article-number":"390"}}