{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,5]],"date-time":"2026-08-05T23:30:19Z","timestamp":1785972619872,"version":"3.56.0"},"reference-count":98,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100007825","name":"National Forestry and Grassland Administration","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100007825","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100020145","name":"Chengdu Research Base of Giant Panda Breeding","doi-asserted-by":"publisher","award":["CGF2024006"],"award-info":[{"award-number":["CGF2024006"]}],"id":[{"id":"10.13039\/100020145","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100020145","name":"Chengdu Research Base of Giant Panda Breeding","doi-asserted-by":"publisher","award":["2023KCPB-02"],"award-info":[{"award-number":["2023KCPB-02"]}],"id":[{"id":"10.13039\/100020145","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100020145","name":"Chengdu Research Base of Giant Panda Breeding","doi-asserted-by":"publisher","award":["CAZG2025C04"],"award-info":[{"award-number":["CAZG2025C04"]}],"id":[{"id":"10.13039\/100020145","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100020145","name":"Chengdu Research Base of Giant Panda Breeding","doi-asserted-by":"publisher","award":["202503KY0004"],"award-info":[{"award-number":["202503KY0004"]}],"id":[{"id":"10.13039\/100020145","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100020145","name":"Chengdu Research Base of Giant Panda Breeding","doi-asserted-by":"publisher","award":["2024CPB-C08"],"award-info":[{"award-number":["2024CPB-C08"]}],"id":[{"id":"10.13039\/100020145","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62173066"],"award-info":[{"award-number":["62173066"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2027,1]]},"DOI":"10.1016\/j.eswa.2026.133544","type":"journal-article","created":{"date-parts":[[2026,7,6]],"date-time":"2026-07-06T23:08:24Z","timestamp":1783379304000},"page":"133544","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PB","title":["ViS2T: A vision and scene graph to text model for captive giant panda video captioning"],"prefix":"10.1016","volume":"332","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-5184-9045","authenticated-orcid":false,"given":"Chenyu","family":"Ma","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chang","family":"Duan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9696-4944","authenticated-orcid":false,"given":"Ke","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Peng","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mengnan","family":"He","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhongrong","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8330-2164","authenticated-orcid":false,"given":"Ping","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ce","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.133544_bib0001","doi-asserted-by":"crossref","DOI":"10.1109\/TPAMI.2024.3522295","article-title":"A review of deep learning for video captioning","author":"Abdar","year":"2024","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.133544_bib0002","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"6077","article-title":"Bottom-up and top-down attention for image captioning and visual question answering","author":"Anderson","year":"2018"},{"key":"10.1016\/j.eswa.2026.133544_bib0003","article-title":"Pretrained transformers for multimodal fake news detection: Explainability using shapley additive explanations for contributions from text, image, and image captions","volume":"162","author":"Athira","year":"2025","journal-title":"Engineering Applications of Artificial Intelligence"},{"key":"10.1016\/j.eswa.2026.133544_bib0004","series-title":"Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization","first-page":"65","article-title":"METEOR: An automatic metric for mt evaluation with improved correlation with human judgments","author":"Banerjee","year":"2005"},{"key":"10.1016\/j.eswa.2026.133544_bib0005","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"4662","article-title":"The unreasonable effectiveness of clip features for image captioning: An experimental analysis","author":"Barraco","year":"2022"},{"key":"10.1016\/j.eswa.2026.133544_bib0006","series-title":"2025 IEEE international conference on acoustics, speech and signal processing (ICASSP)","first-page":"1","article-title":"Enhancing teacher classroom behavior descriptions: A spatio-temporal graph-based method for video captioning","author":"Cai","year":"2025"},{"issue":"4","key":"10.1016\/j.eswa.2026.133544_bib0007","doi-asserted-by":"crossref","first-page":"760","DOI":"10.1111\/2041-210X.14502","article-title":"YOLO-Behaviour: A simple, flexible framework to automatically quantify animal behaviours from videos","volume":"16","author":"Chan","year":"2025","journal-title":"Methods in Ecology and Evolution"},{"key":"10.1016\/j.eswa.2026.133544_bib0008","series-title":"Proceedings of the 49th annual meeting of the association for computational linguistics: Human language technologies","first-page":"190","article-title":"Collecting highly parallel data for paraphrase evaluation","author":"Chen","year":"2011"},{"issue":"Suppl. 2","key":"10.1016\/j.eswa.2026.133544_bib0009","doi-asserted-by":"crossref","first-page":"425","DOI":"10.1007\/s00500-021-06360-6","article-title":"RETRACTED ARTICLE: Memory-attended semantic context-aware network for video captioning","volume":"28","author":"Chen","year":"2024","journal-title":"Soft Computing"},{"issue":"10","key":"10.1016\/j.eswa.2026.133544_bib0010","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3679203","article-title":"Action-aware linguistic skeleton optimization network for non-autoregressive video captioning","volume":"20","author":"Chen","year":"2024","journal-title":"ACM Transactions on Multimedia Computing, Communications and Applications"},{"key":"10.1016\/j.eswa.2026.133544_bib0011","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"24185","article-title":"Internvl: Scaling up vision foundation models and aligning for generic visual-linguistic tasks","author":"Chen","year":"2024"},{"issue":"9","key":"10.1016\/j.eswa.2026.133544_bib0012","doi-asserted-by":"crossref","first-page":"11169","DOI":"10.1109\/TPAMI.2023.3268066","article-title":"RelTR: Relation transformer for scene graph generation","volume":"45","author":"Cong","year":"2023","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.133544_bib0013","series-title":"Advances in neural information processing systems 28","article-title":"Semi-supervised sequence learning","author":"Dai","year":"2015"},{"issue":"2","key":"10.1016\/j.eswa.2026.133544_bib0014","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3550276","article-title":"Semantic embedding guided attention with explicit visual feature fusion for video captioning","volume":"19","author":"Dong","year":"2023","journal-title":"ACM Transactions on Multimedia Computing, Communications and Applications"},{"key":"10.1016\/j.eswa.2026.133544_bib0015","series-title":"2024 IEEE International Conference on Signal, Information and Data Processing (ICSIDP)","first-page":"1","article-title":"Multi-attentional action recognition method for captive giant panda based on surveillance video","author":"Duan","year":"2024"},{"key":"10.1016\/j.eswa.2026.133544_bib0016","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"13854","article-title":"Mammalps: A multi-view video behavior monitoring dataset of wild mammals in the swiss alps","author":"Gabeff","year":"2025"},{"key":"10.1016\/j.eswa.2026.133544_bib0017","unstructured":"Gu, J., Bradbury, J., Xiong, C., Li, V. O. K., & Socher, R. (2017). Non-autoregressive neural machine translation. arXiv preprint arXiv: 1711.02281."},{"key":"10.1016\/j.eswa.2026.133544_bib0018","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"18941","article-title":"Text with knowledge graph augmented transformer for video captioning","author":"Gu","year":"2023"},{"key":"10.1016\/j.eswa.2026.133544_bib0019","series-title":"Proceedings of the IEEE international conference on computer vision","first-page":"2712","article-title":"Youtube2text: Recognizing and describing arbitrary activities using semantic hierarchies and zero-shot recognition","author":"Guadarrama","year":"2013"},{"key":"10.1016\/j.eswa.2026.133544_bib0020","doi-asserted-by":"crossref","unstructured":"Guo, L., Liu, J., Zhu, X., He, X., Jiang, J., & Lu, H. (2020). Non-autoregressive image captioning with counterfactuals-critical multi-agent learning. arXiv preprint arXiv: 2005.04690.","DOI":"10.24963\/ijcai.2020\/107"},{"key":"10.1016\/j.eswa.2026.133544_bib0021","article-title":"Action-driven semantic representation and aggregation for video captioning","author":"Han","year":"2024","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.eswa.2026.133544_bib0022","unstructured":"Hendrycks, D., & Gimpel, K. (2016). Gaussian error linear units (gelus). arXiv preprint arXiv: 1606.08415."},{"issue":"8","key":"10.1016\/j.eswa.2026.133544_bib0023","doi-asserted-by":"crossref","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","article-title":"Long short-term memory","volume":"9","author":"Hochreiter","year":"1997","journal-title":"Neural Computation"},{"key":"10.1016\/j.eswa.2026.133544_bib0024","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"8106","article-title":"Detecting human-object relationships in videos","author":"Ji","year":"2021"},{"key":"10.1016\/j.eswa.2026.133544_bib0025","doi-asserted-by":"crossref","first-page":"2367","DOI":"10.1109\/TMM.2023.3295098","article-title":"Memory-based augmentation network for video captioning","volume":"26","author":"Jing","year":"2023","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.eswa.2026.133544_bib0026","doi-asserted-by":"crossref","unstructured":"Lei, J., Yu, L., Bansal, M., & Berg, T. L. (2018). TVQA: Localized, compositional video question answering. arXiv preprint arXiv: 1809.01696.","DOI":"10.18653\/v1\/D18-1167"},{"issue":"2","key":"10.1016\/j.eswa.2026.133544_bib0027","doi-asserted-by":"crossref","first-page":"1049","DOI":"10.1109\/TPAMI.2023.3327677","article-title":"Learning hierarchical modular networks for video captioning","volume":"46","author":"Li","year":"2023","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.133544_bib0028","series-title":"Proceedings of the international conference on machine learning","first-page":"19730","article-title":"BLIP-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","author":"Li","year":"2023"},{"key":"10.1016\/j.eswa.2026.133544_bib0029","doi-asserted-by":"crossref","first-page":"2726","DOI":"10.1109\/TIP.2022.3158546","article-title":"Long short-term relation transformer with global gating for video captioning","volume":"31","author":"Li","year":"2022","journal-title":"IEEE Transactions on Image Processing"},{"key":"10.1016\/j.eswa.2026.133544_bib0030","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.111176","article-title":"Pseudo-labeling with keyword refining for few-supervised video captioning","volume":"159","author":"Li","year":"2025","journal-title":"Pattern Recognition"},{"key":"10.1016\/j.eswa.2026.133544_bib0031","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2023.121462","article-title":"Analyzing the pregnancy status of giant pandas with hierarchical behavioral information","volume":"237","author":"Li","year":"2024","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.133544_bib0032","series-title":"European conference on computer vision","first-page":"121","article-title":"Oscar: Object-semantics aligned pre-training for vision-language tasks","author":"Li","year":"2020"},{"key":"10.1016\/j.eswa.2026.133544_bib0033","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"8518","article-title":"Scheduled sampling in vision-language pretraining with decoupled encoder-decoder network","volume":"vol. 35","author":"Li","year":"2021"},{"key":"10.1016\/j.eswa.2026.133544_bib0034","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"7492","article-title":"Jointly localizing and describing events for dense video captioning","author":"Li","year":"2018"},{"key":"10.1016\/j.eswa.2026.133544_bib0035","series-title":"Text summarization branches out: Proceedings of the acl-04 workshop","first-page":"74","article-title":"ROUGE: A package for automatic evaluation of summaries","author":"Lin","year":"2004"},{"key":"10.1016\/j.eswa.2026.133544_bib0036","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"17949","article-title":"Swinbert: End-to-end transformers with sparse attention for video captioning","author":"Lin","year":"2022"},{"issue":"10","key":"10.1016\/j.eswa.2026.133544_bib0037","doi-asserted-by":"crossref","first-page":"9718","DOI":"10.1109\/TCSVT.2024.3399933","article-title":"Evcap: Element-aware video captioning","volume":"34","author":"Liu","year":"2024","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.eswa.2026.133544_bib0038","series-title":"Proceedings of the european conference on computer vision (eccv)","first-page":"338","article-title":"Show, tell and discriminate: Image captioning by self-retrieval with partially labeled data","author":"Liu","year":"2018"},{"key":"10.1016\/j.eswa.2026.133544_bib0039","series-title":"Proceedings of the 63rd annual meeting of the association for computational linguistics (volume 1: Long papers)","first-page":"16524","article-title":"Task-specific information decomposition for end-to-end dense video captioning","author":"Liu","year":"2025"},{"key":"10.1016\/j.eswa.2026.133544_bib0040","doi-asserted-by":"crossref","first-page":"1281","DOI":"10.1038\/s41593-018-0209-y","article-title":"Deeplabcut: Markerless pose estimation of user-defined body parts with deep learning","volume":"21","author":"Mathis","year":"2018","journal-title":"Nature Neuroscience"},{"key":"10.1016\/j.eswa.2026.133544_bib0041","unstructured":"Mokady, R., Hertz, A., & Bermano, A. H. (2021). CLIPCap: Clip prefix for image captioning. arXiv preprint arXiv: 2111.09734."},{"key":"10.1016\/j.eswa.2026.133544_bib0042","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"19023","article-title":"Animal kingdom: A large and diverse dataset for animal behavior understanding","author":"Ng","year":"2022"},{"key":"10.1016\/j.eswa.2026.133544_bib0043","doi-asserted-by":"crossref","first-page":"E5716","DOI":"10.1073\/pnas.1719367115","article-title":"Automatically identifying, counting, and describing wild animals in camera-trap images with deep learning","volume":"115","author":"Norouzzadeh","year":"2018","journal-title":"Proceedings of the National Academy of Sciences"},{"key":"10.1016\/j.eswa.2026.133544_bib0044","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"4594","article-title":"Jointly modeling embedding and translation to bridge video and language","author":"Pan","year":"2016"},{"key":"10.1016\/j.eswa.2026.133544_bib0045","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"6504","article-title":"Video captioning with transferred semantic attributes","author":"Pan","year":"2017"},{"key":"10.1016\/j.eswa.2026.133544_bib0046","doi-asserted-by":"crossref","unstructured":"Pan, Z., Li, P., & Wang, W. (2025). Sgcap: Decoding semantic group for zero-shot video captioning. arXiv preprint arXiv: 2508.01270.","DOI":"10.2139\/ssrn.5861646"},{"key":"10.1016\/j.eswa.2026.133544_bib0047","series-title":"Proceedings of the 40th annual meeting of the association for computational linguistics (ACL)","first-page":"311","article-title":"BLEU: A method for automatic evaluation of machine translation","author":"Papineni","year":"2002"},{"issue":"6","key":"10.1016\/j.eswa.2026.133544_bib0048","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3712059","article-title":"Dense video captioning: A survey of techniques, datasets and evaluation protocols","volume":"57","author":"Qasim","year":"2025","journal-title":"ACM Computing Surveys"},{"key":"10.1016\/j.eswa.2026.133544_bib0049","series-title":"Proceedings of the international conference on machine learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.eswa.2026.133544_bib0050","doi-asserted-by":"crossref","DOI":"10.1016\/j.cosrev.2020.100289","article-title":"Deep learning methods for multi-species animal re-identification and tracking\u2014a survey","volume":"38","author":"Ravoor","year":"2020","journal-title":"Computer Science Review"},{"key":"10.1016\/j.eswa.2026.133544_bib0051","doi-asserted-by":"crossref","DOI":"10.1016\/j.patrec.2025.04.007","article-title":"From visual features to key concepts: A dynamic and static concept-driven approach for video captioning","author":"Ren","year":"2025","journal-title":"Pattern Recognition Letters"},{"key":"10.1016\/j.eswa.2026.133544_bib0052","series-title":"Proceedings of the IEEE international conference on computer vision","first-page":"433","article-title":"Translating video content to natural language descriptions","author":"Rohrbach","year":"2013"},{"key":"10.1016\/j.eswa.2026.133544_bib0053","doi-asserted-by":"crossref","first-page":"7928","DOI":"10.3390\/s23187928","article-title":"Captive animal behavior study by video analysis","volume":"23","author":"Rotaru","year":"2023","journal-title":"Sensors"},{"key":"10.1016\/j.eswa.2026.133544_bib0054","doi-asserted-by":"crossref","unstructured":"Sahoo, P., Meharia, P., Ghosh, A., Saha, S., Jain, V., & Chadha, A. (2024). A comprehensive survey of hallucination in large language, image, video and audio foundation models.","DOI":"10.18653\/v1\/2024.findings-emnlp.685"},{"key":"10.1016\/j.eswa.2026.133544_bib0055","article-title":"Beyond observation: Deep learning for animal behavior and ecological conservation","author":"Saoud","year":"2024","journal-title":"Ecological Informatics"},{"key":"10.1016\/j.eswa.2026.133544_bib0056","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"17959","article-title":"End-to-end generative pretraining for multimodal video captioning","author":"Seo","year":"2022"},{"key":"10.1016\/j.eswa.2026.133544_bib0057","series-title":"Proceedings of the 56th annual meeting of the association for computational linguistics (volume 1: Long papers)","first-page":"2556","article-title":"Conceptual captions: A cleaned, hypernymed, image alt-text dataset for automatic image captioning","author":"Sharma","year":"2018"},{"key":"10.1016\/j.eswa.2026.133544_bib0058","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"15558","article-title":"Accurate and fast compressed video captioning","author":"Shen","year":"2023"},{"issue":"7","key":"10.1016\/j.eswa.2026.133544_bib0059","doi-asserted-by":"crossref","first-page":"3210","DOI":"10.1109\/TIP.2018.2814344","article-title":"Self-supervised video hashing with hierarchical binary auto-encoder","volume":"27","author":"Song","year":"2018","journal-title":"IEEE Transactions on Image Processing"},{"key":"10.1016\/j.eswa.2026.133544_bib0060","article-title":"Scene adaptive dynamic multi-modal knowledge for video captioning","author":"Sun","year":"2025","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.133544_bib0061","series-title":"Proceedings of the 29th ACM international conference on multimedia","first-page":"4858","article-title":"Clip4caption: Clip for video caption","author":"Tang","year":"2021"},{"key":"10.1016\/j.eswa.2026.133544_bib0062","unstructured":"Q. Team (2026). Qwen3.5: Towards native multimodal agents."},{"key":"10.1016\/j.eswa.2026.133544_bib0063","doi-asserted-by":"crossref","unstructured":"Trabelsi, M., Boyd, A., Cao, J., & Uzunalioglu, H. (2025). Time series language model for descriptive caption generation. arXiv preprint arXiv: 2501.01832.","DOI":"10.1016\/j.engappai.2025.112673"},{"key":"10.1016\/j.eswa.2026.133544_bib0064","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2022.109204","article-title":"Relation-aware attention for video captioning via graph learning","volume":"136","author":"Tu","year":"2023","journal-title":"Pattern Recognition"},{"key":"10.1016\/j.eswa.2026.133544_bib0065","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition (cvpr)","first-page":"4566","article-title":"Cider: Consensus-based image description evaluation","author":"Vedantam","year":"2015"},{"key":"10.1016\/j.eswa.2026.133544_bib0066","series-title":"Proceedings of the IEEE international conference on computer vision","first-page":"4534","article-title":"Sequence to sequence\u2014video to text","author":"Venugopalan","year":"2015"},{"issue":"4","key":"10.1016\/j.eswa.2026.133544_bib0067","doi-asserted-by":"crossref","first-page":"652","DOI":"10.1109\/TPAMI.2016.2587640","article-title":"Show and tell: Lessons learned from the 2015 MSCOCO image captioning challenge","volume":"39","author":"Vinyals","year":"2016","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.133544_bib0068","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"7622","article-title":"Reconstruction network for video captioning","author":"Wang","year":"2018"},{"key":"10.1016\/j.eswa.2026.133544_bib0069","unstructured":"Wang, J., Fu, L., Li, Y., Zhu, Y., Jing, Y., Wu, X., & Zheng, J. (2026). Diffvc: A non-autoregressive framework based on diffusion model for video captioning. arXiv preprint arXiv: 2604.08084."},{"key":"10.1016\/j.eswa.2026.133544_bib0070","unstructured":"Wang, J., Yang, Z., Hu, X., Li, L., Lin, K., Gan, Z., Liu, Z., Liu, C., & Wang, L. (2022). Git: A generative image-to-text transformer for vision and language."},{"key":"10.1016\/j.eswa.2026.133544_bib0071","article-title":"Clipcap++: An efficient image captioning approach via image encoder optimization and llm fine-tuning","author":"Wang","year":"2025","journal-title":"Applied Soft Computing"},{"key":"10.1016\/j.eswa.2026.133544_bib0072","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"6847","article-title":"End-to-end dense video captioning with parallel decoding","author":"Wang","year":"2021"},{"key":"10.1016\/j.eswa.2026.133544_bib0073","article-title":"Visual evidence-aware for object hallucinations rectification in llm-based video captioning","author":"Wang","year":"2025","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"10.1016\/j.eswa.2026.133544_bib0074","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"8417","article-title":"Event-equalized dense video captioning","author":"Wu","year":"2025"},{"key":"10.1016\/j.eswa.2026.133544_bib0075","series-title":"2020 IEEE international conference on multimedia and expo (ICME)","first-page":"1","article-title":"Video captioning with temporal and region graph convolution network","author":"Xiao","year":"2020"},{"key":"10.1016\/j.eswa.2026.133544_bib0076","series-title":"Proceedings of the European conference on computer vision (ECCV)","first-page":"468","article-title":"Move forward and tell: A progressive generator of video descriptions","author":"Xiong","year":"2018"},{"key":"10.1016\/j.eswa.2026.133544_bib0077","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"5288","article-title":"Msr-vtt: A large video description dataset for bridging video and language","author":"Xu","year":"2016"},{"key":"10.1016\/j.eswa.2026.133544_bib0078","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"2346","article-title":"Jointly modeling deep video and compositional text to bridge vision and language in a unified framework","volume":"vol. 29","author":"Xu","year":"2015"},{"key":"10.1016\/j.eswa.2026.133544_bib0079","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2024.126296","article-title":"CroCaps: A CLIP-assisted cross-domain video captioner","volume":"268","author":"Xu","year":"2025","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.133544_bib0080","doi-asserted-by":"crossref","first-page":"5366","DOI":"10.1109\/TIP.2023.3307969","article-title":"Concept-aware video captioning: Describing videos with effective prior information","volume":"32","author":"Yang","year":"2023","journal-title":"IEEE Transactions on Image Processing"},{"key":"10.1016\/j.eswa.2026.133544_bib0081","unstructured":"Yang, B., Chen, J., Li, W., Yao, W., & Zhou, Y. (2025). Expertized caption auto-enhancement for video-text retrieval. arXiv preprint arXiv: 2502.02885."},{"key":"10.1016\/j.eswa.2026.133544_bib0082","first-page":"490","article-title":"Giant panda pose recognition based on an improved YOLOv5s","volume":"42","author":"Yang","year":"2023","journal-title":"Sichuan Journal of Zoology"},{"key":"10.1016\/j.eswa.2026.133544_bib0083","series-title":"Proceedings of the IEEE international conference on computer vision","first-page":"4507","article-title":"Describing videos by exploiting temporal structure","author":"Yao","year":"2015"},{"key":"10.1016\/j.eswa.2026.133544_bib0084","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"4651","article-title":"Image captioning with semantic attention","author":"You","year":"2016"},{"key":"10.1016\/j.eswa.2026.133544_bib0085","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"4584","article-title":"Video paragraph captioning using hierarchical recurrent neural networks","author":"Yu","year":"2016"},{"key":"10.1016\/j.eswa.2026.133544_bib0086","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.111138","article-title":"Fully exploring object relation interaction and hidden state attention for video captioning","volume":"159","author":"Yuan","year":"2025","journal-title":"Pattern Recognition"},{"key":"10.1016\/j.eswa.2026.133544_bib0087","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"4334","article-title":"Leveraging video descriptions to learn video question answering","volume":"vol. 31","author":"Zeng","year":"2017"},{"key":"10.1016\/j.eswa.2026.133544_bib0088","article-title":"Visual commonsense-aware representation network for video captioning","author":"Zeng","year":"2023","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"10.1016\/j.eswa.2026.133544_bib0089","series-title":"Proceedings of the 2025 conference of the nations of the americas chapter of the association for computational linguistics: Human language technologies, volume 2: Short papers","first-page":"292","article-title":"Pretrained image-text models are secretly video captioners","author":"Zhang","year":"2025"},{"key":"10.1016\/j.eswa.2026.133544_bib0090","series-title":"Proceedings of the 31st acm international conference on multimedia","first-page":"4778","article-title":"Depth-aware sparse transformer for video-language learning","author":"Zhang","year":"2023"},{"key":"10.1016\/j.eswa.2026.133544_bib0091","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"8327","article-title":"Object-aware aggregation with bidirectional temporal graph for video captioning","author":"Zhang","year":"2019"},{"key":"10.1016\/j.eswa.2026.133544_bib0092","doi-asserted-by":"crossref","first-page":"6209","DOI":"10.1109\/TIP.2020.2988435","article-title":"Video captioning with object-aware spatio-temporal correlation and aggregation","volume":"29","author":"Zhang","year":"2020","journal-title":"IEEE Transactions on Image Processing"},{"key":"10.1016\/j.eswa.2026.133544_bib0093","doi-asserted-by":"crossref","DOI":"10.1016\/j.engappai.2024.109102","article-title":"Multi-scale features with temporal information guidance for video captioning","volume":"137","author":"Zhao","year":"2024","journal-title":"Engineering Applications of Artificial Intelligence"},{"key":"10.1016\/j.eswa.2026.133544_bib0094","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"3724","article-title":"Refined semantic enhancement towards frequency diffusion for video captioning","volume":"vol. 37","author":"Zhong","year":"2023"},{"key":"10.1016\/j.eswa.2026.133544_bib0095","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"13041","article-title":"Unified vision-language pre-training for image captioning and vqa","volume":"vol. 34","author":"Zhou","year":"2020"},{"key":"10.1016\/j.eswa.2026.133544_bib0096","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"8739","article-title":"End-to-end dense video captioning with masked transformer","author":"Zhou","year":"2018"},{"key":"10.1016\/j.eswa.2026.133544_bib0097","unstructured":"Zhu, W., Pang, B., Thapliyal, A. V., Wang, W. Y., & Soricut, R. (2022). End-to-end dense video captioning as sequence generation. arXiv preprint arXiv: 2204.08121."},{"issue":"1","key":"10.1016\/j.eswa.2026.133544_bib0098","first-page":"82","article-title":"Giant panda head image segmentation based on dual-model fusion","volume":"43","author":"Zhou","year":"2023","journal-title":"Acta Theriologica Sinica"}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S095741742602453X?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S095741742602453X?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,8,5]],"date-time":"2026-08-05T21:42:46Z","timestamp":1785966166000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S095741742602453X"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2027,1]]},"references-count":98,"alternative-id":["S095741742602453X"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.133544","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2027,1]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"ViS2T: A vision and scene graph to text model for captive giant panda video captioning","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.133544","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"133544"}}