{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,5]],"date-time":"2026-08-05T18:46:45Z","timestamp":1785955605925,"version":"3.56.0"},"reference-count":75,"publisher":"Springer Science and Business Media LLC","issue":"10","license":[{"start":{"date-parts":[[2025,9,25]],"date-time":"2025-09-25T00:00:00Z","timestamp":1758758400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,9,25]],"date-time":"2025-09-25T00:00:00Z","timestamp":1758758400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Sci. China Inf. Sci."],"published-print":{"date-parts":[[2025,10]]},"DOI":"10.1007\/s11432-024-4321-9","type":"journal-article","created":{"date-parts":[[2025,12,1]],"date-time":"2025-12-01T08:46:55Z","timestamp":1764578815000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":101,"title":["VideoChat: chat-centric video understanding"],"prefix":"10.1007","volume":"68","author":[{"given":"Kunchang","family":"Li","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yinan","family":"He","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yi","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yizhuo","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wenhai","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ping","family":"Luo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yali","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Limin","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yu","family":"Qiao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,9,25]]},"reference":[{"key":"4321_CR1","unstructured":"Huang S H, Dong L, Wang W H, et al. Language is not all you need: aligning perception with language models. 2023. ArXiv:2302.14045"},{"key":"4321_CR2","first-page":"34892","volume-title":"Proceedings of the 37th International Conference on Neural Information Processing Systems","author":"H T Liu","year":"2023","unstructured":"Liu H T, Li C Y, Wu Q Y, et al. Visual instruction tuning. In: Proceedings of the 37th International Conference on Neural Information Processing Systems, 2023. 34892\u201334916"},{"key":"4321_CR3","unstructured":"Zhu D Y, Chen J, Shen X Q, et al. MiniGPT-4: enhancing vision-language understanding with advanced large language models. 2023. ArXiv:2304.10592"},{"key":"4321_CR4","unstructured":"Ye Q H, Xu H Y, Xu G H, et al. mPLUG-Owl: modularization empowers large language models with multimodality. 2023. ArXiv:2304.14178"},{"key":"4321_CR5","first-page":"9879","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"A Miech","year":"2020","unstructured":"Miech A, Alayrac J B, Smaira L, et al. End-to-end learning of visual representations from uncurated instructional videos. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2020. 9879\u20139889"},{"key":"4321_CR6","unstructured":"Li T H, Wang L M. Learning spatiotemporal features via video and text pair discrimination. 2001. ArXiv:2001.05691"},{"key":"4321_CR7","first-page":"23634","volume-title":"Proceedings of the 35th International Conference on Neural Information Processing Systems","author":"R Zellers","year":"2021","unstructured":"Zellers R, Lu X M, Hessel J, et al. MERLOT: multimodal neural script knowledge models. In: Proceedings of the 35th International Conference on Neural Information Processing Systems, 2021. 23634\u201323651"},{"key":"4321_CR8","doi-asserted-by":"crossref","unstructured":"Xu H, Ghosh G, Huang P Y, et al. VideoCLIP: contrastive pre-training for zero-shot video-text understanding. 2021. ArXiv:2109.14084","DOI":"10.18653\/v1\/2021.emnlp-main.544"},{"key":"4321_CR9","first-page":"19948","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"K C Li","year":"2023","unstructured":"Li K C, Wang Y L, Li Y Z, et al. Unmasked teacher: towards training-efficient video foundation models. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023. 19948\u201319960"},{"key":"4321_CR10","first-page":"1632","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"K C Li","year":"2023","unstructured":"Li K C, Wang Y L, He Y N, et al. UniFormerV2: unlocking the potential of image ViTs for video understanding. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023. 1632\u20131643"},{"key":"4321_CR11","first-page":"17959","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"X W Hu","year":"2022","unstructured":"Hu X W, Gan Z, Wang J F, et al. Scaling up vision-language pre-training for image captioning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022. 17959\u201317968"},{"key":"4321_CR12","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"Z Y Dou","year":"2022","unstructured":"Dou Z Y, Xu Y C, Gan Z, et al. An empirical study of training end-to-end vision-and-language transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022"},{"key":"4321_CR13","unstructured":"Shen S, Li L H, Tan H, et al. How much can clip benefit vision-and-language tasks? 2021. ArXiv:2107.06383"},{"key":"4321_CR14","unstructured":"Yao L W, Huang R H, Hou L, et al. FILIP: fine-grained interactive language-image pre-training. 2021. ArXiv:2111.07783"},{"key":"4321_CR15","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"S Chen","year":"2019","unstructured":"Chen S, Austin M, Carl V, et al. VideoBERT: a joint model for video and language representation learning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2019"},{"key":"4321_CR16","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"L C Zhu","year":"2020","unstructured":"Zhu L C, Yang Y. ActBERT: learning global-local video-text representations. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2020"},{"key":"4321_CR17","unstructured":"Wang Y, Li K C, Li Y Z, et al. InternVideo: general video foundation models via generative and discriminative learning. 2022. ArXiv:2212.03191"},{"key":"4321_CR18","unstructured":"Chen G, Xing S, Chen Z, et al. InternVideo-Ego4D: a pack of champion solutions to Ego4D challenges. 2022. ArXiv:2211.09529"},{"key":"4321_CR19","volume-title":"Proceedings of the 35th Conference on Neural Information Processing Systems","author":"Z Tong","year":"2022","unstructured":"Tong Z, Song Y B, Wang J, et al. VideoMAE: masked autoencoders are data-efficient learners for self-supervised video pre-training. In: Proceedings of the 35th Conference on Neural Information Processing Systems, 2022"},{"key":"4321_CR20","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"L M Wang","year":"2023","unstructured":"Wang L M, Huang B K, Zhao Z Y, et al. VideoMAE v2: scaling video masked autoencoders with dual masking. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023"},{"key":"4321_CR21","unstructured":"Li L J, Gan Z, Lin K, et al. LAVENDER: unifying video-language understanding as masked language modeling. 2022. ArXiv:2206.07160"},{"key":"4321_CR22","first-page":"6598","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"J P Wang","year":"2023","unstructured":"Wang J P, Ge Y X, Yan R, et al. All in one: exploring unified video-language pre-training. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023. 6598\u20136608"},{"key":"4321_CR23","unstructured":"Fu T J, Li L J, Gan Z, et al. VIOLET: end-to-end video-language transformers with masked visual-token modeling. 2021. ArXiv:2111.12681"},{"key":"4321_CR24","first-page":"16375","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"R Zellers","year":"2022","unstructured":"Zellers R, Lu J S, Lu X M, et al. MERLOT Reserve: neural script knowledge through vision and language and sound. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2022. 16375\u201316387"},{"key":"4321_CR25","volume-title":"ChatGPT","author":"OpenAI","year":"2022","unstructured":"OpenAI. ChatGPT. https:\/\/openai.com\/blog\/ChatGPT\/. 2022"},{"key":"4321_CR26","unstructured":"Achiam J, Adler S, Agarwal S, et al. GPT-4 technical report. 2023. ArXiv:2303.08774"},{"key":"4321_CR27","first-page":"1877","volume-title":"Proceedings of the 34th International Conference on Neural Information Processing Systems","author":"T Brown","year":"2020","unstructured":"Brown T, Mann B, Ryder N, et al. Language models are few-shot learners. In: Proceedings of the 34th International Conference on Neural Information Processing Systems, 2020. 1877\u20131901"},{"key":"4321_CR28","unstructured":"Touvron H, Lavril T, Izacard G, et al. LLaMA: open and efficient foundation language models. 2023. ArXiv:2302.13971"},{"key":"4321_CR29","unstructured":"Chiang W L, Li Z H, Lin Z, et al. Instruction tuning with GPT-4. 2023. ArXiv:2304.03277"},{"key":"4321_CR30","volume-title":"StableLM: stability AI language models","author":"StableLM contributors","year":"2023","unstructured":"StableLM contributors. StableLM: stability AI language models. https:\/\/github.com\/stability-AI\/stableLM. 2023"},{"key":"4321_CR31","volume-title":"Stanford Alpaca: an instruction-following LLaMA model","author":"T Rohan","year":"2023","unstructured":"Rohan T, Ishaan G, Zhang T Y, et al. Stanford Alpaca: an instruction-following LLaMA model. https:\/\/github.com\/tatsulab\/stanford_alpaca. 2023"},{"key":"4321_CR32","first-page":"27730","volume-title":"Proceedings of the 36th International Conference on Neural Information Processing Systems","author":"L Ouyang","year":"2022","unstructured":"Ouyang L, Wu J, Jiang X, et al. Training language models to follow instructions with human feedback. In: Proceedings of the 36th International Conference on Neural Information Processing Systems, 2022. 27730\u201327744"},{"key":"4321_CR33","unstructured":"Swaroop M, Daniel K, Chitta B, et al. Cross-task generalization via natural language crowdsourcing instructions. 2021. ArXiv:2104.08773"},{"key":"4321_CR34","unstructured":"Chung H W, Hou L, Longpre S, et al. Scaling instruction-finetuned language models. 2022. ArXiv:2210.11416"},{"key":"4321_CR35","unstructured":"Zhang S S, Roller S, Goyal N, et al. OPT: open pre-trained transformer language models. 2022. ArXiv:2205.01068"},{"key":"4321_CR36","volume-title":"MOSS","author":"MOSS contributors","year":"2023","unstructured":"MOSS contributors. MOSS. https:\/\/github.com\/OpenLMLab\/MOSS. 2023"},{"key":"4321_CR37","unstructured":"Zeng A H, Liu X, Du Z X, et al. GLM-130B: an open bilingual pre-trained model. 2022. ArXiv:2210.02414"},{"key":"4321_CR38","unstructured":"Awadalla A, Gao I, Gardner J, et al. OpenFlamingo: an open-source framework for training large autoregressive vision-language models. 2023. ArXiv:2308.01390"},{"key":"4321_CR39","first-page":"12888","volume-title":"Proceedings of the International Conference on Machine Learning","author":"J N Li","year":"2022","unstructured":"Li J N, Li D X, Xiong C M, et al. BLIP: bootstrapping language-image pre-training for unified vision-language understanding and generation. In: Proceedings of the International Conference on Machine Learning, 2022. 12888\u201312900"},{"key":"4321_CR40","unstructured":"Liu Z Y, He Y N, Wang W H, et al. InternChat: solving vision-centric tasks by interacting with chatbots beyond language. 2023. ArXiv:2305.05662"},{"key":"4321_CR41","unstructured":"Wu C F, Yin S M, Qi W Z, et al. Visual ChatGPT: talking, drawing and editing with visual foundation models. 2023. ArXiv:2303.04671"},{"key":"4321_CR42","unstructured":"Yang Z Y, Li L J, Wang J F, et al. MM-REACT: prompting ChatGPT for multimodal reasoning and action. 2023. ArXiv:2303.11381"},{"key":"4321_CR43","unstructured":"Shen Y L, Song K T, Tan X, et al. HuggingGPT: solving AI tasks with ChatGPT and its friends in HuggingFace. 2023. ArXiv:2303.17580"},{"key":"4321_CR44","doi-asserted-by":"crossref","unstructured":"Liang Y B, Wu C F, Song T, et al. Taskmatrix.AI: completing tasks by connecting foundation models with millions of APIs. 2023. ArXiv:2303.16434","DOI":"10.34133\/icomputing.0063"},{"key":"4321_CR45","first-page":"5485","volume":"21","author":"C Raffel","year":"2020","unstructured":"Raffel C, Shazeer N, Roberts A, et al. Exploring the limits of transfer learning with a unified text-to-text transformer. J Mach Learn Res. 2020, 21: 5485\u20135551","journal-title":"J Mach Learn Res"},{"key":"4321_CR46","unstructured":"Wu J L, Wang J F, Yang Z Y, et al. GRiT: a generative region-to-text transformer for object understanding. 2022. ArXiv:2212.00280"},{"key":"4321_CR47","unstructured":"Huang X Y, Zhang Y C, Ma J Y, et al. Tag2Text: guiding vision-language model via image tagging. 2023. ArXiv:2303.05657"},{"key":"4321_CR48","unstructured":"Radford A, Kim J W, Xu T, et al. Robust speech recognition via large-scale weak supervision. 2022. ArXiv:2212.04356"},{"key":"4321_CR49","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"W H Wang","year":"2023","unstructured":"Wang W H, Dai J F, Chen Z, et al. Internimage: exploring large-scale vision foundation models with deformable convolutions. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023"},{"key":"4321_CR50","unstructured":"Yuan L, Chen D D, Chen Y L, et al. Florence: a new foundation model for computer vision. 2021. ArXiv:2111.11432"},{"key":"4321_CR51","first-page":"19730","volume-title":"Proceedings of the International Conference on Machine Learning","author":"J N Li","year":"2023","unstructured":"Li J N, Li D X, Savarese S, et al. BLIP-2: bootstrapping language-image pre-training with frozen image encoders and large language models. In: Proceedings of the International Conference on Machine Learning, 2023. 19730\u201319742"},{"key":"4321_CR52","unstructured":"Sun Q, Fang Y X, Wu L, et al. EVA-CLIP: improved training techniques for clip at scale. 2023. ArXiv:2303.15389"},{"key":"4321_CR53","first-page":"1728","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"M Bain","year":"2021","unstructured":"Bain M, Nagrani A, Varol G, et al. Frozen in time: a joint video and image encoder for end-to-end retrieval. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021. 1728\u20131738"},{"key":"4321_CR54","unstructured":"Chen X L, Fang H, Lin T Y, et al. Microsoft COCO captions: data collection and evaluation server. 2015. ArXiv:1504.00325"},{"key":"4321_CR55","doi-asserted-by":"publisher","first-page":"32","DOI":"10.1007\/s11263-016-0981-7","volume":"123","author":"R Krishna","year":"2017","unstructured":"Krishna R, Zhu Y, Groth O, et al. Visual genome: connecting language and vision using crowdsourced dense image annotations. Int J Comput Vis, 2017, 123: 32\u201373","journal-title":"Int J Comput Vis"},{"key":"4321_CR56","first-page":"1143","volume-title":"Proceedings of the 25th International Conference on Neural Information Processing Systems","author":"V Ordonez","year":"2011","unstructured":"Ordonez V, Kulkarni G, Berg T L. Im2Text: describing images using 1 million captioned photographs. In: Proceedings of the 25th International Conference on Neural Information Processing Systems, 2011. 1143\u20131151"},{"key":"4321_CR57","volume-title":"Proceedings of the Annual Meeting of the Association for Computational Linguistics","author":"P Sharma","year":"2018","unstructured":"Sharma P, Ding N, Goodman S, et al. Conceptual captions: a cleaned, hypernymed, image alt-text dataset for automatic image captioning. In: Proceedings of the Annual Meeting of the Association for Computational Linguistics, 2018"},{"key":"4321_CR58","first-page":"3558","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"S Changpinyo","year":"2021","unstructured":"Changpinyo S, Sharma P, Ding N, et al. Conceptual 12M: pushing web-scale image-text pre-training to recognize long-tail visual concepts. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2021. 3558\u20133568"},{"key":"4321_CR59","first-page":"9777","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition","author":"J B Xiao","year":"2021","unstructured":"Xiao J B, Shang X D, Yao A, et al. NExT-QA: next phase of question-answering to explaining temporal actions. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2021. 9777\u20139786"},{"key":"4321_CR60","doi-asserted-by":"crossref","unstructured":"Lei J, Yu L C, Bansal M, et al. TVQA: localized, compositional video question answering. 2018. ArXiv:1809.01696","DOI":"10.18653\/v1\/D18-1167"},{"key":"4321_CR61","unstructured":"Zhou L, Louis N, Corso J J. Weakly-supervised video object grounding from text by loss weighting and object interaction. 2018. ArXiv:1805.02834"},{"key":"4321_CR62","first-page":"11846","volume-title":"Proceedings of the 34th International Conference on Neural Information Processing Systems","author":"J Lei","year":"2021","unstructured":"Lei J, Berg T L, Bansal M. Detecting moments and highlights in videos via natural language queries. In: Proceedings of the 34th International Conference on Neural Information Processing Systems, 2021. 11846\u201311858"},{"key":"4321_CR63","first-page":"5267","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision","author":"J Y Gao","year":"2017","unstructured":"Gao J Y, Sun C, Yang Z H, et al. TALL: temporal activity localization via language query. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2017. 5267\u20135275"},{"key":"4321_CR64","unstructured":"Li B H, Wang R, Wang G Z, et al. Seed-bench: benchmarking multimodal LLMs with generative comprehension. 2023. ArXiv:2307.16125"},{"key":"4321_CR65","unstructured":"Li K C, Wang Y L, He Y N, et al. MVBench: a comprehensive multi-modal video understanding benchmark. 2023. ArXiv:2311.17005"},{"key":"4321_CR66","unstructured":"Ning M N, Zhu B, Xie Y J, et al. Video-bench: a comprehensive benchmark and toolkit for evaluating video-based large language models. 2023. ArXiv:2311.16103"},{"key":"4321_CR67","unstructured":"Khattak M U, Naeem M F, Hassan J, et al. Complex video reasoning and robustness evaluation suite for video-LMMs. 2024. ArXiv:2405.03690"},{"key":"4321_CR68","unstructured":"Maaz M, Rasheed H, Khan S, et al. Video-ChatGPT: towards detailed video understanding via large vision and language models. 2023. ArXiv:2306.05424"},{"key":"4321_CR69","unstructured":"Song E X, Chai W H, Wang G H, et al. MovieChat: from dense token to sparse memory for long video understanding. 2023. ArXiv:2307.16449"},{"key":"4321_CR70","doi-asserted-by":"crossref","unstructured":"Zhang H, Li X, Bing L D. Video-LLaMA: an instruction-tuned audio-visual language model for video understanding. 2023. ArXiv:2306.02858","DOI":"10.18653\/v1\/2023.emnlp-demo.49"},{"key":"4321_CR71","unstructured":"Luo R P, Zhao Z W, Yang M, et al. Valley: video assistant with large language model enhanced ability. 2023. ArXiv:2306.07207"},{"key":"4321_CR72","doi-asserted-by":"crossref","unstructured":"Jin P, Takanobu R, Zhang C W, et al. Chat-UniVi: unified visual representation empowers large language models with image and video understanding. 2023. ArXiv:2311.08046","DOI":"10.1109\/CVPR52733.2024.01300"},{"key":"4321_CR73","unstructured":"Lin B, Zhu B, Ye Y, et al. Video-LLaVa: learning united visual representation by alignment before projection. 2023. ArXiv:2311.10122"},{"key":"4321_CR74","doi-asserted-by":"crossref","unstructured":"Li Y W, Wang C Y, Jia J Y. LLaMA-VID: an image is worth 2 tokens in large language models. 2023. ArXiv:2311.17043","DOI":"10.1007\/978-3-031-72952-2_19"},{"key":"4321_CR75","unstructured":"Ren S S, Yao L L, Li S C, et al. TimeChat: a time-sensitive multimodal large language model for long video understanding. 2023. ArXiv:2312.02051"}],"container-title":["Science China Information Sciences"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11432-024-4321-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11432-024-4321-9\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11432-024-4321-9.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,1]],"date-time":"2025-12-01T08:47:26Z","timestamp":1764578846000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11432-024-4321-9"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,9,25]]},"references-count":75,"journal-issue":{"issue":"10","published-print":{"date-parts":[[2025,10]]}},"alternative-id":["4321"],"URL":"https:\/\/doi.org\/10.1007\/s11432-024-4321-9","relation":{},"ISSN":["1674-733X","1869-1919"],"issn-type":[{"value":"1674-733X","type":"print"},{"value":"1869-1919","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,9,25]]},"assertion":[{"value":"15 May 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 September 2024","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 November 2024","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 September 2025","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"200102"}}