{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,7]],"date-time":"2026-05-07T16:25:10Z","timestamp":1778171110359,"version":"3.51.4"},"reference-count":76,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"4","license":[{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T00:00:00Z","timestamp":1775001600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U21B2024"],"award-info":[{"award-number":["U21B2024"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62425307"],"award-info":[{"award-number":["62425307"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62472303"],"award-info":[{"award-number":["62472303"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62402334"],"award-info":[{"award-number":["62402334"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Circuits Syst. Video Technol."],"published-print":{"date-parts":[[2026,4]]},"DOI":"10.1109\/tcsvt.2025.3639574","type":"journal-article","created":{"date-parts":[[2025,12,3]],"date-time":"2025-12-03T18:44:30Z","timestamp":1764787470000},"page":"5384-5397","source":"Crossref","is-referenced-by-count":1,"title":["Constituency-Tree-Induced Vision\u2013Language Alignment for Multimodal Large Language Models"],"prefix":"10.1109","volume":"36","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7980-7901","authenticated-orcid":false,"given":"Yingchen","family":"Zhai","sequence":"first","affiliation":[{"name":"School of Electrical and Information Engineering, Tianjin University, Tianjin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7526-4356","authenticated-orcid":false,"given":"Ning","family":"Xu","sequence":"additional","affiliation":[{"name":"School of Electrical and Information Engineering, Tianjin University, Tianjin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7635-0961","authenticated-orcid":false,"given":"Hongshuo","family":"Tian","sequence":"additional","affiliation":[{"name":"School of Electrical and Information Engineering, Tianjin University, Tianjin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8788-1725","authenticated-orcid":false,"given":"Bolun","family":"Zheng","sequence":"additional","affiliation":[{"name":"Institute of Information and Control, Hangzhou Dianzi University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1204-0512","authenticated-orcid":false,"given":"Chenggang","family":"Yan","sequence":"additional","affiliation":[{"name":"Institute of Information and Control, Hangzhou Dianzi University, Hangzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-5037-6989","authenticated-orcid":false,"given":"Jinbo","family":"Cao","sequence":"additional","affiliation":[{"name":"30th Research Institute of China Electronics Technology Group Corporation, Chengdu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8487-8520","authenticated-orcid":false,"given":"Rongbao","family":"Kang","sequence":"additional","affiliation":[{"name":"30th Research Institute of China Electronics Technology Group Corporation, Chengdu, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5755-9145","authenticated-orcid":false,"given":"An-An","family":"Liu","sequence":"additional","affiliation":[{"name":"School of Electrical and Information Engineering, Tianjin University, Tianjin, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00904"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.52202\/068431-1723"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.eacl-main.262"},{"key":"ref4","first-page":"32897","article-title":"VLMo: Unified vision-language pre-training with mixture-of-modality-experts","volume-title":"Proc. NeurIPS","author":"Wang"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2023.3291379"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02506"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02506"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01750"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58577-8_7"},{"key":"ref10","volume-title":"Vicuna: An Open-Source Chatbot Impressing GPT-4 With 90%* ChatGPT Quality","author":"Chiang et al","year":"2023"},{"key":"ref11","article-title":"Scaling instruction-finetuned language models","author":"Chung","year":"2022","journal-title":"arXiv:2210.11416"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2142"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01396"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2023.3328298"},{"key":"ref15","article-title":"An image is worth 16$\\times$\n16 words: Transformers for image recognition at scale","author":"Dosovitskiy","journal-title":"arXiv:2010.11929"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N19-1116"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01422"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01855"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00291"},{"key":"ref20","article-title":"Training-free structured diffusion guidance for compositional text-to-image synthesis","author":"Feng","year":"2022","journal-title":"arXiv:2212.05032"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2023.3284979"},{"key":"ref22","first-page":"127","article-title":"Tree-structured reinforcement learning for sequential object localization","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"29","author":"Jie"},{"key":"ref23","first-page":"56615","article-title":"Exploring question decomposition for zero-shot VQA","volume-title":"Proc. NeurIPS","author":"Khan"},{"key":"ref24","first-page":"19730","article-title":"BLIP-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","volume-title":"Proc. ICML","author":"Li"},{"key":"ref25","first-page":"12888","article-title":"BLIP: Bootstrapping language-image pre-training for unified vision-language understanding and generation","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Li"},{"key":"ref26","first-page":"9694","article-title":"Align before fuse: Vision and language representation learning with momentum distillation","volume-title":"Proc. NeurIPS","author":"Li"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01303"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01069"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58577-8_8"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.52202\/075280-1516"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2024.3399933"},{"key":"ref34","article-title":"Grounding DINO: Marrying DINO with grounded pre-training for open-set object detection","author":"Liu","year":"2023","journal-title":"arXiv:2303.05499"},{"key":"ref35","article-title":"Decoupled weight decay regularization","author":"Loshchilov","journal-title":"arXiv:1711.05101"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01516"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.21236\/ADA273556"},{"key":"ref38","first-page":"17359","article-title":"Locating and editing factual associations in GPT","volume-title":"Proc. NeurIPS","author":"Meng"},{"key":"ref39","article-title":"Simple open-vocabulary object detection with vision transformers","author":"Minderer","year":"2022","journal-title":"arXiv:2205.06230"},{"key":"ref40","article-title":"ClipCap: CLIP prefix for image captioning","author":"Mokady","year":"2021","journal-title":"arXiv:2111.09734"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.findings-emnlp.65"},{"key":"ref42","article-title":"DINOv2: Learning robust visual features without supervision","author":"Oquab","year":"2023","journal-title":"arXiv:2304.07193"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW60793.2023.00307"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.60"},{"key":"ref45","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"Proc. Int. Conf. Mach. Learn.","volume":"139","author":"Radford"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW60793.2023.00308"},{"key":"ref47","article-title":"Ensemble distillation for unsupervised constituency parsing","author":"Shayegh","year":"2023","journal-title":"arXiv:2310.01717"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1514"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1145\/3605781"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.findings-emnlp.153"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.52202\/079017-2771"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00914"},{"key":"ref53","article-title":"LLaMA: Open and efficient foundation language models","author":"Touvron","year":"2023","journal-title":"arXiv:2302.13971"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01747"},{"key":"ref55","article-title":"Unsupervised vision-language grammar induction with shared structure modeling","volume-title":"Proc. ICLR","author":"Wan"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.emnlp-main.25"},{"key":"ref57","first-page":"61501","article-title":"VisionLLM: Large language model is also an open-ended decoder for vision-centric tasks","volume-title":"Proc. NeurIPS","author":"Wang"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-emnlp.189"},{"key":"ref59","article-title":"Adaptively clustering neighbor elements for image captioning","author":"Wang","year":"2023","journal-title":"arXiv:2301.01955"},{"key":"ref60","article-title":"SimVLM: Simple visual language model pretraining with weak supervision","author":"Wang","journal-title":"arXiv:2108.10904"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49660.2025.10890039"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-acl.432"},{"key":"ref63","first-page":"21687","article-title":"Strongly incremental constituency parsing with graph neural networks","volume-title":"Proc. NeurIPS","author":"Yang"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00220"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00271"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1093\/nsr\/nwae403"},{"key":"ref67","article-title":"DINO: DETR with improved DeNoising anchor boxes for end-to-end object detection","author":"Zhang","journal-title":"arXiv:2203.03605"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2023.3243725"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00553"},{"key":"ref70","article-title":"LLaMA-adapter: Efficient fine-tuning of language models with zero-init attention","author":"Zhang","year":"2023","journal-title":"arXiv:2303.16199"},{"key":"ref71","article-title":"OPT: Open pre-trained transformer language models","author":"Zhang","year":"2022","journal-title":"arXiv:2205.01068"},{"key":"ref72","article-title":"GPT4RoI: Instruction tuning large language model on region-of-interest","author":"Zhang","year":"2023","journal-title":"arXiv:2307.03601"},{"key":"ref73","first-page":"6634","article-title":"Multi-modal dependency tree for video captioning","volume-title":"Proc. NIPS","volume":"34","author":"Zhao"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1230"},{"key":"ref75","article-title":"MiniGPT-4: Enhancing vision-language understanding with advanced large language models","author":"Zhu","journal-title":"arXiv:2304.10592"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2016.2606648"}],"container-title":["IEEE Transactions on Circuits and Systems for Video Technology"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/76\/11475579\/11275819.pdf?arnumber=11275819","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,4,7]],"date-time":"2026-04-07T20:03:26Z","timestamp":1775592206000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11275819\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,4]]},"references-count":76,"journal-issue":{"issue":"4"},"URL":"https:\/\/doi.org\/10.1109\/tcsvt.2025.3639574","relation":{},"ISSN":["1051-8215","1558-2205"],"issn-type":[{"value":"1051-8215","type":"print"},{"value":"1558-2205","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,4]]}}}