{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,24]],"date-time":"2026-06-24T00:13:34Z","timestamp":1782260014003,"version":"3.54.5"},"reference-count":81,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"11","license":[{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,11,1]],"date-time":"2024-11-01T00:00:00Z","timestamp":1730419200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["52302397"],"award-info":[{"award-number":["52302397"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["52202391"],"award-info":[{"award-number":["52202391"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U20A20155"],"award-info":[{"award-number":["U20A20155"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100002858","name":"China Postdoctoral Science Foundation","doi-asserted-by":"publisher","award":["2023M730173"],"award-info":[{"award-number":["2023M730173"]}],"id":[{"id":"10.13039\/501100002858","id-type":"DOI","asserted-by":"publisher"}]},{"name":"National Key Research and Development Program of China","award":["2022YFC3803700"],"award-info":[{"award-number":["2022YFC3803700"]}]},{"name":"Beihang Frontier Cross Foundation","award":["501QYJC2023113003"],"award-info":[{"award-number":["501QYJC2023113003"]}]},{"name":"Beijing Natural Science Foundation","award":["4222021"],"award-info":[{"award-number":["4222021"]}]},{"DOI":"10.13039\/501100015973","name":"Zhuoyue Bairen Postdoctorate Program of Beihang University","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100015973","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Circuits Syst. Video Technol."],"published-print":{"date-parts":[[2024,11]]},"DOI":"10.1109\/tcsvt.2024.3409453","type":"journal-article","created":{"date-parts":[[2024,6,4]],"date-time":"2024-06-04T21:20:50Z","timestamp":1717536050000},"page":"10538-10550","source":"Crossref","is-referenced-by-count":15,"title":["CFMMC-Align: Coarse-Fine Multi-Modal Contrastive Alignment Network for Traffic Event Video Question Answering"],"prefix":"10.1109","volume":"34","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9828-1028","authenticated-orcid":false,"given":"Kan","family":"Guo","sequence":"first","affiliation":[{"name":"Beijing Advanced Innovation Center for Big Data and Brain Computing, Beijing Key Laboratory for Cooperative Vehicle Infrastructure Systems and Safety Control, School of Transportation Science and Engineering, Beihang University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7796-5650","authenticated-orcid":false,"given":"Daxin","family":"Tian","sequence":"additional","affiliation":[{"name":"Beijing Advanced Innovation Center for Big Data and Brain Computing, Beijing Key Laboratory for Cooperative Vehicle Infrastructure Systems and Safety Control, School of Transportation Science and Engineering, Beihang University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0440-438X","authenticated-orcid":false,"given":"Yongli","family":"Hu","sequence":"additional","affiliation":[{"name":"Beijing Key Laboratory of Multimedia and Intelligent Software Technology, Faculty of Information Technology, Beijing University of Technology, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9051-8852","authenticated-orcid":false,"given":"Chunmian","family":"Lin","sequence":"additional","affiliation":[{"name":"Beijing Advanced Innovation Center for Big Data and Brain Computing, Beijing Key Laboratory for Cooperative Vehicle Infrastructure Systems and Safety Control, School of Transportation Science and Engineering, Beihang University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0872-384X","authenticated-orcid":false,"given":"Yanfeng","family":"Sun","sequence":"additional","affiliation":[{"name":"Beijing Key Laboratory of Multimedia and Intelligent Software Technology, Faculty of Information Technology, Beijing University of Technology, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5331-6162","authenticated-orcid":false,"given":"Jianshan","family":"Zhou","sequence":"additional","affiliation":[{"name":"Beijing Advanced Innovation Center for Big Data and Brain Computing, Beijing Key Laboratory for Cooperative Vehicle Infrastructure Systems and Safety Control, School of Transportation Science and Engineering, Beihang University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5931-8310","authenticated-orcid":false,"given":"Xuting","family":"Duan","sequence":"additional","affiliation":[{"name":"Beijing Advanced Innovation Center for Big Data and Brain Computing, Beijing Key Laboratory for Cooperative Vehicle Infrastructure Systems and Safety Control, School of Transportation Science and Engineering, Beihang University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9803-0256","authenticated-orcid":false,"given":"Junbin","family":"Gao","sequence":"additional","affiliation":[{"name":"Discipline of Business Analytics, The University of Sydney Business School, The University of Sydney, Sydney, NSW, Australia"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3121-1823","authenticated-orcid":false,"given":"Baocai","family":"Yin","sequence":"additional","affiliation":[{"name":"Beijing Key Laboratory of Multimedia and Intelligent Software Technology, Faculty of Information Technology, Beijing University of Technology, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","volume-title":"Reasoning About Change: Time and Causation From the Standpoint of Artificial Intelligence","author":"Shoham","year":"1987"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.501"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2017\/280"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.149"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr.2017.778"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00688"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33018658"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6767"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01527"},{"key":"ref10","first-page":"124","article-title":"Zero-shot video question answering via frozen bidirectional language models","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"35","author":"Yang"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2015.2496795"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2018.2876614"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2023.3264573"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/cvprw56347.2022.00355"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2022.3171553"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00103"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2023.3295058"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2021.3081999"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01032"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01406"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02076"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2021.3086104"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2020.3013254"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01473"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00227"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00641"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00975"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2023.3284038"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1016\/j.aiopen.2021.01.001"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01282"},{"key":"ref33","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"Proc. Int. Conf. Mach. Learn.","volume":"139","author":"Radford"},{"key":"ref34","first-page":"225","article-title":"Linear Hinge loss and average margin","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"11","author":"Gentile"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"ref37","article-title":"LAION-400M: Open dataset of CLIP-filtered 400 million image-text pairs","author":"Schuhmann","year":"2021","journal-title":"arXiv:2111.02114"},{"key":"ref38","article-title":"The kinetics human action video dataset","author":"Kay","year":"2017","journal-title":"arXiv:1705.06950"},{"key":"ref39","article-title":"A short note about kinetics-600","author":"Carreira","year":"2018","journal-title":"arXiv:1808.01340"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00175"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1145\/3077136.3080655"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/WACV45572.2020.9093596"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00171"},{"key":"ref44","first-page":"297","article-title":"Noise-contrastive estimation: A new estimation principle for unnormalized statistical models","volume-title":"Proc. 13th Int. Conf. Artif. Intell. Statist.","author":"Gutmann"},{"key":"ref45","article-title":"An image is worth 16\u00d716 words: Transformers for image recognition at scale","author":"Dosovitskiy","year":"2020","journal-title":"arXiv:2010.11929"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2023.3283289"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2023.3234311"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2023.3243829"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/IJCNN54540.2023.10191491"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2022.3165934"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2023.3281448"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2022.3212463"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.312"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/d18-1167"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00999"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00210"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01012"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00173"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6737"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2021.3097171"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2022.3142526"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00756"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00725"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i14.17556"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.544"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01660"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01589"},{"key":"ref69","first-page":"12888","article-title":"BLIP: Bootstrapping language-image pre-training for unified vision-language understanding and generation","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Li"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00293"},{"key":"ref71","first-page":"9929","article-title":"Understanding contrastive representation learning through alignment and uniformity on the hypersphere","volume-title":"Proc. 37th Int. Conf. Mach. Learn.","volume":"119","author":"Wang"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.552"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-long.197"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.5555\/3524938.3525087"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01549"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547910"},{"key":"ref78","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00965"},{"key":"ref79","first-page":"2953","article-title":"Exploring models and data for image question answering","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"28","author":"Ren"},{"key":"ref80","article-title":"LoRA: Low-rank adaptation of large language models","author":"Hu","year":"2021","journal-title":"arXiv:2106.09685"},{"key":"ref81","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-023-01891-x"}],"container-title":["IEEE Transactions on Circuits and Systems for Video Technology"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/76\/10768433\/10547379.pdf?arnumber=10547379","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,11,27]],"date-time":"2024-11-27T19:48:51Z","timestamp":1732736931000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10547379\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11]]},"references-count":81,"journal-issue":{"issue":"11"},"URL":"https:\/\/doi.org\/10.1109\/tcsvt.2024.3409453","relation":{},"ISSN":["1051-8215","1558-2205"],"issn-type":[{"value":"1051-8215","type":"print"},{"value":"1558-2205","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11]]}}}