{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,24]],"date-time":"2026-03-24T05:21:59Z","timestamp":1774329719914,"version":"3.50.1"},"reference-count":52,"publisher":"Springer Science and Business Media LLC","issue":"16","license":[{"start":{"date-parts":[[2025,11,1]],"date-time":"2025-11-01T00:00:00Z","timestamp":1761955200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,11,1]],"date-time":"2025-11-01T00:00:00Z","timestamp":1761955200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"the Key Program of Chongqing Municipal Education Commission Planning Project","award":["24SKGH055"],"award-info":[{"award-number":["24SKGH055"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2025,11]]},"DOI":"10.1007\/s10489-025-06980-7","type":"journal-article","created":{"date-parts":[[2025,11,8]],"date-time":"2025-11-08T06:01:04Z","timestamp":1762581664000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["MDIF: A multimodal dynamic inference framework for traffic video question answering"],"prefix":"10.1007","volume":"55","author":[{"given":"Wenhao","family":"Guo","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1240-4470","authenticated-orcid":false,"given":"Lingling","family":"Zi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xin","family":"Cong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,11,8]]},"reference":[{"key":"6980_CR1","unstructured":"Krizhevsky A, Sutskever I, Hinton GE (2012) Imagenet classification with deep convolutional neural networks. Advances in neural information processing systems, vol 25"},{"key":"6980_CR2","doi-asserted-by":"crossref","unstructured":"Jang Y, Song Y, Yu Y et\u00a0al (2017) Tgif-qa: toward spatio-temporal reasoning in visual question answering. In: IEEE conference on computer vision and pattern recognition, pp 2758\u20132766","DOI":"10.1109\/CVPR.2017.149"},{"key":"6980_CR3","doi-asserted-by":"publisher","first-page":"1385","DOI":"10.1007\/s11263-019-01189-x","volume":"127","author":"Y Jang","year":"2019","unstructured":"Jang Y, Song Y, Kim CD et al (2019) Video question answering with spatio-temporal reasoning. Int J Comput Vision 127:1385\u20131412","journal-title":"Int J Comput Vision"},{"key":"6980_CR4","doi-asserted-by":"crossref","unstructured":"Niu Y, Tang K, Zhang H et\u00a0al (2021) Counterfactual vqa: a cause-effect look at language bias. In: IEEE\/CVF conference on computer vision and pattern recognition, pp 12700\u201312710","DOI":"10.1109\/CVPR46437.2021.01251"},{"key":"6980_CR5","doi-asserted-by":"crossref","unstructured":"Cai J, Yuan C, Shi C et\u00a0al (2021) Feature augmented memory with global attention network for videoqa. In: Twenty-ninth international conference on international joint conferences on artificial intelligence, pp 998\u20131004","DOI":"10.24963\/ijcai.2020\/139"},{"key":"6980_CR6","doi-asserted-by":"publisher","unstructured":"Huang D, Chen P, Zeng R et\u00a0al (2020) Location-aware graph convolutional networks for video question answering. In: AAAI Conference on Artificial Intelligence, pp 11021\u201311028. https:\/\/doi.org\/10.1609\/aaai.v34i07.6737","DOI":"10.1609\/aaai.v34i07.6737"},{"key":"6980_CR7","unstructured":"Cao B, Xia Y, Ding Y et\u00a0al (2024) Predictive dynamic fusion. arXiv:2406.04802"},{"key":"6980_CR8","volume-title":"Causal inference in statistics: a primer","author":"J Pearl","year":"2016","unstructured":"Pearl J, Glymour M, Jewell NP (2016) Causal inference in statistics: a primer. John Wiley & Sons"},{"key":"6980_CR9","doi-asserted-by":"publisher","unstructured":"Xu D, Zhao Z, Xiao J et\u00a0al (2017) Video question answering via gradually refined attention over appearance and motion. In: Proceedings of the 25th ACM international conference on multimedia, pp 1645\u20131653, https:\/\/doi.org\/10.1145\/3123266.3123427","DOI":"10.1145\/3123266.3123427"},{"key":"6980_CR10","doi-asserted-by":"crossref","unstructured":"Le TM, Le V, Venkatesh S et\u00a0al (2020) Hierarchical conditional relation networks for video question answering. In: IEEE\/CVF conference on computer vision and pattern recognition, pp 9972\u20139981","DOI":"10.1109\/CVPR42600.2020.00999"},{"key":"6980_CR11","doi-asserted-by":"publisher","first-page":"3950","DOI":"10.1109\/TMM.2022.3169065","volume":"25","author":"T Qian","year":"2023","unstructured":"Qian T, Chen J, Chen S et al (2023) Scene graph refinement network for visual question answering. IEEE Trans Multimedia 25:3950\u20133961. https:\/\/doi.org\/10.1109\/TMM.2022.3169065","journal-title":"IEEE Trans Multimedia"},{"key":"6980_CR12","doi-asserted-by":"crossref","unstructured":"Feichtenhofer C (2020) X3d: expanding architectures for efficient video recognition. In: IEEE\/CVF conference on computer vision and pattern recognition, pp 203\u2013213","DOI":"10.1109\/CVPR42600.2020.00028"},{"key":"6980_CR13","doi-asserted-by":"crossref","unstructured":"Lin J, Gan C, Han S (2019) Tsm: temporal shift module for efficient video understanding. In: IEEE\/CVF international conference on computer vision, pp 7083\u20137093","DOI":"10.1109\/ICCV.2019.00718"},{"issue":"21","key":"6980_CR14","doi-asserted-by":"publisher","first-page":"10709","DOI":"10.1007\/s10489-024-05775-6","volume":"54","author":"M Qaraqe","year":"2024","unstructured":"Qaraqe M, Yang YD, Varghese EB et al (2024) Crowd behavior detection: leveraging video swin transformer for crowd size and violence level analysis. Appl Intell 54(21):10709\u201310730. https:\/\/doi.org\/10.1007\/s10489-024-05775-6","journal-title":"Appl Intell"},{"key":"6980_CR15","doi-asserted-by":"publisher","unstructured":"Adegun A, Viriri S, Tapamo JR (2024) Automated classification of remote sensing satellite images using deep learning based vision transformer. Appl Intell 1\u201320. https:\/\/doi.org\/10.1007\/s10489-024-05818-y","DOI":"10.1007\/s10489-024-05818-y"},{"key":"6980_CR16","doi-asserted-by":"crossref","unstructured":"Ibh M, Grasshof S, Witzner D et\u00a0al (2023) Tempose: a new skeleton-based transformer model designed for fine-grained motion recognition in badminton. In: IEEE\/CVF conference on computer vision and pattern recognition, pp 5199\u20135208","DOI":"10.1109\/CVPRW59228.2023.00548"},{"key":"6980_CR17","doi-asserted-by":"publisher","first-page":"7943","DOI":"10.1109\/TMM.2022.3232034","volume":"25","author":"F Wu","year":"2023","unstructured":"Wu F, Wang Q, Bian J et al (2023) A survey on video action recognition in sports: datasets, methods and applications. IEEE Trans Multimedia 25:7943\u20137966. https:\/\/doi.org\/10.1109\/TMM.2022.3232034","journal-title":"IEEE Trans Multimedia"},{"issue":"8","key":"6980_CR18","doi-asserted-by":"publisher","first-page":"1909","DOI":"10.1007\/s11263-023-01790-1","volume":"131","author":"J Mao","year":"2023","unstructured":"Mao J, Shi S, Wang X et al (2023) 3d object detection for autonomous driving: a comprehensive survey. Int J Comput Vision 131(8):1909\u20131963. https:\/\/doi.org\/10.1007\/s11263-023-01790-1","journal-title":"Int J Comput Vision"},{"key":"6980_CR19","doi-asserted-by":"publisher","unstructured":"Mavroudi E, Haro BB, Vidal R (2020) Representation learning on visual-symbolic graphs for video understanding. In: European conference on computer vision, Springer, pp 71\u201390. https:\/\/doi.org\/10.1007\/978-3-030-58526-6_5","DOI":"10.1007\/978-3-030-58526-6_5"},{"key":"6980_CR20","doi-asserted-by":"crossref","unstructured":"Pan J, Chen S, Shou MZ et\u00a0al (2021) Actor-context-actor relation network for spatio-temporal action localization. In: IEEE\/CVF conference on computer vision and pattern recognition, pp 464\u2013474","DOI":"10.1109\/CVPR46437.2021.00053"},{"issue":"3","key":"6980_CR21","doi-asserted-by":"publisher","first-page":"3933","DOI":"10.1109\/TPAMI.2022.3180025","volume":"45","author":"W Wang","year":"2023","unstructured":"Wang W, Gao J, Xu C (2023) Weakly-supervised video object grounding via causal intervention. IEEE Trans Pattern Anal Mach Intell 45(3):3933\u20133948. https:\/\/doi.org\/10.1109\/TPAMI.2022.3180025","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"6980_CR22","doi-asserted-by":"publisher","unstructured":"Wang T, Li Y, Kang B et\u00a0al (2020) The devil is in classification: a simple framework for long-tail instance segmentation. In: Computer vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part XIV 16, Springer, pp 728\u2013744. https:\/\/doi.org\/10.1007\/978-3-030-58568-6_43","DOI":"10.1007\/978-3-030-58568-6_43"},{"issue":"11","key":"6980_CR23","doi-asserted-by":"publisher","first-page":"12996","DOI":"10.1109\/TPAMI.2021.3121705","volume":"45","author":"X Yang","year":"2023","unstructured":"Yang X, Zhang H, Cai J (2023) Deconfounded image captioning: a causal retrospect. IEEE Trans Pattern Anal Mach Intell 45(11):12996\u201313010. https:\/\/doi.org\/10.1109\/TPAMI.2021.3121705","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"issue":"6","key":"6980_CR24","doi-asserted-by":"publisher","first-page":"485","DOI":"10.1007\/s11633-022-1362-z","volume":"19","author":"Y Liu","year":"2022","unstructured":"Liu Y, Wei YS, Yan H et al (2022) Causal reasoning meets visual representation learning: a prospective study. Mach Intell Res 19(6):485\u2013511. https:\/\/doi.org\/10.1007\/s11633-022-1362-z","journal-title":"Mach Intell Res"},{"key":"6980_CR25","doi-asserted-by":"crossref","unstructured":"Nag S, Min K, Tripathi S et\u00a0al (2023) Unbiased scene graph generation in videos. In: IEEE\/CVF conference on computer vision and pattern recognition, pp 22803\u201322813","DOI":"10.1109\/CVPR52729.2023.02184"},{"key":"6980_CR26","doi-asserted-by":"crossref","unstructured":"Yang Z, Wang J, Gan Z et\u00a0al (2023) Reco: Region-controlled text-to-image generation. In: IEEE\/CVF conference on computer vision and pattern recognition, pp 14246\u201314255","DOI":"10.1109\/CVPR52729.2023.01369"},{"key":"6980_CR27","doi-asserted-by":"publisher","unstructured":"Wei Y, Liu Y, Yan H et\u00a0al (2023) Visual causal scene refinement for video question answering. In: Proceedings of the 31st ACM international conference on multimedia. Association for Computing Machinery, New York, NY, USA, MM \u201923, pp 377\u2013386. https:\/\/doi.org\/10.1145\/3581783.3611873","DOI":"10.1145\/3581783.3611873"},{"issue":"10","key":"6980_CR28","doi-asserted-by":"publisher","first-page":"11624","DOI":"10.1109\/TPAMI.2023.3284038","volume":"45","author":"Y Liu","year":"2023","unstructured":"Liu Y, Li G, Lin L (2023) Cross-modal causal relational reasoning for event-level visual question answering. IEEE Trans Pattern Anal Mach Intell 45(10):11624\u201311641. https:\/\/doi.org\/10.1109\/TPAMI.2023.3284038","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"issue":"11","key":"6980_CR29","doi-asserted-by":"publisher","first-page":"10538","DOI":"10.1109\/TCSVT.2024.3409453","volume":"34","author":"K Guo","year":"2024","unstructured":"Guo K, Tian D, Hu Y et al (2024) Cfmmc-align: coarse-fine multi-modal contrastive alignment network for traffic event video question answering. IEEE Trans Circuits Syst Video Technol 34(11):10538\u201310550. https:\/\/doi.org\/10.1109\/TCSVT.2024.3409453","journal-title":"IEEE Trans Circuits Syst Video Technol"},{"key":"6980_CR30","doi-asserted-by":"crossref","unstructured":"Fan C, Zhang X, Zhang S et\u00a0al (2019) Heterogeneous memory enhanced multimodal attention model for video question answering. In: IEEE\/CVF conference on computer vision and pattern recognition, pp 1999\u20132007","DOI":"10.1109\/CVPR.2019.00210"},{"key":"6980_CR31","doi-asserted-by":"crossref","unstructured":"Zang C, Wang H, Pei M et\u00a0al (2023) Discovering the real association: multimodal causal reasoning in video question answering. In: IEEE\/CVF conference on computer vision and pattern recognition, pp 19027\u201319036","DOI":"10.1109\/CVPR52729.2023.01824"},{"key":"6980_CR32","doi-asserted-by":"crossref","unstructured":"Ahmad M, Park G, Park D et\u00a0al (2023) Mmtf: multi-modal temporal fusion for commonsense video question answering. In: IEEE\/CVF international conference on computer vision, pp 4657\u20134662","DOI":"10.1109\/ICCVW60793.2023.00502"},{"issue":"2","key":"6980_CR33","doi-asserted-by":"publisher","first-page":"1615","DOI":"10.1109\/TCSVT.2024.3475510","volume":"35","author":"T Yu","year":"2025","unstructured":"Yu T, Fu K, Wang S et al (2025) Prompting video-language foundation models with domain-specific fine-grained heuristics for video question answering. IEEE Trans Circuits Syst Video Technol 35(2):1615\u20131630. https:\/\/doi.org\/10.1109\/TCSVT.2024.3475510","journal-title":"IEEE Trans Circuits Syst Video Technol"},{"key":"6980_CR34","unstructured":"Zhao H, Cai Z, Si S et\u00a0al (2023) Mmicl: empowering vision-language model with multi-modal in-context learning. arXiv:2309.07915"},{"key":"6980_CR35","doi-asserted-by":"crossref","unstructured":"Bitton Y, Stanovsky G, Elhadad M et\u00a0al (2021) Data efficient masked language modeling for vision and language. arXiv:2109.02040","DOI":"10.18653\/v1\/2021.findings-emnlp.259"},{"key":"6980_CR36","doi-asserted-by":"crossref","unstructured":"Liu Z, Lin Y, Cao Y et\u00a0al (2021) Swin transformer: hierarchical vision transformer using shifted windows. In: IEEE\/CVF international conference on computer vision, pp 10012\u201310022","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"6980_CR37","doi-asserted-by":"crossref","unstructured":"Xie S, Girshick R, Doll\u00e1r P et\u00a0al (2017) Aggregated residual transformations for deep neural networks. In: IEEE conference on computer vision and pattern recognition, pp 1492\u20131500","DOI":"10.1109\/CVPR.2017.634"},{"key":"6980_CR38","doi-asserted-by":"crossref","unstructured":"Pennington J, Socher R, Manning CD (2014) Glove: global vectors for word representation. In: Proceedings of the 2014 conference on empirical methods in natural language processing (EMNLP), pp 1532\u20131543","DOI":"10.3115\/v1\/D14-1162"},{"key":"6980_CR39","doi-asserted-by":"publisher","unstructured":"Devlin J, Chang MW, Lee K et\u00a0al (2019) BERT: pre-training of deep bidirectional transformers for language understanding. In: Burstein J, Doran C, Solorio T (eds) Proceedings of the 2019 conference of the North American chapter of the association for computational linguistics: human language technologies, volume 1 (Long and Short Papers). Association for Computational Linguistics, Minneapolis, Minnesota, pp 4171\u20134186. https:\/\/doi.org\/10.18653\/v1\/N19-1423. https:\/\/aclanthology.org\/N19-1423\/","DOI":"10.18653\/v1\/N19-1423"},{"key":"6980_CR40","unstructured":"Ren S, He K, Girshick R et\u00a0al (2015) Faster r-cnn: towards real-time object detection with region proposal networks. In: Cortes C, Lawrence N, Lee D et\u00a0al (eds) Advances in neural information processing systems, vol\u00a028. Curran Associates, Inc.. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2015\/file\/14bfa6bb14875e45bba028a21ed38046-Paper.pdf"},{"key":"6980_CR41","unstructured":"Pearl J, Mackenzie D (2018) The book of why: the new science of cause and effect. Basic books"},{"key":"6980_CR42","unstructured":"Xu K, Ba J, Kiros R et\u00a0al (2015) Show, attend and tell: neural image caption generation with visual attention. In: Bach F, Blei D (eds) Proceedings of the 32nd international conference on machine learning, proceedings of machine learning research, vol\u00a037. PMLR, Lille, France, pp 2048\u20132057. https:\/\/proceedings.mlr.press\/v37\/xuc15.html"},{"key":"6980_CR43","doi-asserted-by":"crossref","unstructured":"Xu L, Huang H, Liu J (2021) Sutd-trafficqa: a question answering benchmark and an efficient network for video reasoning over traffic events. In: IEEE\/CVF conference on computer vision and pattern recognition, pp 9878\u20139888","DOI":"10.1109\/CVPR46437.2021.00975"},{"key":"6980_CR44","unstructured":"Chen D, Dolan WB (2011) Collecting highly parallel data for paraphrase evaluation. In: Proceedings of the 49th annual meeting of the association for computational linguistics: human language technologies, pp 190\u2013200"},{"key":"6980_CR45","doi-asserted-by":"crossref","unstructured":"Seo A, Kang GC, Park J et\u00a0al (2021) Attend what you need: motion-appearance synergistic networks for video question answering. arXiv:2106.10446","DOI":"10.18653\/v1\/2021.acl-long.481"},{"key":"6980_CR46","doi-asserted-by":"publisher","first-page":"3369","DOI":"10.1109\/TMM.2021.3097171","volume":"24","author":"J Wang","year":"2022","unstructured":"Wang J, Bao BK, Xu C (2022) Dualvgr: a dual-visual graph reasoning unit for video question answering. IEEE Trans Multimedia 24:3369\u20133380. https:\/\/doi.org\/10.1109\/TMM.2021.3097171","journal-title":"IEEE Trans Multimedia"},{"key":"6980_CR47","doi-asserted-by":"crossref","unstructured":"Kim N, Ha SJ, Kang JW (2021) Video question answering using language-guided deep compressed-domain video feature. In: IEEE\/CVF international conference on computer vision, pp 1708\u20131717","DOI":"10.1109\/ICCV48922.2021.00173"},{"issue":"2","key":"6980_CR48","doi-asserted-by":"publisher","first-page":"581","DOI":"10.1007\/s11263-023-01891-x","volume":"132","author":"P Gao","year":"2024","unstructured":"Gao P, Geng S, Zhang R et al (2024) Clip-adapter: better vision-language models with feature adapters. Int J Comput Vision 132(2):581\u2013595","journal-title":"Int J Comput Vision"},{"key":"6980_CR49","doi-asserted-by":"publisher","first-page":"5477","DOI":"10.1109\/TIP.2021.3076556","volume":"30","author":"W Jin","year":"2021","unstructured":"Jin W, Zhao Z, Cao X et al (2021) Adaptive spatio-temporal graph enhanced vision-language representation for video qa. IEEE Trans Image Process 30:5477\u20135489","journal-title":"IEEE Trans Image Process"},{"key":"6980_CR50","doi-asserted-by":"publisher","first-page":"1684","DOI":"10.1109\/TIP.2022.3142526","volume":"31","author":"Y Liu","year":"2022","unstructured":"Liu Y, Zhang X, Huang F et al (2022) Cross-attentional spatio-temporal semantic graph networks for video question answering. IEEE Trans Image Process 31:1684\u20131696","journal-title":"IEEE Trans Image Process"},{"key":"6980_CR51","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S et\u00a0al (2016) Deep residual learning for image recognition. In: IEEE conference on computer vision and pattern recognition, pp 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"6980_CR52","doi-asserted-by":"crossref","unstructured":"Liu Z, Hu H, Lin Y et\u00a0al (2022) Swin transformer v2: scaling up capacity and resolution. In: IEEE\/CVF conference on computer vision and pattern recognition, pp 12009\u201312019","DOI":"10.1109\/CVPR52688.2022.01170"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-025-06980-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10489-025-06980-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-025-06980-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,26]],"date-time":"2025-11-26T08:05:13Z","timestamp":1764144313000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10489-025-06980-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11]]},"references-count":52,"journal-issue":{"issue":"16","published-print":{"date-parts":[[2025,11]]}},"alternative-id":["6980"],"URL":"https:\/\/doi.org\/10.1007\/s10489-025-06980-7","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"value":"0924-669X","type":"print"},{"value":"1573-7497","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,11]]},"assertion":[{"value":"25 March 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 October 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"8 November 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflicts of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflicts of Interest"}},{"value":"All authors have read and agreed to the published version of the manuscript.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Consent for Publication"}}],"article-number":"1087"}}