{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T19:34:34Z","timestamp":1783712074462,"version":"3.55.0"},"reference-count":67,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2021ZD0111901"],"award-info":[{"award-number":["2021ZD0111901"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U21B2025"],"award-info":[{"award-number":["U21B2025"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"Natural Science Foundation of China","doi-asserted-by":"publisher","award":["U19B2036"],"award-info":[{"award-number":["U19B2036"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. on Image Process."],"published-print":{"date-parts":[[2024]]},"DOI":"10.1109\/tip.2024.3358726","type":"journal-article","created":{"date-parts":[[2024,1,31]],"date-time":"2024-01-31T18:41:34Z","timestamp":1706726494000},"page":"1109-1121","source":"Crossref","is-referenced-by-count":11,"title":["Event Graph Guided Compositional Spatial\u2013Temporal Reasoning for Video Question Answering"],"prefix":"10.1109","volume":"33","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-5917-5400","authenticated-orcid":false,"given":"Ziyi","family":"Bai","sequence":"first","affiliation":[{"name":"Key Laboratory of Intelligent Information Processing, Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1830-2595","authenticated-orcid":false,"given":"Ruiping","family":"Wang","sequence":"additional","affiliation":[{"name":"Key Laboratory of Intelligent Information Processing, Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8494-3492","authenticated-orcid":false,"given":"Difei","family":"Gao","sequence":"additional","affiliation":[{"name":"Key Laboratory of Intelligent Information Processing, Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3024-4404","authenticated-orcid":false,"given":"Xilin","family":"Chen","sequence":"additional","affiliation":[{"name":"Key Laboratory of Intelligent Information Processing, Institute of Computing Technology, Chinese Academy of Sciences, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1145\/1866029.1866080"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.279"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.670"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00522"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00686"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33018658"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00210"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01527"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2021.3097171"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01113"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.330"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01246-5_41"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/WACV45572.2020.9093404"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01025"},{"key":"ref15","article-title":"Activity graph transformer for temporal action localization","author":"Nawhal","year":"2021","journal-title":"arXiv:2101.08540"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1016\/j.cub.2004.09.041"},{"key":"ref17","first-page":"1","article-title":"STAR: A benchmark for situated reasoning in real-world videos","volume-title":"Proc. 35th Conf. Neural Inf. Process. Syst. Track Datasets Benchmarks (Round 2)","author":"Wu"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00688"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00999"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2020.2963950"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2022.3205212"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00947"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01643"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1145\/3343031.3351065"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6737"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00172"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6767"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2021.3051756"},{"key":"ref29","article-title":"Hierarchical object-oriented spatio-temporal reasoning for video question answering","author":"Hoang Dang","year":"2021","journal-title":"arXiv:2106.13432"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i3.20184"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i1.19922"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2020.3032542"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/WACV48630.2021.00112"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00304"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00756"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00877"},{"key":"ref37","first-page":"23634","article-title":"MERLOT: Multimodal neural script knowledge models","volume-title":"Proc. 35th Conf. Neural Inf. Process. Syst.","author":"Zellers"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00725"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00171"},{"key":"ref40","article-title":"Flamingo: A visual language model for few-shot learning","author":"Alayrac","year":"2022","journal-title":"arXiv:2204.14198"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.acl-long.29"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01819"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01429"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02211"},{"key":"ref45","article-title":"Benchmarking graph neural networks","author":"Dwivedi","year":"2020","journal-title":"arXiv:2003.00982"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.640"},{"key":"ref47","first-page":"21171","article-title":"Hierarchical graph transformer with adaptive node sampling","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Zhang"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00377"},{"key":"ref49","first-page":"5998","article-title":"Attention is all you need","volume-title":"Proc. Adv. Neural Inform. Process. Syst. (NIPS)","author":"Vaswani"},{"key":"ref50","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2018","journal-title":"arXiv:1810.04805"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01606"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46448-0_31"},{"key":"ref53","volume-title":"A Scene Graph Generation Codebase in Pytorch","author":"Tang","year":"2020"},{"key":"ref54","first-page":"91","article-title":"Faster R-CNN: Towards real-time object detection with region proposal networks","volume-title":"Proc. Int. Conf. Adv. Neural Inf. Process. Syst.","volume":"28","author":"Ren"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.106"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.634"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-016-0981-7"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00399"},{"key":"ref61","article-title":"Decoupled weight decay regularization","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Loshchilov"},{"key":"ref62","first-page":"8026","article-title":"PyTorch: An imperative style, high-performance deep learning library","volume-title":"Proc. 33rd Conf. Neural Inf. Process. Syst. (NeurIPS)","volume":"32","author":"Paszke"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19812-0_22"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01225-0_1"},{"key":"ref65","article-title":"MIST: Multi-modal iterative spatial\u2013temporal transformer for long-form video question answering","author":"Gao","year":"2022","journal-title":"arXiv:2212.09522"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01589"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00293"}],"container-title":["IEEE Transactions on Image Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/83\/10346232\/10418133.pdf?arnumber=10418133","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,2,6]],"date-time":"2024-02-06T22:18:16Z","timestamp":1707257896000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10418133\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"references-count":67,"URL":"https:\/\/doi.org\/10.1109\/tip.2024.3358726","relation":{},"ISSN":["1057-7149","1941-0042"],"issn-type":[{"value":"1057-7149","type":"print"},{"value":"1941-0042","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024]]}}}