{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,2]],"date-time":"2026-06-02T09:29:07Z","timestamp":1780392547123,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":53,"publisher":"ACM","license":[{"start":{"date-parts":[[2019,10,15]],"date-time":"2019-10-15T00:00:00Z","timestamp":1571097600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"NSF of China","award":["61571362"],"award-info":[{"award-number":["61571362"]}]},{"name":"Agency for Science, Technology and Research (A*STAR)","award":["A18A2b0046"],"award-info":[{"award-number":["A18A2b0046"]}]},{"name":"Fundamental Research Funds for the Central Universities","award":["3102019ZY1004"],"award-info":[{"award-number":["3102019ZY1004"]}]},{"name":"Natural Science Basic Research Plan in Shaanxi Province of China","award":["2018JM6015"],"award-info":[{"award-number":["2018JM6015"]}]},{"name":"National Research Foundation Singapore","award":["Strategic Capability Research Centres Funding Initiative"],"award-info":[{"award-number":["Strategic Capability Research Centres Funding Initiative"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2019,10,15]]},"DOI":"10.1145\/3343031.3351040","type":"proceedings-article","created":{"date-parts":[[2019,10,21]],"date-time":"2019-10-21T16:32:26Z","timestamp":1571675546000},"page":"521-529","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":47,"title":["Explainable Video Action Reasoning via Prior Knowledge and State Transitions"],"prefix":"10.1145","author":[{"given":"Tao","family":"Zhuo","sequence":"first","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhiyong","family":"Cheng","sequence":"additional","affiliation":[{"name":"Qilu University of Technology (Shandong Academy of Sciences), Jinan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Peng","family":"Zhang","sequence":"additional","affiliation":[{"name":"Northwestern Polytechnical University, Xi'an, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yongkang","family":"Wong","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mohan","family":"Kankanhalli","sequence":"additional","affiliation":[{"name":"National University of Singapore, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2019,10,15]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Joint discovery of object states and manipulating actions","author":"Alayrac Jean-Baptiste"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"crossref","volume-title":"Probabilistic event logic for interval-based event recognition","author":"Brendel William","DOI":"10.1109\/CVPR.2011.5995491"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"crossref","volume-title":"Quo vadis, action recognition? a new model and the kinetics dataset","author":"Carreira Joao","DOI":"10.1109\/CVPR.2017.502"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"crossref","volume-title":"Detecting Visual Relationships With Deep Relational Networks","author":"Dai Bo","DOI":"10.1109\/CVPR.2017.352"},{"key":"e_1_3_2_1_5_1","volume-title":"Probabilistic Inductive Logic Programming","author":"Raedt Luc De"},{"key":"e_1_3_2_1_6_1","volume-title":"Victor Escorcia and Juan Carlos Niebles","author":"Fabian Caba Heilbron Bernard Ghanem","year":"2015"},{"key":"e_1_3_2_1_7_1","volume-title":"Rehg","author":"Fathi Alireza","year":"2013"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"crossref","volume-title":"Spatiotemporal multiplier networks for video action recognition","author":"Feichtenhofer Christoph","DOI":"10.1109\/CVPR.2017.787"},{"key":"e_1_3_2_1_9_1","volume-title":"What Have We Learned From Deep Representations for Action Recognition?","author":"Feichtenhofer Christoph"},{"key":"e_1_3_2_1_10_1","article-title":"Learning Perceptual Causality from Video","volume":"7","author":"Fire Amy","year":"2015","journal-title":"Transactions on Intelligent Systems and Technology"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"crossref","volume-title":"Supervised sequence labelling with recurrent neural networks","author":"Graves Alex","DOI":"10.1007\/978-3-642-24797-2"},{"key":"e_1_3_2_1_12_1","volume-title":"Neural graph matching networks for fewshot 3d action recognition","author":"Guo Michelle"},{"key":"e_1_3_2_1_13_1","volume-title":"Deep residual learning for image recognition","author":"He Kaiming"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"crossref","unstructured":"Joris Ijsselmuiden and Rainer Stiefelhagen. 2010. Towards high-level human activity recognition through computer vision and temporal logic. In AAAI .  Joris Ijsselmuiden and Rainer Stiefelhagen. 2010. Towards high-level human activity recognition through computer vision and temporal logic. In AAAI .","DOI":"10.1007\/978-3-642-16111-7_49"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1006\/cviu.2000.0896"},{"key":"e_1_3_2_1_16_1","volume-title":"Structural-RNN: Deep learning on spatio-temporal graphs","author":"Jain Ashesh"},{"key":"e_1_3_2_1_17_1","volume-title":"Kingma and Jimmy Ba","author":"Diederik","year":"2014"},{"key":"e_1_3_2_1_18_1","unstructured":"Thomas N Kipf and Max Welling. 2017. Semi-supervised classification with graph convolutional networks. In ICLR .  Thomas N Kipf and Max Welling. 2017. Semi-supervised classification with graph convolutional networks. In ICLR ."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"crossref","unstructured":"Alexander Klaser Marcin Marszalek and Cordelia Schmid. 2008. A Spatio-Temporal Descriptor Based on 3D-Gradients. In BMVC . https:\/\/hal.inria.fr\/inria-00514853  Alexander Klaser Marcin Marszalek and Cordelia Schmid. 2008. A Spatio-Temporal Descriptor Based on 3D-Gradients. In BMVC . https:\/\/hal.inria.fr\/inria-00514853","DOI":"10.5244\/C.22.99"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1177\/0278364913478446"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"crossref","volume-title":"Referring Relationships","author":"Krishna Ranjay","DOI":"10.1109\/CVPR.2018.00718"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-016-0981-7"},{"key":"e_1_3_2_1_23_1","volume-title":"Ensemble deep learning for skeleton-based action recognition using temporal sliding LSTM networks","author":"Lee Inwoong"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2016.2537337"},{"key":"e_1_3_2_1_25_1","volume-title":"Jointly Recognizing Object Fluents and Tasks in Egocentric Videos","author":"Liu Yang"},{"key":"e_1_3_2_1_26_1","volume-title":"Visual relationship detection with language priors","author":"Lu Cewu"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"crossref","volume-title":"Discriminative correlation filter with channel and spatial reliability","author":"Lukezic Alan","DOI":"10.1109\/CVPR.2017.515"},{"key":"e_1_3_2_1_28_1","volume-title":"Multi-agent event recognition in structured scenarios","author":"Morariu Vlad I"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1093\/logcom\/4.5.467"},{"key":"e_1_3_2_1_30_1","volume-title":"Artificial Intelligence: foundations of computational agents","author":"Poole David L"},{"key":"e_1_3_2_1_31_1","volume-title":"Learning human-object interactions by graph parsing neural networks","author":"Qi Siyuan"},{"key":"e_1_3_2_1_32_1","volume-title":"Markov logic networks. Machine learning","author":"Richardson Matthew","year":"2006"},{"key":"e_1_3_2_1_33_1","volume-title":"Artificial Intelligence: A Modern Approach","author":"Russell Stuart","year":"2009","edition":"3"},{"key":"e_1_3_2_1_34_1","volume-title":"Learning realistic human actions from movies","author":"Schmid Cordelia"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"crossref","unstructured":"J. Shotton A. Fitzgibbon M. Cook T. Sharp M. Finocchio R. Moore A. Kipman and A. Blake. 2011. Real-time Human Pose Recognition in Parts from Single Depth Images. In CVPR. IEEE.  J. Shotton A. Fitzgibbon M. Cook T. Sharp M. Finocchio R. Moore A. Kipman and A. Blake. 2011. Real-time Human Pose Recognition in Parts from Single Depth Images. In CVPR. IEEE.","DOI":"10.1109\/CVPR.2011.5995316"},{"key":"e_1_3_2_1_36_1","unstructured":"Karen Simonyan and Andrew Zisserman. 2014a. Two-Stream Convolutional Networks for Action Recognition in Videos. In NeurIPS. 568--576.  Karen Simonyan and Andrew Zisserman. 2014a. Two-Stream Convolutional Networks for Action Recognition in Videos. In NeurIPS. 568--576."},{"key":"e_1_3_2_1_37_1","volume-title":"Very Deep Convolutional Networks for Large-Scale Image Recognition. CoRR","author":"Simonyan Karen","year":"2014"},{"key":"e_1_3_2_1_38_1","volume-title":"Amir Roshan Zamir, and Mubarak Shah","author":"Soomro Khurram","year":"2012"},{"key":"e_1_3_2_1_39_1","unstructured":"John F Sowa. 2014. Principles of semantic networks: Explorations in the representation of knowledge .Morgan Kaufmann.  John F Sowa. 2014. Principles of semantic networks: Explorations in the representation of knowledge .Morgan Kaufmann."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"crossref","unstructured":"Christian Szegedy Sergey Ioffe Vincent Vanhoucke and Alexander A Alemi. 2017. Inception-v4 inception-resnet and the impact of residual connections on learning. In AAAI .  Christian Szegedy Sergey Ioffe Vincent Vanhoucke and Alexander A Alemi. 2017. Inception-v4 inception-resnet and the impact of residual connections on learning. In AAAI .","DOI":"10.1609\/aaai.v31i1.11231"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"crossref","volume-title":"A closer look at spatiotemporal convolutions for action recognition","author":"Tran Du","DOI":"10.1109\/CVPR.2018.00675"},{"key":"e_1_3_2_1_42_1","volume-title":"Event modeling and recognition using markov logic networks","author":"Tran Son D"},{"key":"e_1_3_2_1_43_1","doi-asserted-by":"crossref","volume-title":"Action Recognition with Improved Trajectories","author":"Wang Heng","DOI":"10.1109\/ICCV.2013.441"},{"key":"e_1_3_2_1_44_1","volume-title":"Appearance-and-Relation Networks for Video Classification","author":"Wang Limin"},{"key":"e_1_3_2_1_45_1","volume-title":"Temporal segment networks for action recognition in videos","author":"Wang Limin","year":"2018"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2012.2185041"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2012.2187181"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2017.2716819"},{"key":"e_1_3_2_1_49_1","volume-title":"Actions Transformations","author":"Wang Xiaolong"},{"key":"e_1_3_2_1_50_1","volume-title":"Videos as space-time region graphs","author":"Wang Xiaolong"},{"key":"e_1_3_2_1_51_1","volume-title":"Scene Graph Generation by Iterative Message Passing","author":"Xu Danfei"},{"key":"e_1_3_2_1_52_1","volume-title":"Dual-stream recurrent neural network for video captioning","author":"Xu Ning","year":"2018"},{"key":"e_1_3_2_1_53_1","volume-title":"Visual semantic planning using deep successor representations","author":"Zhu Yuke"}],"event":{"name":"MM '19: The 27th ACM International Conference on Multimedia","location":"Nice France","acronym":"MM '19","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 27th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3343031.3351040","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3343031.3351040","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T23:13:11Z","timestamp":1750201991000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3343031.3351040"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,10,15]]},"references-count":53,"alternative-id":["10.1145\/3343031.3351040","10.1145\/3343031"],"URL":"https:\/\/doi.org\/10.1145\/3343031.3351040","relation":{},"subject":[],"published":{"date-parts":[[2019,10,15]]},"assertion":[{"value":"2019-10-15","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}