{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,21]],"date-time":"2026-02-21T18:25:38Z","timestamp":1771698338977,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":79,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,6,7]],"date-time":"2023-06-07T00:00:00Z","timestamp":1686096000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,6,7]]},"DOI":"10.1145\/3587819.3590988","type":"proceedings-article","created":{"date-parts":[[2023,6,8]],"date-time":"2023-06-08T17:24:19Z","timestamp":1686245059000},"page":"289-300","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":11,"title":["Video-based Contrastive Learning on Decision Trees: from Action Recognition to Autism Diagnosis"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7931-500X","authenticated-orcid":false,"given":"Mindi","family":"Ruan","sequence":"first","affiliation":[{"name":"West Virginia University, Morgantown, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2710-2969","authenticated-orcid":false,"given":"Xiangxu","family":"Yu","sequence":"additional","affiliation":[{"name":"Washington University at Saint Louis, St. Louis, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8845-5055","authenticated-orcid":false,"given":"Na","family":"Zhang","sequence":"additional","affiliation":[{"name":"West Virginia University, Morgantown, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1165-5005","authenticated-orcid":false,"given":"Chuanbo","family":"Hu","sequence":"additional","affiliation":[{"name":"West Virginia University, Morgantown, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2562-0225","authenticated-orcid":false,"given":"Shuo","family":"Wang","sequence":"additional","affiliation":[{"name":"Washington University at Saint Louis, St. Louis, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2067-2763","authenticated-orcid":false,"given":"Xin","family":"Li","sequence":"additional","affiliation":[{"name":"West Virginia University, Morgantown, United States of America"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,6,8]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1093\/jamia\/ocy039"},{"key":"e_1_3_2_1_2_1","volume-title":"Youtube-8m: A large-scale video classification benchmark. arXiv preprint arXiv:1609.08675","author":"Abu-El-Haija Sami","year":"2016","unstructured":"Sami Abu-El-Haija, Nisarg Kothari, Joonseok Lee, Paul Natsev, George Toderici, Balakrishnan Varadarajan, and Sudheendra Vijayanarasimhan. 2016. Youtube-8m: A large-scale video classification benchmark. arXiv preprint arXiv:1609.08675 (2016)."},{"key":"e_1_3_2_1_3_1","volume-title":"Human motion analysis: A review. Computer vision and image understanding 73, 3","author":"Aggarwal Jake K","year":"1999","unstructured":"Jake K Aggarwal and Quin Cai. 1999. Human motion analysis: A review. Computer vision and image understanding 73, 3 (1999), 428--440."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-006-0009-9"},{"key":"e_1_3_2_1_5_1","volume-title":"Random forests. Machine learning 45, 1","author":"Breiman Leo","year":"2001","unstructured":"Leo Breiman. 2001. Random forests. Machine learning 45, 1 (2001), 5--32."},{"key":"e_1_3_2_1_6_1","volume-title":"Classification and regression trees","author":"Breiman Leo","unstructured":"Leo Breiman, Jerome H Friedman, Richard A Olshen, and Charles J Stone. 2017. Classification and regression trees. Routledge."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1177\/1362361318766247"},{"key":"e_1_3_2_1_8_1","volume-title":"A short note on the kinetics-700 human action dataset. arXiv preprint arXiv:1907.06987","author":"Carreira Joao","year":"2019","unstructured":"Joao Carreira, Eric Noland, Chloe Hillier, and Andrew Zisserman. 2019. A short note on the kinetics-700 human action dataset. arXiv preprint arXiv:1907.06987 (2019)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"key":"e_1_3_2_1_10_1","volume-title":"Rachel Aiello, Jeffrey Baker, Kimberly Carpenter, Scott Compton, Naomi Davis, Brian Eichner, Steven Espinosa, Jacqueline Flowers, et al.","author":"Chang Zhuoqing","year":"2021","unstructured":"Zhuoqing Chang, J Matias Di Martino, Rachel Aiello, Jeffrey Baker, Kimberly Carpenter, Scott Compton, Naomi Davis, Brian Eichner, Steven Espinosa, Jacqueline Flowers, et al. 2021. Computational methods to measure patterns of gaze in toddlers with autism spectrum disorder. JAMA pediatrics 175, 8 (2021), 827--836."},{"key":"e_1_3_2_1_11_1","volume-title":"Proceedings, Part XXIV 16","author":"Cheng Ke","year":"2020","unstructured":"Ke Cheng, Yifan Zhang, Congqi Cao, Lei Shi, Jian Cheng, and Hanqing Lu. 2020. Decoupling gcn with dropgraph module for skeleton-based action recognition. In Computer Vision-ECCV 2020: 16th European Conference, Glasgow, UK, August 23--28, 2020, Proceedings, Part XXIV 16. Springer, 536--553."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00026"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV51458.2022.00229"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1037\/pri0000121"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.213"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58577-8_14"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01255"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1002\/aur.1263"},{"key":"e_1_3_2_1_19_1","volume-title":"Unsupervised on-line learning of decision trees for hierarchical data analysis. Advances in neural information processing systems 10","author":"Held Marcus","year":"1997","unstructured":"Marcus Held and Joachim Buhmann. 1997. Unsupervised on-line learning of decision trees for hierarchical data analysis. Advances in neural information processing systems 10 (1997)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58555-6_35"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1145\/3459637.3481908"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.buildenv.2019.106424"},{"key":"e_1_3_2_1_23_1","volume-title":"3D convolutional neural networks for human action recognition","author":"Ji Shuiwang","year":"2012","unstructured":"Shuiwang Ji, Wei Xu, Ming Yang, and Kai Yu. 2012. 3D convolutional neural networks for human action recognition. IEEE transactions on pattern analysis and machine intelligence 35, 1 (2012), 221--231."},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.354"},{"key":"e_1_3_2_1_25_1","unstructured":"Will Kay Joao Carreira Karen Simonyan Brian Zhang Chloe Hillier Sudheendra Vijayanarasimhan Fabio Viola Tim Green Trevor Back Paul Natsev et al. 2017. The kinetics human action video dataset. arXiv preprint arXiv:1705.06950 (2017)."},{"key":"e_1_3_2_1_26_1","volume-title":"Semi-supervised classification with graph convolutional networks. arXiv preprint arXiv:1609.02907","author":"Kipf Thomas N","year":"2016","unstructured":"Thomas N Kipf and Max Welling. 2016. Semi-supervised classification with graph convolutional networks. arXiv preprint arXiv:1609.02907 (2016)."},{"key":"e_1_3_2_1_27_1","volume-title":"Human action recognition and prediction: A survey. arXiv preprint arXiv:1806.11230","author":"Kong Yu","year":"2018","unstructured":"Yu Kong and Yun Fu. 2018. Human action recognition and prediction: A survey. arXiv preprint arXiv:1806.11230 (2018)."},{"key":"e_1_3_2_1_28_1","volume-title":"Proceedings of the International Conference on Computer Vision (ICCV).","author":"Kuehne H.","unstructured":"H. Kuehne, H. Jhuang, E. Garrote, T. Poggio, and T. Serre. 2011. HMDB: a large video database for human motion recognition. In Proceedings of the International Conference on Computer Vision (ICCV)."},{"key":"e_1_3_2_1_29_1","volume-title":"Lada Adamic, Sinan Aral, Albert Laszlo Barabasi, Devon Brewer, Nicholas Christakis, Noshir Contractor, James Fowler, Myron Gutmann, et al.","author":"Lazer David","year":"2009","unstructured":"David Lazer, Alex Sandy Pentland, Lada Adamic, Sinan Aral, Albert Laszlo Barabasi, Devon Brewer, Nicholas Christakis, Noshir Contractor, James Fowler, Myron Gutmann, et al. 2009. Life in the network: the coming age of computational social science. Science (New York, NY) 323, 5915 (2009), 721."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00371"},{"key":"e_1_3_2_1_31_1","volume-title":"Trear: Transformer-based rgb-d egocentric action recognition","author":"Li Xiangyu","year":"2021","unstructured":"Xiangyu Li, Yonghong Hou, Pichao Wang, Zhimin Gao, Mingliang Xu, and Wanqing Li. 2021. Trear: Transformer-based rgb-d egocentric action recognition. IEEE Transactions on Cognitive and Developmental Systems (2021)."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58539-6_17"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3413548"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/FG.2017.134"},{"key":"e_1_3_2_1_35_1","volume-title":"Ntu rgb+ d 120: A large-scale benchmark for 3d human activity understanding","author":"Liu Jun","year":"2019","unstructured":"Jun Liu, Amir Shahroudy, Mauricio Perez, Gang Wang, Ling-Yu Duan, and Alex C Kot. 2019. Ntu rgb+ d 120: A large-scale benchmark for 3d human activity understanding. IEEE transactions on pattern analysis and machine intelligence 42, 10 (2019), 2684--2701."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACII.2017.8273597"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-60639-8_40"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1023\/A:1005592401947"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01313"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/34.868684"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jaac.2011.03.012"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1023\/A:1016374617369"},{"key":"e_1_3_2_1_43_1","volume-title":"Interaction relational network for mutual action recognition","author":"Perez Mauricio","year":"2021","unstructured":"Mauricio Perez, Jun Liu, and Alex C Kot. 2021. Interaction relational network for mutual action recognition. IEEE Transactions on Multimedia (2021)."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-68796-0_50"},{"key":"e_1_3_2_1_45_1","volume-title":"A survey on vision-based human action recognition. Image and vision computing 28, 6","author":"Poppe Ronald","year":"2010","unstructured":"Ronald Poppe. 2010. A survey on vision-based human action recognition. Image and vision computing 28, 6 (2010), 976--990."},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2013.103"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/MPRV.2014.23"},{"key":"e_1_3_2_1_48_1","volume-title":"International Workshop on Machine Learning for Multimodal Interaction. Springer, 76--86","author":"Rienks Rutger","year":"2005","unstructured":"Rutger Rienks and Dirk Heylen. 2005. Dominance detection in meetings using easily obtainable features. In International Workshop on Machine Learning for Multimodal Interaction. Springer, 76--86."},{"key":"e_1_3_2_1_49_1","volume-title":"Incipient status in small groups. Social forces 58, 1","author":"Rosa Eugene","year":"1979","unstructured":"Eugene Rosa and Allan Mazur. 1979. Incipient status in small groups. Social forces 58, 1 (1979), 18--37."},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1002\/aur.2376"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2009.5459361"},{"key":"e_1_3_2_1_52_1","volume-title":"A simple neural network module for relational reasoning. Advances in neural information processing systems 30","author":"Santoro Adam","year":"2017","unstructured":"Adam Santoro, David Raposo, David G Barrett, Mateusz Malinowski, Razvan Pascanu, Peter Battaglia, and Timothy Lillicrap. 2017. A simple neural network module for relational reasoning. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01107"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.115"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00810"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01230"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2019.2942030"},{"key":"e_1_3_2_1_58_1","volume-title":"Two-stream convolutional networks for action recognition in videos. arXiv preprint arXiv:1406.2199","author":"Simonyan Karen","year":"2014","unstructured":"Karen Simonyan and Andrew Zisserman. 2014. Two-stream convolutional networks for action recognition in videos. arXiv preprint arXiv:1406.2199 (2014)."},{"key":"e_1_3_2_1_59_1","volume-title":"Imitation and action in autism: a critical review. Psychological bulletin 116, 2","author":"Smith Isabel M","year":"1994","unstructured":"Isabel M Smith and Susan E Bryson. 1994. Imitation and action in autism: a critical review. Psychological bulletin 116, 2 (1994), 259."},{"key":"e_1_3_2_1_60_1","volume-title":"Amir Roshan Zamir, and Mubarak Shah","author":"Soomro Khurram","year":"2012","unstructured":"Khurram Soomro, Amir Roshan Zamir, and Mubarak Shah. 2012. UCF101: A dataset of 101 human actions classes from videos in the wild. arXiv preprint arXiv:1212.0402 (2012)."},{"key":"e_1_3_2_1_61_1","volume-title":"Peter Washington, Haik Kalantarian, and Dennis Paul Wall.","author":"Tariq Qandeel","year":"2018","unstructured":"Qandeel Tariq, Jena Daniels, Jessey Nicole Schwartz, Peter Washington, Haik Kalantarian, and Dennis Paul Wall. 2018. Mobile detection of autism through machine learning on home video: A development and prospective validation study. PLoS medicine 15, 11 (2018), e1002705."},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.603"},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10803-006-0137-7"},{"key":"e_1_3_2_1_64_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01940"},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2013.441"},{"key":"e_1_3_2_1_66_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00417"},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00840"},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"publisher","DOI":"10.1016\/S0149-7634(01)00014-8"},{"key":"e_1_3_2_1_69_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01020"},{"key":"e_1_3_2_1_70_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00325"},{"key":"e_1_3_2_1_71_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.12328"},{"key":"e_1_3_2_1_72_1","doi-asserted-by":"crossref","unstructured":"Chao-Lung Yang Aji Setyoko Hendrik Tampubolon and Kai-Lung Hua. 2020. Pairwise Adjacency Matrix on Spatial Temporal Graph Convolution Network for Skeleton-Based Two-Person Interaction Recognition. In 2020 IEEE International Conference on Image Processing (ICIP). 2166--2170.","DOI":"10.1109\/ICIP40778.2020.9190680"},{"key":"e_1_3_2_1_73_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01367"},{"key":"e_1_3_2_1_74_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2010.5540235"},{"key":"e_1_3_2_1_75_1","first-page":"609","article-title":"Obtaining calibrated probability estimates from decision trees and naive bayesian classifiers","volume":"1","author":"Zadrozny Bianca","year":"2001","unstructured":"Bianca Zadrozny and Charles Elkan. 2001. Obtaining calibrated probability estimates from decision trees and naive bayesian classifiers. In ICML, Vol. 1. 609--616.","journal-title":"ICML"},{"key":"e_1_3_2_1_76_1","volume-title":"Discriminative few shot learning of facial dynamics in interview videos for autism trait classification","author":"Zhang Na","year":"2022","unstructured":"Na Zhang, Mindi Ruan, Shuo Wang, Lynn Paul, and Xin Li. 2022. Discriminative few shot learning of facial dynamics in interview videos for autism trait classification. IEEE Transactions on Affective Computing (2022)."},{"key":"e_1_3_2_1_77_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475473"},{"key":"e_1_3_2_1_78_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00432"},{"key":"e_1_3_2_1_79_1","volume-title":"A comprehensive study of deep video action recognition. arXiv preprint arXiv:2012.06567","author":"Zhu Yi","year":"2020","unstructured":"Yi Zhu, Xinyu Li, Chunhui Liu, Mohammadreza Zolfaghari, Yuanjun Xiong, Chongruo Wu, Zhi Zhang, Joseph Tighe, R Manmatha, and Mu Li. 2020. A comprehensive study of deep video action recognition. arXiv preprint arXiv:2012.06567 (2020)."}],"event":{"name":"MMSys '23: 14th Conference on ACM Multimedia Systems","location":"Vancouver BC Canada","acronym":"MMSys '23","sponsor":["SIGMM ACM Special Interest Group on Multimedia","SIGCOMM ACM Special Interest Group on Data Communication","SIGMOBILE ACM Special Interest Group on Mobility of Systems, Users, Data and Computing"]},"container-title":["Proceedings of the 14th ACM Multimedia Systems Conference"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3587819.3590988","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3587819.3590988","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T18:08:01Z","timestamp":1750183681000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3587819.3590988"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,6,7]]},"references-count":79,"alternative-id":["10.1145\/3587819.3590988","10.1145\/3587819"],"URL":"https:\/\/doi.org\/10.1145\/3587819.3590988","relation":{},"subject":[],"published":{"date-parts":[[2023,6,7]]},"assertion":[{"value":"2023-06-08","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}