{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,5]],"date-time":"2026-06-05T04:33:30Z","timestamp":1780634010895,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":51,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,26]],"date-time":"2023-10-26T00:00:00Z","timestamp":1698278400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Young Elite Scientist Sponsorship Program of Beijing Association for Science and Technology","award":["BYESS2021178"],"award-info":[{"award-number":["BYESS2021178"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62171325"],"award-info":[{"award-number":["62171325"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Hubei Key R&D Project","award":["2022BAA033"],"award-info":[{"award-number":["2022BAA033"]}]},{"name":"CAAI-Huawei MindSpore Open Fund"},{"name":"National Key R&D Project","award":["2021YFC3320301"],"award-info":[{"award-number":["2021YFC3320301"]}]},{"name":"Young Elite Scientist Sponsorship Program of China Association for Science and Technology","award":["YESS20200140"],"award-info":[{"award-number":["YESS20200140"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,26]]},"DOI":"10.1145\/3581783.3611892","type":"proceedings-article","created":{"date-parts":[[2023,10,27]],"date-time":"2023-10-27T07:27:12Z","timestamp":1698391632000},"page":"6623-6633","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":8,"title":["Uncovering the Unseen: Discover Hidden Intentions by Micro-Behavior Graph Reasoning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-4620-4378","authenticated-orcid":false,"given":"Zhuo","family":"Zhou","sequence":"first","affiliation":[{"name":"School of Computer Science, Wuhan University, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4417-6628","authenticated-orcid":false,"given":"Wenxuan","family":"Liu","sequence":"additional","affiliation":[{"name":"School of Computer Science, Wuhan University, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9482-0111","authenticated-orcid":false,"given":"Danni","family":"Xu","sequence":"additional","affiliation":[{"name":"School of Computing, National University of Singapore, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3846-9157","authenticated-orcid":false,"given":"Zheng","family":"Wang","sequence":"additional","affiliation":[{"name":"National Engineering Research Center for Multimedia Software, School of Computer Science, Wuhan University, Wuhan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3508-756X","authenticated-orcid":false,"given":"Jian","family":"Zhao","sequence":"additional","affiliation":[{"name":"Intelligent Game and Decision Laboratory, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2023,10,27]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10639-015-9388-2"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"crossref","unstructured":"Sagie Benaim Ariel Ephrat Oran Lang Inbar Mosseri William T. Freeman Michael Rubinstein Michal Irani and Tali Dekel. 2020. SpeedNet: Learning the Speediness in Videos. In CVPR. 9919--9928.","DOI":"10.1109\/CVPR42600.2020.00994"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"crossref","unstructured":"Jo\u00e3o Carreira and Andrew Zisserman. 2017. Quo Vadis Action Recognition? A New Model and the Kinetics Dataset. In CVPR. 4724--4733.","DOI":"10.1109\/CVPR.2017.502"},{"key":"e_1_3_2_2_4_1","volume-title":"Stochastic Backpropagation: A Memory Efficient Strategy for Training Video Models. In CVPR. 8291--8300.","author":"Cheng Feng","year":"2022","unstructured":"Feng Cheng, Mingze Xu, Yuanjun Xiong, Hao Chen, Xinyu Li, Wei Li, and Wei Xia. 2022. Stochastic Backpropagation: A Memory Efficient Strategy for Training Video Models. In CVPR. 8291--8300."},{"key":"e_1_3_2_2_5_1","volume-title":"Rehg","author":"Chong Eunji","year":"2020","unstructured":"Eunji Chong, Yongxin Wang, Nataniel Ruiz, and James M. Rehg. 2020. Detecting Attended Visual Targets in Video. In CVPR. 5395--5405."},{"key":"e_1_3_2_2_6_1","first-page":"1","article-title":"A novel interaction for competence assessment using micro-behaviors: Extending CACHET to graphs and charts","volume":"438","author":"Colarusso Fiorenzo","year":"2023","unstructured":"Fiorenzo Colarusso, Peter C.-H. Cheng, Grecia Garcia Garcia, Aaron Stockdill, Daniel Raggi, and Mateja Jamnik. 2023. A novel interaction for competence assessment using micro-behaviors: Extending CACHET to graphs and charts. In CHI. 438:1--438:14.","journal-title":"CHI."},{"key":"e_1_3_2_2_7_1","volume-title":"B\u00fclthoff","author":"de la Rosa Stephan","year":"2016","unstructured":"Stephan de la Rosa, Ylva Ferstl, and Heinrich H. B\u00fclthoff. 2016. Visual adaptation dominates bimodal visual-motor action adaptation. Scientific Reports, Vol. 6 (2016)."},{"key":"e_1_3_2_2_8_1","volume-title":"Marigold","author":"Dom\u00ednguez-Zamora F. Javier","year":"2018","unstructured":"F. Javier Dom\u00ednguez-Zamora, Shaila M. Gunn, and Daniel S. Marigold. 2018. Adaptive Gaze Strategies to Reduce Environmental Uncertainty During a Sequential Visuomotor Behaviour. Scientific Reports, Vol. 8 (2018)."},{"key":"e_1_3_2_2_9_1","volume-title":"Flight or Flourish","author":"Buisson-Narsai I. Du","unstructured":"I. Du Buisson-Narsai. 2020. Fight, Flight or Flourish. KR Publishing. https:\/\/books.google.com.sg\/books?id=fc5HEAAAQBAJ"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"crossref","unstructured":"Zhijie Fang Weiqun Wang Shixin Ren Jiaxing Wang Weiguo Shi Xu Liang Chen-Chen Fan and Zeng-Guang Hou. 2020. Learning Regional Attention Convolutional Neural Network for Motion Intention Recognition Based on EEG Data. In IJCAI. 1570--1576.","DOI":"10.24963\/ijcai.2020\/218"},{"key":"e_1_3_2_2_11_1","volume-title":"Ali Diba, Mehdi Noroozi, Ehsan Adeli, Luc Van Gool, and J\u00fcrgen Gall.","author":"Fayyaz Mohsen","year":"2021","unstructured":"Mohsen Fayyaz, Emad Bahrami Rad, Ali Diba, Mehdi Noroozi, Ehsan Adeli, Luc Van Gool, and J\u00fcrgen Gall. 2021. 3D CNNs With Adaptive Temporal Feature Resolutions. In CVPR. 4731--4740."},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"crossref","unstructured":"Christoph Feichtenhofer. 2020. X3D: Expanding Architectures for Efficient Video Recognition. In CVPR. 200--210.","DOI":"10.1109\/CVPR42600.2020.00028"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"crossref","unstructured":"Christoph Feichtenhofer Haoqi Fan Jitendra Malik and Kaiming He. 2019. SlowFast Networks for Video Recognition. In ICCV. 6201--6210.","DOI":"10.1109\/ICCV.2019.00630"},{"key":"e_1_3_2_2_14_1","volume-title":"Scientific Reports","volume":"10","author":"Gigliotti Maria Francesca","year":"2020","unstructured":"Maria Francesca Gigliotti, Adriana Sampaio, Angela Bartolo, and Yann Coello. 2020. The combined effects of motor and social goals on the kinematics of object-directed motor action. Scientific Reports, Vol. 10 (2020)."},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1038\/s42003-022-04324-6"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"crossref","unstructured":"Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Deep Residual Learning for Image Recognition. In CVPR. 770--778.","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"crossref","unstructured":"Yunhao Li Wei Shen Zhongpai Gao Yucheng Zhu Guangtao Zhai and Guodong Guo. 2021. Looking here or there? Gaze Following in 360-Degree Images. In ICCV. 3722--3731.","DOI":"10.1109\/ICCV48922.2021.00372"},{"key":"e_1_3_2_2_18_1","volume-title":"TSM: Temporal Shift Module for Efficient Video Understanding. In ICCV. 7082--7092.","author":"Lin Ji","year":"2019","unstructured":"Ji Lin, Chuang Gan, and Song Han. 2019. TSM: Temporal Shift Module for Efficient Video Understanding. In ICCV. 7082--7092."},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/LRA.2020.2976305"},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2022.3229646"},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2023.3273459"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"crossref","unstructured":"Wenjing Meng Deqing Yang and Yanghua Xiao. 2020. Incorporating user micro-behaviors and item knowledge into multi-task learning for session-based recommendation. In SIGIR. 1091--1100.","DOI":"10.1145\/3397271.3401098"},{"key":"e_1_3_2_2_23_1","first-page":"1712","article-title":"Micro interactions and Multi dimensional Graphical User Interfaces in the Design of Wrist Worn Wearables","volume":"59","author":"Motti Vivian Genaro","year":"2015","unstructured":"Vivian Genaro Motti and Kelly E. Caine. 2015. Micro interactions and Multi dimensional Graphical User Interfaces in the Design of Wrist Worn Wearables. HFESAM, Vol. 59 (2015), 1712--1716.","journal-title":"HFESAM"},{"key":"e_1_3_2_2_24_1","volume-title":"Scientific Reports","volume":"9","author":"Nogueira-Campos Anaelli Aparecida","year":"2019","unstructured":"Anaelli Aparecida Nogueira-Campos, Pauline M. Hilt, Luciano Fadiga, Christian Veronesi, Alessandro D'Ausilio, and Thierry Pozzo. 2019. Anticipatory postural adjustments during joint action coordination. Scientific Reports, Vol. 9 (2019)."},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1080\/10447318.2021.1921364"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1080\/10447318.2021.1921364"},{"key":"e_1_3_2_2_27_1","volume-title":"Coupling perception to action through incidental sensory consequences of motor behaviour. Nature Reviews Psychology","author":"Rolfs Martin","year":"2022","unstructured":"Martin Rolfs and Richard Schweitzer. 2022. Coupling perception to action through incidental sensory consequences of motor behaviour. Nature Reviews Psychology (2022)."},{"key":"e_1_3_2_2_28_1","first-page":"45","article-title":"Micro-affirmations and micro-inequities","volume":"1","author":"Rowe Mary","year":"2008","unstructured":"Mary Rowe. 2008. Micro-affirmations and micro-inequities. Journal of the International Ombudsman Association, Vol. 1, 1 (2008), 45--48.","journal-title":"Journal of the International Ombudsman Association"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1038\/s41467-022-31716-3"},{"key":"e_1_3_2_2_30_1","volume-title":"Ullman","author":"Shu Tianmin","year":"2021","unstructured":"Tianmin Shu, Abhishek Bhandwaldar, Chuang Gan, Kevin A. Smith, Shari Liu, Dan Gutfreund, Elizabeth S. Spelke, Joshua B. Tenenbaum, and Tomer D. Ullman. 2021. AGENT: A Benchmark for Core Psychological Reasoning. In ICML. 9614--9625."},{"key":"e_1_3_2_2_31_1","unstructured":"Karen Simonyan and Andrew Zisserman. 2014. Two-Stream Convolutional Networks for Action Recognition in Videos. In NeurIPS. 568--576."},{"key":"e_1_3_2_2_32_1","volume-title":"Ullman","author":"Smith Kevin","year":"2019","unstructured":"Kevin Smith, Lingjie Mei, Shunyu Yao, Jiajun Wu, Elizabeth S. Spelke, Josh Tenenbaum, and Tomer D. Ullman. 2019. Modeling Expectation Violation in Intuitive Physics with Coarse Probabilistic Object Representations. In NeurIPS. 8983--8993."},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1038\/nn.2112"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"crossref","unstructured":"Waqas Sultani Chen Chen and Mubarak Shah. 2018. Real-World Anomaly Detection in Surveillance Videos. In CVPR. 6479--6488.","DOI":"10.1109\/CVPR.2018.00678"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"crossref","unstructured":"Du Tran Lubomir D. Bourdev Rob Fergus Lorenzo Torresani and Manohar Paluri. 2015. Learning Spatiotemporal Features with 3D Convolutional Networks. In ICCV. 4489--4497.","DOI":"10.1109\/ICCV.2015.510"},{"key":"e_1_3_2_2_36_1","volume-title":"Luca Cagliero, and Paolo Garza.","author":"Vaiani Lorenzo","year":"2022","unstructured":"Lorenzo Vaiani, Moreno La Quatra, Luca Cagliero, and Paolo Garza. 2022. ViPER: Video-based Perceiver for Emotion Recognition. In ACM MM. 67--73."},{"key":"e_1_3_2_2_37_1","volume-title":"Nervousness, Trust, and Deception from Behavioral Features in Videos. Ph.,D. Dissertation","author":"Walls Bradley L","unstructured":"Bradley L Walls. 2020. An Ai Model to Predict Dominance, Nervousness, Trust, and Deception from Behavioral Features in Videos. Ph.,D. Dissertation. The University of Arizona."},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"crossref","unstructured":"Jinpeng Wang Yuting Gao Ke Li Yiqi Lin Andy J. Ma Hao Cheng Pai Peng Feiyue Huang Rongrong Ji and Xing Sun. 2021. Removing the Background by Adding the Background: Towards Background Robust Self-Supervised Video Representation Learning. In CVPR. 11804--11813.","DOI":"10.1109\/CVPR46437.2021.01163"},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"crossref","unstructured":"Jiang Wang Xiaohan Nie Yin Xia Ying Wu and Song-Chun Zhu. 2014. Cross-View Action Modeling Learning and Recognition. In CVPR. 2649--2656.","DOI":"10.1109\/CVPR.2014.339"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"crossref","unstructured":"Limin Wang Yuanjun Xiong Zhe Wang Yu Qiao Dahua Lin Xiaoou Tang and Luc Van Gool. 2016. Temporal Segment Networks: Towards Good Practices for Deep Action Recognition. In ECCV. 20--36.","DOI":"10.1007\/978-3-319-46484-8_2"},{"key":"e_1_3_2_2_41_1","unstructured":"Yunbo Wang Lu Jiang Ming-Hsuan Yang Li-Jia Li Mingsheng Long and Li Fei-Fei. 2019. Eidetic 3d lstm: A model for video prediction and beyond. In ICLR."},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"crossref","unstructured":"Ping Wei Yang Liu Tianmin Shu Nanning Zheng and Song-Chun Zhu. 2018. Where and Why Are They Looking? Jointly Inferring Human Attention and Intentions in Complex Tasks. In CVPR. 6801--6809.","DOI":"10.1109\/CVPR.2018.00711"},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"crossref","unstructured":"Ping Wei Dan Xie Nanning Zheng and Song-Chun Zhu. 2017. Inferring Human Attention by Learning Latent Intentions. In IJCAI. 1297--1303.","DOI":"10.24963\/ijcai.2017\/180"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"crossref","unstructured":"Danni Xu Ruimin Hu Zheng Wang Linbo Luo Dengshi Li and Wenjun Zeng. 2022. Gaze- and Spacing-flow Unveil Intentions: Hidden Follower Discovery. In ACM MM. 2115--2123.","DOI":"10.1145\/3503161.3548207"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"crossref","unstructured":"Danni Xu Ruimin Hu Zixiang Xiong Zheng Wang Linbo Luo and Dengshi Li. 2021. Trajectory is not Enough: Hidden Following Detection. In ACM MM. 5373--5381.","DOI":"10.1145\/3474085.3475664"},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"crossref","unstructured":"Jian Zhao Yu Cheng Yan Xu Lin Xiong Jianshu Li Fang Zhao Jayashree Karlekar Sugiri Pranata Shengmei Shen Junliang Xing Shuicheng Yan and Jiashi Feng. 2018. Towards Pose Invariant Face Recognition in the Wild. In CVPR. 2207--2216.","DOI":"10.1109\/CVPR.2018.00235"},{"key":"e_1_3_2_2_47_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-019-01181-5"},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-019-01252-7"},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2021.3072171"},{"key":"e_1_3_2_2_50_1","volume-title":"VCD: View-Constraint Disentanglement for Action Recognition. In ICASSP. 2170--2174.","author":"Zhong Xian","year":"2022","unstructured":"Xian Zhong, Zhuo Zhou, Wenxuan Liu, Kui Jiang, Xuemei Jia, Wenxin Huang, and Zheng Wang. 2022b. VCD: View-Constraint Disentanglement for Action Recognition. In ICASSP. 2170--2174."},{"key":"e_1_3_2_2_51_1","unstructured":"Bolei Zhou \u00c0gata Lapedriza Jianxiong Xiao Antonio Torralba and Aude Oliva. 2014. Learning Deep Features for Scene Recognition using Places Database. In NeurIPS. 487--495."}],"event":{"name":"MM '23: The 31st ACM International Conference on Multimedia","location":"Ottawa ON Canada","acronym":"MM '23","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 31st ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3611892","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3581783.3611892","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T00:02:41Z","timestamp":1755820961000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3611892"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,26]]},"references-count":51,"alternative-id":["10.1145\/3581783.3611892","10.1145\/3581783"],"URL":"https:\/\/doi.org\/10.1145\/3581783.3611892","relation":{},"subject":[],"published":{"date-parts":[[2023,10,26]]},"assertion":[{"value":"2023-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}