{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T01:13:20Z","timestamp":1783559600587,"version":"3.55.0"},"publisher-location":"Cham","reference-count":36,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032222664","type":"print"},{"value":"9783032222671","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-22267-1_6","type":"book-chapter","created":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T00:15:51Z","timestamp":1783556151000},"page":"65-77","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Learning Human-Object Interactions in\u00a0Videos with\u00a0Optical Flow"],"prefix":"10.1007","author":[{"given":"Qiyue","family":"Li","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Pengchao","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yingbin","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yichuan","family":"Yin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,2]]},"reference":[{"key":"6_CR1","unstructured":"Ba, J.L., Kiros, J.R., Hinton, G.E.: Layer normalization. arXiv preprint arXiv:1607.06450 (2016)"},{"issue":"3","key":"6_CR2","first-page":"2276","volume":"27","author":"C Baldassano","year":"2017","unstructured":"Baldassano, C., Beck, D.M., Fei-Fei, L.: Human-object interactions are more than the sum of their parts. Cereb. Cortex 27(3), 2276\u20132288 (2017)","journal-title":"Cereb. Cortex"},{"key":"6_CR3","doi-asserted-by":"publisher","unstructured":"Bottou, L.: Large-scale machine learning with stochastic gradient descent. In: Proceedings of COMPSTAT\u20192010, pp. 177\u2013186. Springer, Heidelberg (2010). https:\/\/doi.org\/10.1007\/978-3-7908-2604-3_16","DOI":"10.1007\/978-3-7908-2604-3_16"},{"key":"6_CR4","doi-asserted-by":"crossref","unstructured":"Carreira, J., Zisserman, A.: Quo vadis, action recognition? A new model and the kinetics dataset. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6299\u20136308 (2017)","DOI":"10.1109\/CVPR.2017.502"},{"key":"6_CR5","doi-asserted-by":"crossref","unstructured":"Chiou, M.J., Liao, C.Y., Wang, L.W., Zimmermann, R., Feng, J.: St-hoi: a spatial-temporal baseline for human-object interaction detection in videos. In: Proceedings of the 2021 ACM Workshop on Intelligent Cross-Data Analysis and Retrieval, pp. 9\u201317 (2021)","DOI":"10.1145\/3463944.3469097"},{"key":"6_CR6","doi-asserted-by":"crossref","unstructured":"Crasto, N., Weinzaepfel, P., Alahari, K., Schmid, C.: Mars: motion-augmented RGB stream for action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 7882\u20137891 (2019)","DOI":"10.1109\/CVPR.2019.00807"},{"key":"6_CR7","doi-asserted-by":"crossref","unstructured":"Gao, D., Zhou, L., Ji, L., Zhu, L., Yang, Y., Shou, M.Z.: Mist: multi-modal iterative spatial-temporal transformer for long-form video question answering. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14773\u201314783 (2023)","DOI":"10.1109\/CVPR52729.2023.01419"},{"key":"6_CR8","doi-asserted-by":"crossref","unstructured":"Gkountakos, K., Ioannidis, K., Tsikrika, T., Vrochidis, S., Kompatsiaris, I.: A crowd analysis framework for detecting violence scenes. In: Proceedings of the 2020 International Conference on Multimedia Retrieval, pp. 276\u2013280 (2020)","DOI":"10.1145\/3372278.3390725"},{"key":"6_CR9","doi-asserted-by":"crossref","unstructured":"Goyal, R., et\u00a0al.: The \u201csomething something\" video database for learning and evaluating visual common sense. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 5842\u20135850 (2017)","DOI":"10.1109\/ICCV.2017.622"},{"key":"6_CR10","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"issue":"1","key":"6_CR11","doi-asserted-by":"publisher","first-page":"21","DOI":"10.1615\/JMachLearnModelComput.2023047367","volume":"4","author":"AD Jagtap","year":"2023","unstructured":"Jagtap, A.D., Karniadakis, G.E.: How important are activation functions in regression and classification? A survey, performance comparison, and future directions. J. Mach. Learn. Model. Comput. 4(1), 21\u201375 (2023)","journal-title":"J. Mach. Learn. Model. Comput."},{"key":"6_CR12","doi-asserted-by":"crossref","unstructured":"Jain, A., Zamir, A.R., Savarese, S., Saxena, A.: Structural-RNN: deep learning on spatio-temporal graphs. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5308\u20135317 (2016)","DOI":"10.1109\/CVPR.2016.573"},{"issue":"25","key":"6_CR13","doi-asserted-by":"publisher","first-page":"35619","DOI":"10.1007\/s11042-021-11878-w","volume":"81","author":"V Jain","year":"2022","unstructured":"Jain, V., Al-Turjman, F., Chaudhary, G., Nayar, D., Gupta, V., Kumar, A.: Video captioning: a review of theory, techniques and practices. Multimedia Tools Appl. 81(25), 35619\u201335653 (2022)","journal-title":"Multimedia Tools Appl."},{"key":"6_CR14","unstructured":"Joulin, A., Ciss\u00e9, M., Grangier, D., J\u00e9gou, H., et\u00a0al.: Efficient softmax approximation for gpus. In: International Conference on Machine Learning, pp. 1302\u20131310. PMLR (2017)"},{"issue":"5","key":"6_CR15","doi-asserted-by":"publisher","first-page":"1366","DOI":"10.1007\/s11263-022-01594-9","volume":"130","author":"Y Kong","year":"2022","unstructured":"Kong, Y., Fu, Y.: Human action recognition and prediction: a survey. Int. J. Comput. Vision 130(5), 1366\u20131401 (2022)","journal-title":"Int. J. Comput. Vision"},{"issue":"8","key":"6_CR16","doi-asserted-by":"publisher","first-page":"951","DOI":"10.1177\/0278364913478446","volume":"32","author":"HS Koppula","year":"2013","unstructured":"Koppula, H.S., Gupta, R., Saxena, A.: Learning human activities and object affordances from rgb-d videos. Int. J. Rob. Res. 32(8), 951\u2013970 (2013)","journal-title":"Int. J. Rob. Res."},{"key":"6_CR17","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.neunet.2023.01.019","volume":"163","author":"Q Li","year":"2023","unstructured":"Li, Q., Xie, X., Zhang, J., Shi, G.: Few-shot human-object interaction video recognition with transformers. Neural Netw. 163, 1\u20139 (2023)","journal-title":"Neural Netw."},{"key":"6_CR18","doi-asserted-by":"crossref","unstructured":"Li, Q., Yan, J., Zhang, J., Li, Y., Zheng, Z.: Learning human-object interactions in videos by heterogeneous graph neural networks. In: 2024 IEEE 17th International Conference on Signal Processing (ICSP), pp. 459\u2013464. IEEE (2024)","DOI":"10.1109\/ICSP62129.2024.10846027"},{"key":"6_CR19","doi-asserted-by":"crossref","unstructured":"Liu, C., Jin, Y., Xu, K., Gong, G., Mu, Y.: Beyond short-term snippet: video relation detection with spatio-temporal global context. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10840\u201310849 (2020)","DOI":"10.1109\/CVPR42600.2020.01085"},{"key":"6_CR20","unstructured":"Mao, A., Mohri, M., Zhong, Y.: Cross-entropy loss functions: theoretical analysis and applications. In: International Conference on Machine Learning, pp. 23803\u201323828. PMLR (2023)"},{"key":"6_CR21","doi-asserted-by":"crossref","unstructured":"Materzynska, J., Xiao, T., Herzig, R., Xu, H., Wang, X., Darrell, T.: Something-else: compositional action recognition with spatial-temporal interaction networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1049\u20131059 (2020)","DOI":"10.1109\/CVPR42600.2020.00113"},{"issue":"4","key":"6_CR22","doi-asserted-by":"publisher","first-page":"835","DOI":"10.1109\/TPAMI.2012.175","volume":"35","author":"A Prest","year":"2012","unstructured":"Prest, A., Ferrari, V., Schmid, C.: Explicit modeling of human-object interactions in realistic videos. IEEE Trans. Pattern Anal. Mach. Intell. 35(4), 835\u2013848 (2012)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"6_CR23","doi-asserted-by":"crossref","unstructured":"Qi, S., Wang, W., Jia, B., Shen, J., Zhu, S.C.: Learning human-object interactions by graph parsing neural networks. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 401\u2013417 (2018)","DOI":"10.1007\/978-3-030-01240-3_25"},{"issue":"6","key":"6_CR24","doi-asserted-by":"publisher","first-page":"1137","DOI":"10.1109\/TPAMI.2016.2577031","volume":"39","author":"S Ren","year":"2016","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster r-cnn: towards real-time object detection with region proposal networks. IEEE Trans. Pattern Anal. Mach. Intell. 39(6), 1137\u20131149 (2016)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"6_CR25","doi-asserted-by":"crossref","unstructured":"Shang, X., Di, D., Xiao, J., Cao, Y., Yang, X., Chua, T.S.: Annotating objects and relations in user-generated videos. In: International Conference on Multimedia Retrieval, pp. 279\u2013287 (2019)","DOI":"10.1145\/3323873.3325056"},{"issue":"4","key":"6_CR26","doi-asserted-by":"publisher","first-page":"525","DOI":"10.1177\/0018720816644364","volume":"58","author":"TB Sheridan","year":"2016","unstructured":"Sheridan, T.B.: Human-robot interaction: status and challenges. Hum. Factors 58(4), 525\u2013532 (2016)","journal-title":"Hum. Factors"},{"key":"6_CR27","unstructured":"Simonyan, K., Zisserman, A.: Two-stream convolutional networks for action recognition in videos. In: Advances in Neural Information Processing Systems, pp. 568\u2013576 (2014)"},{"key":"6_CR28","doi-asserted-by":"crossref","unstructured":"Sultani, W., Chen, C., Shah, M.: Real-world anomaly detection in surveillance videos. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6479\u20136488 (2018)","DOI":"10.1109\/CVPR.2018.00678"},{"key":"6_CR29","doi-asserted-by":"crossref","unstructured":"Sunkesula, S.P.R., Dabral, R., Ramakrishnan, G.: Lighten: learning interactions with graph and hierarchical temporal networks for hoi in videos. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 691\u2013699 (2020)","DOI":"10.1145\/3394171.3413778"},{"key":"6_CR30","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"402","DOI":"10.1007\/978-3-030-58536-5_24","volume-title":"Computer Vision \u2013 ECCV 2020","author":"Z Teed","year":"2020","unstructured":"Teed, Z., Deng, J.: RAFT: recurrent all-pairs field transforms for optical flow. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12347, pp. 402\u2013419. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58536-5_24"},{"key":"6_CR31","first-page":"24261","volume":"34","author":"IO Tolstikhin","year":"2021","unstructured":"Tolstikhin, I.O., et al.: MLP-mixer: an all-MLP architecture for vision. Adv. Neural. Inf. Process. Syst. 34, 24261\u201324272 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"issue":"10","key":"6_CR32","doi-asserted-by":"publisher","first-page":"5814","DOI":"10.1109\/TCSVT.2023.3259430","volume":"33","author":"N Wang","year":"2023","unstructured":"Wang, N., et al.: Exploring spatio-temporal graph convolution for video-based human-object interaction recognition. IEEE Trans. Circuits Syst. Video Technol. 33(10), 5814\u20135827 (2023)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"6_CR33","doi-asserted-by":"crossref","unstructured":"Wang, X., Gupta, A.: Videos as space-time region graphs. In: Proceedings of the European Conference on Computer Vision, pp. 399\u2013417 (2018)","DOI":"10.1007\/978-3-030-01228-1_25"},{"issue":"1","key":"6_CR34","doi-asserted-by":"publisher","first-page":"11","DOI":"10.1038\/s41746-019-0087-z","volume":"2","author":"S Yeung","year":"2019","unstructured":"Yeung, S.: A computer vision system for deep learning-based detection of patient mobilization activities in the ICU. NPJ Dig. Med. 2(1), 11 (2019)","journal-title":"NPJ Dig. Med."},{"key":"6_CR35","doi-asserted-by":"crossref","unstructured":"Zeng, L.A., Zheng, W.S.: Multimodal action quality assessment. IEEE Trans. Image Process. (2024)","DOI":"10.1109\/TIP.2024.3362135"},{"key":"6_CR36","doi-asserted-by":"crossref","unstructured":"Zhou, J., Wu, Y.: Temporal feature enhancement dilated convolution network for weakly-supervised temporal action localization. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 6028\u20136037 (2023)","DOI":"10.1109\/WACV56688.2023.00597"}],"container-title":["Lecture Notes in Computer Science","Advances in Computer Graphics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-22267-1_6","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T00:15:55Z","timestamp":1783556155000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-22267-1_6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9783032222664","9783032222671"],"references-count":36,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-22267-1_6","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"2 July 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"CGI","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Computer Graphics International Conference","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Hong Kong","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"14 July 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18 July 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"42","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"cgi2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/www.cgs-network.org\/cgi25","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}