{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T15:07:08Z","timestamp":1784214428329,"version":"3.55.0"},"publisher-location":"Cham","reference-count":68,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031915741","type":"print"},{"value":"9783031915758","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-91575-8_20","type":"book-chapter","created":{"date-parts":[[2025,5,25]],"date-time":"2025-05-25T17:57:19Z","timestamp":1748195839000},"page":"324-342","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["A Vision-Based Framework for\u00a0Human Behavior Understanding in\u00a0Industrial Assembly Lines"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2467-8727","authenticated-orcid":false,"given":"Konstantinos","family":"Papoutsakis","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3106-4758","authenticated-orcid":false,"given":"Nikolaos","family":"Bakalos","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-8833-1411","authenticated-orcid":false,"given":"Konstantinos","family":"Fragkoulis","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-9295-471X","authenticated-orcid":false,"given":"Athena","family":"Zacharia","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Georgia","family":"Kapetadimitri","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8943-4598","authenticated-orcid":false,"given":"Maria","family":"Pateraki","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,5,12]]},"reference":[{"key":"20_CR1","doi-asserted-by":"publisher","unstructured":"Aganian, D., Stephan, B., Eisenbach, M., Stretz, C., Gross, H.M.: Attach dataset: annotated two-handed assembly actions for human action understanding. In: 2023 IEEE International Conference on Robotics and Automation (ICRA), pp. 11367\u201311373 (2023). https:\/\/doi.org\/10.1109\/ICRA48891.2023.10160633","DOI":"10.1109\/ICRA48891.2023.10160633"},{"key":"20_CR2","doi-asserted-by":"crossref","unstructured":"Bacharidis, K., Argyros, A.: Extracting action hierarchies from action labels and their use in deep action recognition. In: 2020 25th International Conference on Pattern Recognition (ICPR) pp. 339\u2013346. IEEE (2021)","DOI":"10.1109\/ICPR48806.2021.9412033"},{"key":"20_CR3","doi-asserted-by":"crossref","unstructured":"Baradel, F., Neverova, N., Wolf, C., Mille, J., Mori, G.: Object level visual reasoning in videos. In: ECCV (2018)","DOI":"10.1007\/978-3-030-01261-8_7"},{"key":"20_CR4","doi-asserted-by":"crossref","unstructured":"Ben-Shabat, Y., et al.: The ikea asm dataset: Understanding people assembling furniture through actions, objects and pose. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV), pp. 847\u2013859 (January 2021)","DOI":"10.1109\/WACV48630.2021.00089"},{"issue":"1","key":"20_CR5","doi-asserted-by":"publisher","first-page":"172","DOI":"10.1109\/TPAMI.2019.2929257","volume":"43","author":"Z Cao","year":"2019","unstructured":"Cao, Z., Hidalgo, G., Simon, T., Wei, S.E., Sheikh, Y.: Openpose: realtime multi-person 2d pose estimation using part affinity fields. IEEE Trans. Pattern Anal. Mach. Intell. 43(1), 172\u2013186 (2019)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"20_CR6","doi-asserted-by":"crossref","unstructured":"Carreira, J., Zisserman, A.: Quo vadis, action recognition? a new model and the kinetics dataset. In: proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6299\u20136308 (2017)","DOI":"10.1109\/CVPR.2017.502"},{"issue":"1","key":"20_CR7","doi-asserted-by":"publisher","first-page":"745","DOI":"10.1038\/s41597-022-01843-z","volume":"9","author":"G Cicirelli","year":"2022","unstructured":"Cicirelli, G., et al.: The ha4m dataset: multi-modal monitoring of an assembly task for human action recognition in manufacturing. Sci. Data 9(1), 745 (2022)","journal-title":"Sci. Data"},{"key":"20_CR8","unstructured":"Cuturi, M., Blondel, M.: Soft-DTW: A differentiable loss function for time-series. In: Proceedings of the 34th International Conference on Machine Learning - Volume 70, pp. 894\u2013903. ICML\u201917, JMLR.org (2017)"},{"key":"20_CR9","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2019)"},{"key":"20_CR10","doi-asserted-by":"publisher","unstructured":"Feichtenhofer, C., Fan, H., Malik, J., He, K.: Slowfast networks for video recognition. In: 2019 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 6201\u20136210 (2019). https:\/\/doi.org\/10.1109\/ICCV.2019.00630","DOI":"10.1109\/ICCV.2019.00630"},{"key":"20_CR11","doi-asserted-by":"crossref","unstructured":"Garcia-Hernando, G., Yuan, S., Baek, S., Kim, T.K.: First-person hand action benchmark with RGB-d videos and 3d hand pose annotations. In: Proceedings of Computer Vision and Pattern Recognition (CVPR) (2018)","DOI":"10.1109\/CVPR.2018.00050"},{"key":"20_CR12","doi-asserted-by":"crossref","unstructured":"Girdhar, R., Carreira, J., Doersch, C., Zisserman, A.: Video action transformer network. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 244\u2013253 (2019)","DOI":"10.1109\/CVPR.2019.00033"},{"key":"20_CR13","doi-asserted-by":"crossref","unstructured":"Gkioxari, G., Girshick, R., Doll\u00e1r, P., He, K.: Detecting and recognizing human-object interactions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 8359\u20138367 (2018)","DOI":"10.1109\/CVPR.2018.00872"},{"key":"20_CR14","doi-asserted-by":"crossref","unstructured":"Gong, J., Foo, L.G., Fan, Z., Ke, Q., Rahmani, H., Liu, J.: Diffpose: Toward more reliable 3d pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13041\u201313051 (2023)","DOI":"10.1109\/CVPR52729.2023.01253"},{"key":"20_CR15","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., Girshick, R.: Mask R-CNN. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2961\u20132969 (2017)","DOI":"10.1109\/ICCV.2017.322"},{"key":"20_CR16","doi-asserted-by":"publisher","unstructured":"Herath, S., Harandi, M., Porikli, F.: Going deeper into action recognition. Image Vision Comput. 60(C), 4-21 (04 2017). https:\/\/doi.org\/10.1016\/j.imavis.2017.01.010,","DOI":"10.1016\/j.imavis.2017.01.010"},{"key":"20_CR17","doi-asserted-by":"crossref","unstructured":"Ji, J., Krishna, R., Fei-Fei, L., Niebles, J.C.: Action genome: actions as compositions of spatio-temporal scene graphs. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10236\u201310247 (2020)","DOI":"10.1109\/CVPR42600.2020.01025"},{"issue":"4","key":"20_CR18","doi-asserted-by":"publisher","first-page":"199","DOI":"10.1016\/0003-6870(77)90164-8","volume":"8","author":"O Karhu","year":"1977","unstructured":"Karhu, O., Kansi, P., Kuorinka, I.: Correcting working postures in industry: a practical method for analysis. Appl. Ergon. 8(4), 199\u2013201 (1977)","journal-title":"Appl. Ergon."},{"key":"20_CR19","doi-asserted-by":"crossref","unstructured":"Kim, T.S., Reiter, A.: Interpretable 3d human action analysis with temporal convolutional networks. In: 2017 IEEE Conference on Computer Vision and Pattern Recognition Workshops (CVPRW), pp. 1623\u20131631. IEEE (2017)","DOI":"10.1109\/CVPRW.2017.207"},{"key":"20_CR20","doi-asserted-by":"crossref","unstructured":"Konstantinidis, D., Dimitropoulos, K., Daras, P.: Towards real-time generalized ergonomic risk assessment for the prevention of musculoskeletal disorders. In: 14th ACM International Conference on Pervasive Technologies Related to Assistive Environments Conference (PETRA) (June-July 2021)","DOI":"10.1145\/3453892.3461344"},{"key":"20_CR21","doi-asserted-by":"crossref","unstructured":"LeCun, Y., Bengio, Y., Hinton, G.: Deep learning. Nature 521(7553), 436\u2013444 (2015)","DOI":"10.1038\/nature14539"},{"key":"20_CR22","doi-asserted-by":"crossref","unstructured":"Lei, Q., Du, J.X., Zhang, H., Ye, S., Chen, D.: A survey of vision-based human action evaluation methods. Sensors (Basel, Switzerland) 19 (2019)","DOI":"10.3390\/s19194129"},{"key":"20_CR23","doi-asserted-by":"crossref","unstructured":"Li, M., Chen, S., Chen, X., Zhang, Y., Wang, Y., Tian, Q.: Symbiotic graph neural networks for 3d skeleton-based human action recognition and motion prediction. IEEE Trans. Pattern Anal. Mach. Intell. 44(6), 3316\u20133333 (2021)","DOI":"10.1109\/TPAMI.2021.3053765"},{"key":"20_CR24","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2019.2916873","author":"J Liu","year":"2019","unstructured":"Liu, J., Shahroudy, A., Perez, M., Wang, G., Duan, L.Y., Kot, A.C.: Ntu rgb+d 120: a large-scale benchmark for 3d human activity understanding. IEEE Trans. Pattern Anal. Mach. Intell. (2019). https:\/\/doi.org\/10.1109\/TPAMI.2019.2916873","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"4","key":"20_CR25","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3524497","volume":"55","author":"W Liu","year":"2022","unstructured":"Liu, W., Bao, Q., Sun, Y., Mei, T.: Recent advances of monocular 2d and 3d human pose estimation: a deep learning perspective. ACM Comput. Surv. 55(4), 1\u201341 (2022)","journal-title":"ACM Comput. Surv."},{"key":"20_CR26","doi-asserted-by":"crossref","unstructured":"Ma, C.Y., Kadav, A., Melvin, I., Kira, Z., AlRegib, G., Graf, H.P.: Attend and interact: higher-order object interactions for video understanding. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6790\u20136800 (2018)","DOI":"10.1109\/CVPR.2018.00710"},{"key":"20_CR27","doi-asserted-by":"publisher","unstructured":"Mahasseni, B., Todorovic, S.: Regularizing long short term memory with 3d human-skeleton sequences for action recognition. In: 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3054\u20133062 (2016). https:\/\/doi.org\/10.1109\/CVPR.2016.333","DOI":"10.1109\/CVPR.2016.333"},{"key":"20_CR28","doi-asserted-by":"crossref","unstructured":"Materzynska, J., Xiao, T., Herzig, R., Xu, H., Wang, X., Darrell, T.: Something-else: Compositional action recognition with spatial-temporal interaction networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1049\u20131059 (2020)","DOI":"10.1109\/CVPR42600.2020.00113"},{"issue":"14","key":"20_CR29","doi-asserted-by":"publisher","first-page":"1529","DOI":"10.1177\/0278364919882089","volume":"38","author":"P Maurice","year":"2019","unstructured":"Maurice, P., et al.: Human movement and ergonomics: an industry-oriented dataset for collaborative robotics. Int. J. Robot. Res. 38(14), 1529\u20131537 (2019)","journal-title":"Int. J. Robot. Res."},{"key":"20_CR30","doi-asserted-by":"crossref","unstructured":"McAtamney, L., Hignett, S.: Rapid entire body assessment. In: Handbook of Human Factors and Ergonomics Methods, pp. 97\u2013108. CRC Press (2004)","DOI":"10.1201\/9780203489925-17"},{"key":"20_CR31","doi-asserted-by":"publisher","unstructured":"Moeslund, T.B., Hilton, A., Kr\u00fcger, V.: A survey of advances in vision-based human motion capture and analysis. Comput. Vision Image Understanding 104(2), 90\u2013126 (2006). https:\/\/doi.org\/10.1016\/j.cviu.2006.08.002, https:\/\/www.sciencedirect.com\/science\/article\/pii\/S1077314206001263, special Issue on Modeling People: Vision-based understanding of a person\u2019s shape, appearance, movement and behaviour","DOI":"10.1016\/j.cviu.2006.08.002"},{"key":"20_CR32","doi-asserted-by":"publisher","unstructured":"Moriwaki, K., Nakano, G., Inoshita, T.: The brio-ta dataset: Understanding anomalous assembly process in manufacturing. In: 2022 IEEE International Conference on Image Processing (ICIP), pp. 1991\u20131995 (2022). https:\/\/doi.org\/10.1109\/ICIP46576.2022.9897369","DOI":"10.1109\/ICIP46576.2022.9897369"},{"key":"20_CR33","doi-asserted-by":"publisher","unstructured":"Morshed, M.G., Sultana, T., Alam, A., Lee, Y.K.: Human action recognition: a taxonomy-based survey, updates, and opportunities. Sensors 23(4) (2023). https:\/\/doi.org\/10.3390\/s23042182, https:\/\/www.mdpi.com\/1424-8220\/23\/4\/2182","DOI":"10.3390\/s23042182"},{"issue":"9","key":"20_CR34","doi-asserted-by":"publisher","first-page":"1290","DOI":"10.1080\/001401398186315","volume":"41","author":"E Occhipinti","year":"1998","unstructured":"Occhipinti, E.: OCRA: a concise index for the assessment of exposure to repetitive movements of the upper limbs. Ergonomics 41(9), 1290\u20131311 (1998)","journal-title":"Ergonomics"},{"key":"20_CR35","doi-asserted-by":"publisher","unstructured":"Papoutsakis, K., et al.: Detection of physical strain and fatigue in industrial environments using visual and non-visual low-cost sensors. Technologies 10(2) (2022). https:\/\/doi.org\/10.3390\/technologies10020042, https:\/\/www.mdpi.com\/2227-7080\/10\/2\/42","DOI":"10.3390\/technologies10020042"},{"key":"20_CR36","unstructured":"Papoutsakis, K.E., Argyros, A.A.: Unsupervised and explainable assessment of video similarity. In: BMVC, p.\u00a0151 (2019)"},{"key":"20_CR37","doi-asserted-by":"crossref","unstructured":"Parsa, B., Banerjee, A.G.: A multi-task learning approach for human activity segmentation and ergonomics risk assessment. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV), pp. 2352\u20132362 (January 2021)","DOI":"10.1109\/WACV48630.2021.00240"},{"key":"20_CR38","doi-asserted-by":"crossref","unstructured":"Parsa, B., narayanan, A.L., Dariush, B.: Spatio-temporal pyramid graph convolutions for human action recognition and postural assessment. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV) (March 2020)","DOI":"10.1109\/WACV45572.2020.9093368"},{"issue":"4","key":"20_CR39","doi-asserted-by":"publisher","first-page":"3153","DOI":"10.1109\/LRA.2019.2925305","volume":"4","author":"B Parsa","year":"2019","unstructured":"Parsa, B., Samani, E.U., Hendrix, R., Devine, C., Singh, S.M., Devasia, S., Banerjee, A.G.: Toward ergonomic risk prediction via segmentation of indoor object manipulation actions using spatiotemporal convolutional networks. IEEE Robot. Autom. Letters 4(4), 3153\u20133160 (2019). https:\/\/doi.org\/10.1109\/LRA.2019.2925305","journal-title":"IEEE Robot. Autom. Letters"},{"key":"20_CR40","doi-asserted-by":"crossref","unstructured":"Pavlakos, G., Zhu, L., Zhou, X., Daniilidis, K.: Learning to estimate 3D human pose and shape from a single color image. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 459\u2013468 (2018)","DOI":"10.1109\/CVPR.2018.00055"},{"issue":"6","key":"20_CR41","doi-asserted-by":"publisher","first-page":"976","DOI":"10.1016\/j.imavis.2009.11.014","volume":"28","author":"R Poppe","year":"2010","unstructured":"Poppe, R.: A survey on vision-based human action recognition. Image Vis. Comput. 28(6), 976\u2013990 (2010). https:\/\/doi.org\/10.1016\/j.imavis.2009.11.014","journal-title":"Image Vis. Comput."},{"key":"20_CR42","doi-asserted-by":"crossref","unstructured":"Qammaz, A., Argyros, A.: A unified approach for occlusion tolerant 3d facial pose capture and gaze estimation using mocapnets. In: IEEE\/CVF International Conference on Computer Vision Workshops (AMFG 2023 - ICCVW 2023), pp. 3178\u20133188. IEEE, Paris, France (October 2023)","DOI":"10.1109\/ICCVW60793.2023.00342"},{"key":"20_CR43","doi-asserted-by":"crossref","unstructured":"Qammaz, A., Argyros, A.A.: Occlusion-tolerant and personalized 3D human pose estimation in rgb images. In: IEEE International Conference on Pattern Recognition (ICPR 2020) (January 2021). http:\/\/users.ics.forth.gr\/ argyros\/res_mocapnet_II.html","DOI":"10.1109\/ICPR48806.2021.9411956"},{"key":"20_CR44","doi-asserted-by":"crossref","unstructured":"Ragusa, F., et al.: Enigma-51: towards a fine-grained understanding of human behavior in industrial scenarios. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 4549\u20134559 (2024)","DOI":"10.1109\/WACV57701.2024.00449"},{"issue":"6","key":"20_CR45","doi-asserted-by":"publisher","first-page":"616","DOI":"10.1080\/1463922X.2012.678283","volume":"14","author":"K Schaub","year":"2013","unstructured":"Schaub, K., Caragnano, G., Britzke, B., Bruder, R.: The European assembly worksheet. Theor. Issues Ergon. Sci. 14(6), 616\u2013639 (2013)","journal-title":"Theor. Issues Ergon. Sci."},{"issue":"3\u20134","key":"20_CR46","doi-asserted-by":"publisher","first-page":"398","DOI":"10.1504\/IJHFMS.2012.051581","volume":"3","author":"KG Schaub","year":"2012","unstructured":"Schaub, K.G., et al.: Ergonomic assessment of automotive assembly tasks with digital human modelling and the \u2018ergonomics assessment worksheet\u2019(eaws). Int. J. Human Factors Modell. Simulation 3(3\u20134), 398\u2013426 (2012)","journal-title":"Int. J. Human Factors Modell. Simulation"},{"key":"20_CR47","doi-asserted-by":"crossref","unstructured":"Schoonbeek, T.J., Houben, T., Onvlee, H., Van\u00a0der Sommen, F., et\u00a0al.: Industreal: a dataset for procedure step recognition handling execution errors in egocentric videos in an industrial-like setting. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision, pp. 4365\u20134374 (2024)","DOI":"10.1109\/WACV57701.2024.00431"},{"key":"20_CR48","doi-asserted-by":"crossref","unstructured":"Sener, F., et al.: Assembly101: a large-scale multi-view video dataset for understanding procedural activities. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 21096\u201321106 (2022)","DOI":"10.1109\/CVPR52688.2022.02042"},{"key":"20_CR49","doi-asserted-by":"publisher","DOI":"10.1016\/j.autcon.2021.103725","volume":"128","author":"J Seo","year":"2021","unstructured":"Seo, J., Lee, S.: Automated postural ergonomic risk assessment using vision-based posture classification. Autom. Constr. 128, 103725 (2021). https:\/\/doi.org\/10.1016\/j.autcon.2021.103725","journal-title":"Autom. Constr."},{"key":"20_CR50","doi-asserted-by":"publisher","unstructured":"Shafti, A., Ataka, A., Lazpita, B.U., Shiva, A., Wurdemann, H., Althoefer, K.: Real-time robot-assisted ergonomics*. In: 2019 International Conference on Robotics and Automation (ICRA), pp. 1975\u20131981 (2019). https:\/\/doi.org\/10.1109\/ICRA.2019.8793739","DOI":"10.1109\/ICRA.2019.8793739"},{"key":"20_CR51","doi-asserted-by":"crossref","unstructured":"Shin, S., Kim, J., Halilaj, E., Black, M.J.: Wham: Reconstructing world-grounded humans with accurate 3d motion. In: Computer Vision and Pattern Recognition (CVPR) (2024)","DOI":"10.1109\/CVPR52733.2024.00202"},{"key":"20_CR52","doi-asserted-by":"crossref","unstructured":"Tekin, B., Bogo, F., Pollefeys, M.: H+ o: Unified egocentric recognition of 3d hand-object poses and interactions. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4511\u20134520 (2019)","DOI":"10.1109\/CVPR.2019.00464"},{"key":"20_CR53","doi-asserted-by":"crossref","unstructured":"Tenorth, M., Bandouch, J., Beetz, M.: The TUM Kitchen Data Set of Everyday Manipulation Activities for Motion Tracking and Action Recognition. In: IEEE International Workshop THEMIS, ICCV2009 (2009)","DOI":"10.1109\/ICCVW.2009.5457583"},{"key":"20_CR54","doi-asserted-by":"crossref","unstructured":"Tripathi, S., M\u00fcller, L., Huang, C.H.P., Omid, T., Black, M.J., Tzionas, D.: 3D human pose estimation via intuitive physics. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (June 2023)","DOI":"10.1109\/CVPR52729.2023.00457"},{"key":"20_CR55","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Advances Neural Information Processing Systems vol. 30 (2017)"},{"key":"20_CR56","doi-asserted-by":"publisher","first-page":"118","DOI":"10.1016\/j.cviu.2018.04.007","volume":"171","author":"P Wang","year":"2018","unstructured":"Wang, P., Li, W., Ogunbona, P., Wan, J., Escalera, S.: RGB-D-based human motion recognition with deep learning: a survey. Comput. Vis. Image Underst. 171, 118\u2013139 (2018)","journal-title":"Comput. Vis. Image Underst."},{"key":"20_CR57","doi-asserted-by":"crossref","unstructured":"Wang, Y., Ajaykumar, G., Huang, C.M.: See what i see: enabling user-centric robotic assistance using first-person demonstrations. In: Proceedings of the 2020 ACM\/IEEE International Conference on Human-Robot Interaction, pp. 639\u2013648 (2020)","DOI":"10.1145\/3319502.3374820"},{"key":"20_CR58","doi-asserted-by":"crossref","unstructured":"Wang, Y., Daniilidis, K.: Refit: recurrent fitting network for 3d human recovery. In: International Conference on Computer Vision (ICCV) (2023)","DOI":"10.1109\/ICCV51070.2023.01346"},{"key":"20_CR59","unstructured":"Womack, J.: From lean tools to lean management. Lean Enterprise Inst. 21 (2006)"},{"key":"20_CR60","doi-asserted-by":"publisher","unstructured":"Wu, C.Y., et al.: Memvit: Memory-augmented multiscale vision transformer for efficient long-term video recognition. In: 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 13577\u201313587 (2022). https:\/\/doi.org\/10.1109\/CVPR52688.2022.01322","DOI":"10.1109\/CVPR52688.2022.01322"},{"key":"20_CR61","doi-asserted-by":"crossref","unstructured":"Xu, B., Wong, Y., Li, J., Zhao, Q., Kankanhalli, M.S.: Learning to detect human-object interactions with knowledge. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (June 2019)","DOI":"10.1109\/CVPR.2019.00212"},{"key":"20_CR62","doi-asserted-by":"crossref","unstructured":"Yan, S., Xiong, Y., Lin, D.: Spatial temporal graph convolutional networks for skeleton-based action recognition. In: AAAI (2018)","DOI":"10.1609\/aaai.v32i1.12328"},{"key":"20_CR63","doi-asserted-by":"publisher","first-page":"152","DOI":"10.1016\/j.aei.2017.11.001","volume":"34","author":"X Yan","year":"2017","unstructured":"Yan, X., Li, H., Wang, C., Seo, J., Zhang, H., Wang, H.: Development of ergonomic posture recognition technique based on 2d ordinary camera for construction hazard prevention through view-invariant features in 2d skeleton motion. Adv. Eng. Inform. 34, 152\u2013163 (2017). https:\/\/doi.org\/10.1016\/j.aei.2017.11.001","journal-title":"Adv. Eng. Inform."},{"key":"20_CR64","doi-asserted-by":"crossref","unstructured":"Zhang, C., Gupta, A., Zisserman, A.: Temporal query networks for fine-grained video understanding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4486\u20134496 (2021)","DOI":"10.1109\/CVPR46437.2021.00446"},{"key":"20_CR65","doi-asserted-by":"publisher","unstructured":"Zhang, H.B., et al.: A comprehensive survey of vision-based human action recognition methods. Sensors 19(5) (2019). https:\/\/doi.org\/10.3390\/s19051005, https:\/\/www.mdpi.com\/1424-8220\/19\/5\/1005","DOI":"10.3390\/s19051005"},{"key":"20_CR66","doi-asserted-by":"crossref","unstructured":"Zhang, P., Lan, C., Xing, J., Zeng, W., Xue, J., Zheng, N.: View adaptive recurrent neural networks for high performance human action recognition from skeleton data. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2117\u20132126 (2017)","DOI":"10.1109\/ICCV.2017.233"},{"key":"20_CR67","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Shen, H., Liu, Y., et\u00a0al.: Efficient long-range transformers: You need to attend more, but not necessarily at every layer. arXiv preprint arXiv:2310.12442 (2023)","DOI":"10.18653\/v1\/2023.findings-emnlp.183"},{"issue":"1","key":"20_CR68","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3603618","volume":"56","author":"C Zheng","year":"2023","unstructured":"Zheng, C., et al.: Deep learning-based human pose estimation: a survey. ACM Comput. Surv. 56(1), 1\u201337 (2023)","journal-title":"ACM Comput. Surv."}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024 Workshops"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-91575-8_20","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,25]],"date-time":"2025-05-25T17:57:32Z","timestamp":1748195852000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-91575-8_20"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9783031915741","9783031915758"],"references-count":68,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-91575-8_20","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"12 May 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}