{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,13]],"date-time":"2026-02-13T08:59:25Z","timestamp":1770973165701,"version":"3.50.1"},"reference-count":54,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2024,12,26]],"date-time":"2024-12-26T00:00:00Z","timestamp":1735171200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,12,26]],"date-time":"2024-12-26T00:00:00Z","timestamp":1735171200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2025,2]]},"DOI":"10.1007\/s00530-024-01612-5","type":"journal-article","created":{"date-parts":[[2024,12,26]],"date-time":"2024-12-26T02:37:52Z","timestamp":1735180672000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Research on passengers behavior recognition method in public transport vehicles based on efficient 3D CNN"],"prefix":"10.1007","volume":"31","author":[{"given":"Yumeng","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kaixing","family":"Fan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ying","family":"Yu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,12,26]]},"reference":[{"key":"1612_CR1","doi-asserted-by":"publisher","DOI":"10.1016\/j.jpubtr.2022.100009","volume":"24","author":"R Bridgelall","year":"2022","unstructured":"Bridgelall, R.: Using artificial intelligence to derive a public transit risk index. J. Public Transp. 24, 100009 (2022)","journal-title":"J. Public Transp."},{"issue":"10","key":"1612_CR2","doi-asserted-by":"publisher","first-page":"18076","DOI":"10.1109\/TITS.2022.3151264","volume":"23","author":"MN Azadani","year":"2022","unstructured":"Azadani, M.N., Boukerche, A.: Siamese temporal convolutional networks for driver identification using driver steering behavior analysis. IEEE Trans. Intell. Transp. Syst. 23(10), 18076\u201318087 (2022)","journal-title":"IEEE Trans. Intell. Transp. Syst."},{"issue":"3","key":"1612_CR3","first-page":"400","volume":"12","author":"D Stanojevi\u0107","year":"2020","unstructured":"Stanojevi\u0107, D., Stanojevi\u0107, P., Jovanovi\u0107, D., et al.: Impact of riders\u2019 lifestyle on their risky behavior and road traffic accident risk. J. Transp. Saf. Secur. 12(3), 400\u2013418 (2020)","journal-title":"J. Transp. Saf. Secur."},{"key":"1612_CR4","doi-asserted-by":"publisher","DOI":"10.1016\/j.jth.2020.100905","volume":"18","author":"X Wang","year":"2020","unstructured":"Wang, X., Yuen, K.F., Shi, W., et al.: The determinants of passengers\u2019 safety behaviour on public transport. J. Transp. Health 18, 100905 (2020)","journal-title":"J. Transp. Health"},{"issue":"1","key":"1612_CR5","doi-asserted-by":"publisher","first-page":"197","DOI":"10.1007\/s00530-022-00981-z","volume":"29","author":"S Liu","year":"2023","unstructured":"Liu, S., He, N., Wang, C., et al.: Lightweight human pose estimation algorithm based on polarized self-attention. Multimed. Syst. 29(1), 197\u2013210 (2023)","journal-title":"Multimed. Syst."},{"issue":"12","key":"1612_CR6","doi-asserted-by":"publisher","first-page":"13029","DOI":"10.1109\/JSEN.2021.3069927","volume":"21","author":"E Ramanujam","year":"2021","unstructured":"Ramanujam, E., Perumal, T., Padmavathi, S.: Human activity recognition with smartphone and wearable sensors using deep learning techniques: a review. IEEE Sens. J. 21(12), 13029\u201313040 (2021)","journal-title":"IEEE Sens. J."},{"key":"1612_CR7","doi-asserted-by":"crossref","unstructured":"Tu, I., Bhalerao, A., Griffiths, N., et al.: Dual viewpoint passenger state classification using 3D CNNs. In: 2018 IEEE Intelligent Vehicles Symposium (IV), pp. 2163\u20132169. IEEE (2018)","DOI":"10.1109\/IVS.2018.8500564"},{"key":"1612_CR8","doi-asserted-by":"crossref","unstructured":"Kao, S.F., Lin, H.Y.: Passenger detection, counting, and action recognition for self-driving public transport vehicles. In: 2021 IEEE Intelligent Vehicles Symposium (IV), pp. 572\u2013577. IEEE (2021)","DOI":"10.1109\/IV48863.2021.9575797"},{"key":"1612_CR9","doi-asserted-by":"crossref","unstructured":"Tseng, C.H., Lin, H.Y.: A vision-based system for abnormal behavior detection and recognition of bus passengers. In: 2022 IEEE 25th International Conference on Intelligent Transportation Systems (ITSC), pp. 2134\u20132139. IEEE (2022)","DOI":"10.1109\/ITSC55140.2022.9921801"},{"key":"1612_CR10","doi-asserted-by":"crossref","unstructured":"Wang, H., Schmid, C.: Action recognition with improved trajectories. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 3551\u20133558 (2013)","DOI":"10.1109\/ICCV.2013.441"},{"key":"1612_CR11","doi-asserted-by":"publisher","first-page":"346","DOI":"10.1016\/j.patcog.2017.02.030","volume":"68","author":"M Liu","year":"2017","unstructured":"Liu, M., Liu, H., Chen, C.: Enhanced skeleton visualization for view invariant human action recognition. Pattern Recogn. 68, 346\u2013362 (2017)","journal-title":"Pattern Recogn."},{"key":"1612_CR12","doi-asserted-by":"crossref","unstructured":"Toshev, A., Szegedy, C.: Deeppose: human pose estimation via deep neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1653\u20131660 (2014)","DOI":"10.1109\/CVPR.2014.214"},{"key":"1612_CR13","doi-asserted-by":"crossref","unstructured":"Sun, X., Shang, J., Liang, S., et al.: Compositional human pose regression. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2602\u20132611 (2017)","DOI":"10.1109\/ICCV.2017.284"},{"key":"1612_CR14","doi-asserted-by":"crossref","unstructured":"Chen, Y., Wang, Z., Peng, Y., et al.: Cascaded pyramid network for multi-person pose estimation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7103\u20137112 (2018)","DOI":"10.1109\/CVPR.2018.00742"},{"key":"1612_CR15","doi-asserted-by":"crossref","unstructured":"Cao, Z., Simon, T., Wei, S.E., et al.: Realtime multi-person 2d pose estimation using part affinity fields. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7291\u20137299 (2017)","DOI":"10.1109\/CVPR.2017.143"},{"key":"1612_CR16","doi-asserted-by":"crossref","unstructured":"Sun, K., Xiao, B., Liu, D., et al.: Deep high-resolution representation learning for human pose estimation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5693\u20135703 (2019)","DOI":"10.1109\/CVPR.2019.00584"},{"key":"1612_CR17","doi-asserted-by":"crossref","unstructured":"Du, Y., Wang, W., Wang, L.: Hierarchical recurrent neural network for skeleton based action recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1110\u20131118 (2015)","DOI":"10.1109\/CVPR.2015.7298714"},{"issue":"10","key":"1612_CR18","doi-asserted-by":"publisher","first-page":"2481","DOI":"10.1109\/TMM.2019.2960588","volume":"22","author":"D Avola","year":"2019","unstructured":"Avola, D., Cascio, M., Cinque, L., et al.: 2-D skeleton-based action recognition via two-branch stacked LSTM-RNNs. IEEE Trans. Multimed. 22(10), 2481\u20132496 (2019)","journal-title":"IEEE Trans. Multimed."},{"issue":"9","key":"1612_CR19","doi-asserted-by":"publisher","first-page":"4382","DOI":"10.1109\/TIP.2018.2837386","volume":"27","author":"H Wang","year":"2018","unstructured":"Wang, H., Wang, L.: Beyond joints: Learning representations from primitive geometries for skeleton-based action recognition and detection. IEEE Trans. Image Process. 27(9), 4382\u20134394 (2018)","journal-title":"IEEE Trans. Image Process."},{"key":"1612_CR20","doi-asserted-by":"publisher","DOI":"10.1016\/j.asoc.2021.107236","volume":"104","author":"W Xu","year":"2021","unstructured":"Xu, W., Wu, M., Zhu, J., et al.: Multi-scale skeleton adaptive weighted GCN for skeleton-based human action recognition in IoT. Appl. Soft Comput. 104, 107236 (2021)","journal-title":"Appl. Soft Comput."},{"key":"1612_CR21","doi-asserted-by":"publisher","first-page":"41403","DOI":"10.1109\/ACCESS.2022.3164711","volume":"10","author":"Q Wang","year":"2022","unstructured":"Wang, Q., Zhang, K., Asghar, M.A.: Skeleton-based ST-GCN for human action recognition with extended skeleton graph and partitioning strategy. IEEE Access 10, 41403\u201341410 (2022)","journal-title":"IEEE Access"},{"issue":"1","key":"1612_CR22","doi-asserted-by":"publisher","first-page":"342","DOI":"10.1109\/TCSVT.2022.3201186","volume":"33","author":"X Xiong","year":"2022","unstructured":"Xiong, X., Min, W., Wang, Q., et al.: Human skeleton feature optimizer and adaptive structure enhancement graph convolution network for action recognition. IEEE Trans. Circuits Syst. Video Technol. 33(1), 342\u2013353 (2022)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"1612_CR23","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Fan, H., Malik, J., et al.: Slowfast networks for video recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6202\u20136211 (2019)","DOI":"10.1109\/ICCV.2019.00630"},{"key":"1612_CR24","doi-asserted-by":"crossref","unstructured":"Ma, X., Dai, X., Bai, Y., et al.: Rewrite the stars. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5694\u20135703 (2024)","DOI":"10.1109\/CVPR52733.2024.00544"},{"key":"1612_CR25","unstructured":"Yang, L., Zhang, R.Y., Li, L., et al.: Simam: a simple, parameter-free attention module for convolutional neural networks. In: International Conference on Machine Learning, pp. 11863\u201311874. PMLR (2021)"},{"key":"1612_CR26","unstructured":"Soomro, K., Zamir, A.R., Shah, M.: UCF101: a dataset of 101 human actions classes from videos in the wild (2012). arXiv:1212.0402"},{"key":"1612_CR27","doi-asserted-by":"crossref","unstructured":"Kuehne, H., Jhuang, H., Garrote, E., et al.: HMDB: a large video database for human motion recognition. In: 2011 International Conference on Computer Vision, pp. 2556\u20132563. IEEE (2011)","DOI":"10.1109\/ICCV.2011.6126543"},{"key":"1612_CR28","doi-asserted-by":"crossref","unstructured":"Shao, D., Zhao, Y., Dai, B., et al.: Finegym: a hierarchical video dataset for fine-grained action understanding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2616\u20132625 (2020)","DOI":"10.1109\/CVPR42600.2020.00269"},{"key":"1612_CR29","unstructured":"Yang, B., Bender, G., Le, Q.V., et al.: Condconv: conditionally parameterized convolutions for efficient inference. In: Advances in Neural Information Processing Systems, p. 32 (2019)"},{"key":"1612_CR30","doi-asserted-by":"crossref","unstructured":"Wang, X., Girshick, R., Gupta, A., et al.: Non-local neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7794\u20137803 (2018)","DOI":"10.1109\/CVPR.2018.00813"},{"key":"1612_CR31","doi-asserted-by":"crossref","unstructured":"Stergiou, A., Poppe, R.: Spatio-temporal FAST 3D convolutions for human action recognition. In: 2019 18th IEEE International Conference on Machine Learning and Applications (ICMLA), pp. 183\u2013190. IEEE (2019)","DOI":"10.1109\/ICMLA.2019.00036"},{"issue":"12","key":"1612_CR32","first-page":"3482","volume":"39","author":"M Guo","year":"2019","unstructured":"Guo, M., Song, Q., Xu, Z., et al.: Human behavior recognition algorithm based on three-dimensional residual dense network. J. Comput. Appl. 39(12), 3482 (2019)","journal-title":"J. Comput. Appl."},{"key":"1612_CR33","doi-asserted-by":"crossref","unstructured":"Hu, J., Shen, L., Sun, G.: Squeeze-and-excitation networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7132\u20137141 (2018)","DOI":"10.1109\/CVPR.2018.00745"},{"issue":"3","key":"1612_CR34","doi-asserted-by":"publisher","DOI":"10.1088\/1757-899X\/569\/3\/032035","volume":"569","author":"X Wang","year":"2019","unstructured":"Wang, X., Miao, Z., Zhang, R., et al.: I3d-lstm: a new model for human action recognition. IOP Conf Ser Mater Sci Eng 569(3), 032035 (2019)","journal-title":"IOP Conf Ser Mater Sci Eng"},{"key":"1612_CR35","doi-asserted-by":"crossref","unstructured":"Choutas, V., Weinzaepfel, P., Revaud, J., et al.: Potion: pose motion representation for action recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7024\u20137033 (2018)","DOI":"10.1109\/CVPR.2018.00734"},{"key":"1612_CR36","doi-asserted-by":"crossref","unstructured":"Yan A, Wang Y, Li Z, et al. PA3D: Pose-action 3D machine for video recognition. In: Proceedings of the ieee\/cvf conference on computer vision and pattern recognition. 2019: 7922\u20137931.","DOI":"10.1109\/CVPR.2019.00811"},{"key":"1612_CR37","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C.: X3d: expanding architectures for efficient video recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 203\u2013213 (2020)","DOI":"10.1109\/CVPR42600.2020.00028"},{"key":"1612_CR38","doi-asserted-by":"crossref","unstructured":"Duan, H., Zhao, Y., Chen, K., et al.: Revisiting skeleton-based action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2969\u20132978 (2022)","DOI":"10.1109\/CVPR52688.2022.00298"},{"key":"1612_CR39","doi-asserted-by":"crossref","unstructured":"Yue-Hei Ng, J., Hausknecht, M., Vijayanarasimhan, S., et al.: Beyond short snippets: Deep networks for video classification. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4694\u20134702 (2015)","DOI":"10.1109\/CVPR.2015.7299101"},{"key":"1612_CR40","doi-asserted-by":"crossref","unstructured":"Tran, D., Bourdev, L., Fergus, R., et al.: Learning spatiotemporal features with 3d convolutional networks. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 4489\u20134497 (2015)","DOI":"10.1109\/ICCV.2015.510"},{"issue":"3","key":"1612_CR41","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pone.0265115","volume":"17","author":"G Yang","year":"2022","unstructured":"Yang, G., Yang, Y., Lu, Z., et al.: STA-TSN: spatial-temporal attention temporal segment network for action recognition in video. PLoS ONE 17(3), e0265115 (2022)","journal-title":"PLoS ONE"},{"key":"1612_CR42","doi-asserted-by":"publisher","DOI":"10.1016\/j.cviu.2022.103406","volume":"219","author":"I Dave","year":"2022","unstructured":"Dave, I., Gupta, R., Rizve, M.N., et al.: Tclr: temporal contrastive learning for video representation. Comput. Vis. Image Underst. 219, 103406 (2022)","journal-title":"Comput. Vis. Image Underst."},{"issue":"11","key":"1612_CR43","doi-asserted-by":"publisher","first-page":"2740","DOI":"10.1109\/TPAMI.2018.2868668","volume":"41","author":"L Wang","year":"2018","unstructured":"Wang, L., Xiong, Y., Wang, Z., et al.: Temporal segment networks for action recognition in videos. IEEE Trans. Pattern Anal. Mach. Intell. 41(11), 2740\u20132755 (2018)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"1612_CR44","doi-asserted-by":"crossref","unstructured":"Zhou, B., Andonian, A., Oliva, A., et al.: Temporal relational reasoning in videos. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 803\u2013818 (2018)","DOI":"10.1007\/978-3-030-01246-5_49"},{"key":"1612_CR45","doi-asserted-by":"crossref","unstructured":"Lin, J., Gan, C., Han, S.: Tsm: temporal shift module for efficient video understanding. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7083\u20137093 (2019)","DOI":"10.1109\/ICCV.2019.00718"},{"key":"1612_CR46","doi-asserted-by":"crossref","unstructured":"Carreira, J., Zisserman, A.: Quo vadis, action recognition? a new model and the kinetics dataset. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6299\u20136308 (2017)","DOI":"10.1109\/CVPR.2017.502"},{"key":"1612_CR47","doi-asserted-by":"crossref","unstructured":"Yan, S., Xiong, Y., Lin, D.: Spatial temporal graph convolutional networks for skeleton-based action recognition. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol 32, no 1 (2018)","DOI":"10.1609\/aaai.v32i1.12328"},{"key":"1612_CR48","first-page":"8046","volume":"34","author":"M Kim","year":"2021","unstructured":"Kim, M., Kwon, H., Wang, C., et al.: Relational self-attention: what\u2019s missing in attention for video understanding. Adv. Neural. Inf. Process. Syst. 34, 8046\u20138059 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1612_CR49","doi-asserted-by":"crossref","unstructured":"Chen, Y., Zhang, Z., Yuan, C., et al.: Channel-wise topology refinement graph convolution for skeleton-based action recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 13359\u201313368 (2021)","DOI":"10.1109\/ICCV48922.2021.01311"},{"key":"1612_CR50","doi-asserted-by":"crossref","unstructured":"Chee Leong, M., Li Tan, H., Zhang, H., et al.: Joint Learning on the Hierarchy Representation for Fine-Grained Human Action Recognition (2021). arXiv:2110.05853","DOI":"10.1109\/ICIP42928.2021.9506157"},{"key":"1612_CR51","doi-asserted-by":"crossref","unstructured":"Kwon, H., Kim, M., Kwak, S., et al.: Learning self-similarity in space and time as generalized motion for video action recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 13065\u201313075 (2021)","DOI":"10.1109\/ICCV48922.2021.01282"},{"issue":"21","key":"1612_CR52","doi-asserted-by":"publisher","first-page":"4466","DOI":"10.3390\/electronics12214466","volume":"12","author":"C Liang","year":"2023","unstructured":"Liang, C., Yang, J., Du, R., et al.: Non-uniform motion aggregation with graph convolutional networks for skeleton-based human action recognition. Electronics 12(21), 4466 (2023)","journal-title":"Electronics"},{"issue":"4","key":"1612_CR53","doi-asserted-by":"publisher","first-page":"2058","DOI":"10.3390\/app13042058","volume":"13","author":"J Shi","year":"2023","unstructured":"Shi, J., Zhang, Y., Wang, W., et al.: A novel two-stream transformer-based framework for multi-modality human action recognition. Appl. Sci. 13(4), 2058 (2023)","journal-title":"Appl. Sci."},{"key":"1612_CR54","doi-asserted-by":"crossref","unstructured":"Hara, K., Kataoka, H., Satoh, Y.: Can spatiotemporal 3d cnns retrace the history of 2d cnns and imagenet? In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6546\u20136555 (2018)","DOI":"10.1109\/CVPR.2018.00685"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-024-01612-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-024-01612-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-024-01612-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,2,28]],"date-time":"2025-02-28T11:06:54Z","timestamp":1740740814000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-024-01612-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,26]]},"references-count":54,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2025,2]]}},"alternative-id":["1612"],"URL":"https:\/\/doi.org\/10.1007\/s00530-024-01612-5","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,12,26]]},"assertion":[{"value":"4 July 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 December 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 December 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"30"}}