{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T15:34:07Z","timestamp":1783438447258,"version":"3.54.6"},"reference-count":52,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2024,12,14]],"date-time":"2024-12-14T00:00:00Z","timestamp":1734134400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,12,14]],"date-time":"2024-12-14T00:00:00Z","timestamp":1734134400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"International Partnership Program of the Chinese Academy of Sciences","award":["104GJHZ2023053FN"],"award-info":[{"award-number":["104GJHZ2023053FN"]}]},{"name":"International Partnership Program of the Chinese Academy of Sciences","award":["104GJHZ2023053FN"],"award-info":[{"award-number":["104GJHZ2023053FN"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62103410"],"award-info":[{"award-number":["62103410"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62203438"],"award-info":[{"award-number":["62203438"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2025,2]]},"DOI":"10.1007\/s00530-024-01604-5","type":"journal-article","created":{"date-parts":[[2024,12,14]],"date-time":"2024-12-14T10:36:16Z","timestamp":1734172576000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["HierGAT: hierarchical spatial-temporal network with graph and transformer for video HOI detection"],"prefix":"10.1007","volume":"31","author":[{"given":"Junxian","family":"Wu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yujia","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Michael","family":"Kampffmeyer","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yi","family":"Pan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chenyu","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shiying","family":"Sun","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hui","family":"Chang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiaoguang","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,12,14]]},"reference":[{"issue":"10","key":"1604_CR1","doi-asserted-by":"publisher","first-page":"1775","DOI":"10.1109\/TPAMI.2009.83","volume":"31","author":"A Gupta","year":"2009","unstructured":"Gupta, A., Kembhavi, A., Davis, L.S.: Observing human\u2013object interactions: using spatial and functional compatibility for recognition. IEEE Trans. Pattern Anal. Mach. Intell. 31(10), 1775\u20131789 (2009)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"1604_CR2","doi-asserted-by":"crossref","unstructured":"Kim, B., Choi, T., Kang, J., Kim, H.J.: Uniondet: union-level detector towards real-time human\u2013object interaction detection. In: Proceedings of the European Conference on Computer Vision, pp. 498\u2013514 (2020). Springer","DOI":"10.1007\/978-3-030-58555-6_30"},{"key":"1604_CR3","doi-asserted-by":"crossref","unstructured":"Chen, M., Liao, Y., Liu, S., Chen, Z., Wang, F., Qian, C.: Reformulating hoi detection as adaptive set prediction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 9004\u20139013 (2021)","DOI":"10.1109\/CVPR46437.2021.00889"},{"issue":"3","key":"1604_CR4","doi-asserted-by":"publisher","first-page":"2831","DOI":"10.1007\/s10489-024-05324-1","volume":"54","author":"L Xia","year":"2024","unstructured":"Xia, L., Ding, X.: Human\u2013object interaction detection based on cascade multi-scale transformer. Appl. Intell. 54(3), 2831\u20132850 (2024)","journal-title":"Appl. Intell."},{"key":"1604_CR5","unstructured":"Li, L., Wei, J., Wang, W., Yang, Y.: Neural-logic human\u2013object interaction detection. In: Advances in Neural Information Processing Systems, vol. 36 (2024)"},{"issue":"3","key":"1604_CR6","doi-asserted-by":"publisher","first-page":"483","DOI":"10.1007\/s00530-020-00724-y","volume":"27","author":"Y Hu","year":"2021","unstructured":"Hu, Y., Lu, M., Xie, C., Lu, X.: Video-based driver action recognition via hybrid spatial-temporal deep learning framework. Multimed. Syst. 27(3), 483\u2013501 (2021)","journal-title":"Multimed. Syst."},{"key":"1604_CR7","doi-asserted-by":"crossref","unstructured":"Xing, H., Burschka, D.: Understanding spatio-temporal relations in human\u2013object interaction using pyramid graph convolutional network. In: IEEE\/RSJ International Conference on Intelligent Robots and Systems, pp. 5195\u20135201 (2022). IEEE","DOI":"10.1109\/IROS47612.2022.9981771"},{"issue":"10","key":"1604_CR8","doi-asserted-by":"publisher","first-page":"5814","DOI":"10.1109\/TCSVT.2023.3259430","volume":"33","author":"N Wang","year":"2023","unstructured":"Wang, N., Zhu, G., Li, H., Feng, M., Zhao, X., Ni, L., Shen, P., Mei, L., Zhang, L.: Exploring spatio-temporal graph convolution for video-based human\u2013object interaction recognition. IEEE Trans. Circuits Syst. Video Technol. 33(10), 5814\u20135827 (2023)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"1604_CR9","doi-asserted-by":"crossref","unstructured":"Tran, H., Le, V., Venkatesh, S., Tran, T.: Persistent-transient duality: a multi-mechanism approach for modeling human\u2013object interaction. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9858\u20139867 (2023)","DOI":"10.1109\/ICCV51070.2023.00904"},{"issue":"6","key":"1604_CR10","doi-asserted-by":"publisher","first-page":"2206","DOI":"10.1109\/TCSVT.2020.3019293","volume":"31","author":"A Banerjee","year":"2020","unstructured":"Banerjee, A., Singh, P.K., Sarkar, R.: Fuzzy integral-based cnn classifier fusion for 3d skeleton action recognition. IEEE Trans. Circuits Syst. Video Technol. 31(6), 2206\u20132216 (2020)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"issue":"5","key":"1604_CR11","doi-asserted-by":"publisher","first-page":"969","DOI":"10.1007\/s00530-021-00773-x","volume":"27","author":"NS Russel","year":"2021","unstructured":"Russel, N.S., Selvaraj, A.: Fusion of spatial and dynamic cnn streams for action recognition. Multimed. Syst. 27(5), 969\u2013984 (2021)","journal-title":"Multimed. Syst."},{"key":"1604_CR12","doi-asserted-by":"crossref","unstructured":"Nagarajan, T., Feichtenhofer, C., Grauman, K.: Grounded human\u2013object interaction hotspots from video. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8688\u20138697 (2019)","DOI":"10.1109\/ICCV.2019.00878"},{"key":"1604_CR13","doi-asserted-by":"crossref","unstructured":"Zeng, R., Huang, W., Tan, M., Rong, Y., Zhao, P., Huang, J., Gan, C.: Graph convolutional networks for temporal action localization. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 7094\u20137103 (2019)","DOI":"10.1109\/ICCV.2019.00719"},{"key":"1604_CR14","doi-asserted-by":"crossref","unstructured":"Sunkesula, S.P.R., Dabral, R., Ramakrishnan, G.: Lighten: learning interactions with graph and hierarchical temporal networks for hoi in videos. In: Proceedings of the 28th ACM International Conference on Multimedia, pp. 691\u2013699 (2020)","DOI":"10.1145\/3394171.3413778"},{"key":"1604_CR15","doi-asserted-by":"crossref","unstructured":"Wang, N., Zhu, G., Zhang, L., Shen, P., Li, H., Hua, C.: Spatio-temporal interaction graph parsing networks for human\u2013object interaction recognition. In: Proceedings of the 29th ACM International Conference on Multimedia, pp. 4985\u20134993 (2021)","DOI":"10.1145\/3474085.3475636"},{"key":"1604_CR16","doi-asserted-by":"crossref","unstructured":"Qiao, T., Men, Q., Li, F.W., Kubotani, Y., Morishima, S., Shum, H.P.: Geometric features informed multi-person human\u2013object interaction recognition in videos. In: Proceedings of the European Conference on Computer Vision, pp. 474\u2013491 (2022). Springer","DOI":"10.1007\/978-3-031-19772-7_28"},{"key":"1604_CR17","doi-asserted-by":"crossref","unstructured":"Morais, R., Le, V., Venkatesh, S., Tran, T.: Learning asynchronous and sparse human\u2013object interaction in videos. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16041\u201316050 (2021)","DOI":"10.1109\/CVPR46437.2021.01578"},{"key":"1604_CR18","first-page":"23345","volume":"35","author":"D Tu","year":"2022","unstructured":"Tu, D., Sun, W., Min, X., Zhai, G., Shen, W.: Video-based human\u2013object interaction detection from tubelet tokens. Adv. Neural. Inf. Process. Syst. 35, 23345\u201323357 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1604_CR19","unstructured":"Wang, Y., Li, K., Li, Y., He, Y., Huang, B., Zhao, Z., Zhang, H., Xu, J., Liu, Y., Wang, Z., et al.: Internvideo: general video foundation models via generative and discriminative learning. arXiv preprint arXiv:2212.03191 (2022)"},{"key":"1604_CR20","unstructured":"Gupta, S., Malik, J.: Visual semantic role labeling. arXiv preprint arXiv:1505.04474 (2015)"},{"key":"1604_CR21","doi-asserted-by":"crossref","unstructured":"Gkioxari, G., Girshick, R., Doll\u00e1r, P., He, K.: Detecting and recognizing human\u2013object interactions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 8359\u20138367 (2018)","DOI":"10.1109\/CVPR.2018.00872"},{"key":"1604_CR22","doi-asserted-by":"crossref","unstructured":"Mallya, A., Lazebnik, S.: Learning models for actions and person\u2013object interactions with transfer to question answering. In: Proceedings of the European Conference on Computer Vision, pp. 414\u2013428 (2016). Springer","DOI":"10.1007\/978-3-319-46448-0_25"},{"key":"1604_CR23","unstructured":"Gao, C., Zou, Y., Huang, J.-B.: ican: Instance-centric attention network for human\u2013object interaction detection. arXiv preprint arXiv:1808.10437 (2018)"},{"issue":"6","key":"1604_CR24","doi-asserted-by":"publisher","first-page":"2827","DOI":"10.1109\/TPAMI.2021.3049156","volume":"44","author":"T Zhou","year":"2021","unstructured":"Zhou, T., Qi, S., Wang, W., Shen, J., Zhu, S.-C.: Cascaded parsing of human\u2013object interaction recognition. IEEE Trans. Pattern Anal. Mach. Intell. 44(6), 2827\u20132840 (2021)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"1604_CR25","doi-asserted-by":"publisher","first-page":"978","DOI":"10.1016\/j.neucom.2022.05.014","volume":"500","author":"Y Cheng","year":"2022","unstructured":"Cheng, Y., Duan, H., Wang, C., Wang, Z.: Human\u2013object interaction detection with depth-augmented clues. Neurocomputing 500, 978\u2013988 (2022)","journal-title":"Neurocomputing"},{"key":"1604_CR26","doi-asserted-by":"crossref","unstructured":"Liao, Y., Liu, S., Wang, F., Chen, Y., Qian, C., Feng, J.: Ppdm: parallel point detection and matching for real-time human\u2013object interaction detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 482\u2013490 (2020)","DOI":"10.1109\/CVPR42600.2020.00056"},{"issue":"6","key":"1604_CR27","doi-asserted-by":"publisher","first-page":"3853","DOI":"10.1109\/TCSVT.2021.3119892","volume":"32","author":"D Yang","year":"2021","unstructured":"Yang, D., Zou, Y., Zhang, C., Cao, M., Chen, J.: Rr-net: relation reasoning for end-to-end human\u2013object interaction detection. IEEE Trans. Circuits Syst. Video Technol. 32(6), 3853\u20133865 (2021)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"1604_CR28","doi-asserted-by":"crossref","unstructured":"Ulutan, O., Iftekhar, A., Manjunath, B.S.: Vsgnet: spatial attention network for detecting human object interactions using graph convolutions. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13617\u201313626 (2020)","DOI":"10.1109\/CVPR42600.2020.01363"},{"key":"1604_CR29","doi-asserted-by":"crossref","unstructured":"Park, J., Park, J.-W., Lee, J.-S.: Viplo: vision transformer based pose-conditioned self-loop graph for human\u2013object interaction detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 17152\u201317162 (2023)","DOI":"10.1109\/CVPR52729.2023.01645"},{"key":"1604_CR30","unstructured":"Koppula, H.S., Gupta, R., Saxena, A.: Human activity learning using object affordances from rgb-d videos. arXiv preprint arXiv:1208.0967 (2012)"},{"key":"1604_CR31","doi-asserted-by":"crossref","unstructured":"Jain, A., Zamir, A.R., Savarese, S., Saxena, A.: Structural-rnn: deep learning on spatio-temporal graphs. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 5308\u20135317 (2016)","DOI":"10.1109\/CVPR.2016.573"},{"key":"1604_CR32","doi-asserted-by":"crossref","unstructured":"Qi, S., Wang, W., Jia, B., Shen, J., Zhu, S.-C.: Learning human\u2013object interactions by graph parsing neural networks. In: Proceedings of the European Conference on Computer Vision, pp. 401\u2013417 (2018)","DOI":"10.1007\/978-3-030-01240-3_25"},{"key":"1604_CR33","doi-asserted-by":"crossref","unstructured":"Kuehne, H., Gall, J., Serre, T.: An end-to-end generative framework for video segmentation and recognition. In: IEEE Winter Conference on Applications of Computer Vision, pp. 1\u20138 (2016). IEEE","DOI":"10.1109\/WACV.2016.7477701"},{"key":"1604_CR34","doi-asserted-by":"crossref","unstructured":"Pirsiavash, H., Ramanan, D.: Parsing videos of actions with segmental grammars. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 612\u2013619 (2014)","DOI":"10.1109\/CVPR.2014.85"},{"issue":"6","key":"1604_CR35","doi-asserted-by":"publisher","first-page":"6647","DOI":"10.1109\/TPAMI.2020.3021756","volume":"45","author":"S Li","year":"2020","unstructured":"Li, S., Farha, Y.A., Liu, Y., Cheng, M.-M., Gall, J.: Ms-tcn++: multi-stage temporal convolutional network for action segmentation. IEEE Trans. Pattern Anal. Mach. Intell. 45(6), 6647\u20136658 (2020)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"1604_CR36","doi-asserted-by":"crossref","unstructured":"Lea, C., Flynn, M.D., Vidal, R., Reiter, A., Hager, G.D.: Temporal convolutional networks for action segmentation and detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 156\u2013165 (2017)","DOI":"10.1109\/CVPR.2017.113"},{"key":"1604_CR37","doi-asserted-by":"crossref","unstructured":"Huang, Y., Sugano, Y., Sato, Y.: Improving action segmentation via graph-based temporal reasoning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14024\u201314034 (2020)","DOI":"10.1109\/CVPR42600.2020.01404"},{"key":"1604_CR38","doi-asserted-by":"crossref","unstructured":"Wang, Z., Gao, Z., Wang, L., Li, Z., Wu, G.: Boundary-aware cascade networks for temporal action segmentation. In: Proceedings of the European Conference on Computer Vision, pp. 34\u201351 (2020). Springer","DOI":"10.1007\/978-3-030-58595-2_3"},{"key":"1604_CR39","unstructured":"Yi, F., Wen, H., Jiang, T.: Asformer: transformer for action segmentation. In: British Machine Vision Conference (2021)"},{"key":"1604_CR40","doi-asserted-by":"crossref","unstructured":"Zhang, R., Wang, S., Duan, Y., Tang, Y., Zhang, Y., Tan, Y.-P.: Hoi-aware adaptive network for weakly-supervised action segmentation. In: Proceedings of the Thirty-Second International Joint Conference on Artificial Intelligence, pp. 1722\u20131730 (2023)","DOI":"10.24963\/ijcai.2023\/191"},{"key":"1604_CR41","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.neunet.2023.01.019","volume":"163","author":"Q Li","year":"2023","unstructured":"Li, Q., Xie, X., Zhang, J., Shi, G.: Few-shot human\u2013object interaction video recognition with transformers. Neural Netw. 163, 1\u20139 (2023)","journal-title":"Neural Netw."},{"key":"1604_CR42","doi-asserted-by":"crossref","unstructured":"Ji, J., Desai, R., Niebles, J.C.: Detecting human\u2013object relationships in videos. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 8106\u20138116 (2021)","DOI":"10.1109\/ICCV48922.2021.00800"},{"key":"1604_CR43","doi-asserted-by":"crossref","unstructured":"Cong, Y., Liao, W., Ackermann, H., Rosenhahn, B., Yang, M.Y.: Spatial-temporal transformer for dynamic scene graph generation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 16372\u201316382 (2021)","DOI":"10.1109\/ICCV48922.2021.01606"},{"key":"1604_CR44","doi-asserted-by":"publisher","DOI":"10.1016\/j.cviu.2023.103741","volume":"233","author":"Z Ni","year":"2023","unstructured":"Ni, Z., Mascar\u00f3, E.V., Ahn, H., Lee, D.: Human\u2013object interaction prediction in videos through gaze following. Comput. Vis. Image Underst. 233, 103741 (2023)","journal-title":"Comput. Vis. Image Underst."},{"key":"1604_CR45","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster r-cnn: towards real-time object detection with region proposal networks. In: Advances in Neural Information Processing Systems, vol. 28 (2015)"},{"key":"1604_CR46","doi-asserted-by":"publisher","first-page":"32","DOI":"10.1007\/s11263-016-0981-7","volume":"123","author":"R Krishna","year":"2017","unstructured":"Krishna, R., Zhu, Y., Groth, O., Johnson, J., Hata, K., Kravitz, J., Chen, S., Kalantidis, Y., Li, L.-J., Shamma, D.A., et al.: Visual genome: connecting language and vision using crowdsourced dense image annotations. Int. J. Comput. Vis. 123, 32\u201373 (2017)","journal-title":"Int. J. Comput. Vis."},{"key":"1604_CR47","doi-asserted-by":"crossref","unstructured":"Cho, K., van Merrienboer, B., G\u00fcl\u00e7ehre, \u00c7., Bahdanau, D., Bougares, F., Schwenk, H., Bengio, Y.: Learning phrase representations using rnn encoder-decoder for statistical machine translation. In: EMNLP (2014)","DOI":"10.3115\/v1\/D14-1179"},{"issue":"8","key":"1604_CR48","doi-asserted-by":"publisher","first-page":"951","DOI":"10.1177\/0278364913478446","volume":"32","author":"HS Koppula","year":"2013","unstructured":"Koppula, H.S., Gupta, R., Saxena, A.: Learning human activities and object affordances from rgb-d videos. Int. J. Robot. Res. 32(8), 951\u2013970 (2013)","journal-title":"Int. J. Robot. Res."},{"issue":"1","key":"1604_CR49","doi-asserted-by":"publisher","first-page":"187","DOI":"10.1109\/LRA.2019.2949221","volume":"5","author":"CR Dreher","year":"2019","unstructured":"Dreher, C.R., W\u00e4chter, M., Asfour, T.: Learning object-action relations from bimanual human demonstration using graph networks. IEEE Robot. Autom. Lett. 5(1), 187\u2013194 (2019)","journal-title":"IEEE Robot. Autom. Lett."},{"key":"1604_CR50","doi-asserted-by":"crossref","unstructured":"Qiao, T., Li, R., Li, F.W., Shum, H.P.: From category to scenery: an end-to-end framework for multi-person human\u2013object interaction recognition in videos. In: International Conference on Pattern Recognition (2024)","DOI":"10.1007\/978-3-031-78354-8_17"},{"key":"1604_CR51","doi-asserted-by":"crossref","unstructured":"Sener, O., Saxena, A.: rcrf: Recursive belief estimation over crfs in rgb-d activity videos. In: Robotics: Science and Systems (2015)","DOI":"10.15607\/RSS.2015.XI.024"},{"issue":"1","key":"1604_CR52","doi-asserted-by":"publisher","first-page":"14","DOI":"10.1109\/TPAMI.2015.2430335","volume":"38","author":"HS Koppula","year":"2015","unstructured":"Koppula, H.S., Saxena, A.: Anticipating human activities using object affordances for reactive robotic response. IEEE Trans. Pattern Anal. Mach. Intell. 38(1), 14\u201329 (2015)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-024-01604-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-024-01604-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-024-01604-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,2,28]],"date-time":"2025-02-28T11:02:21Z","timestamp":1740740541000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-024-01604-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,14]]},"references-count":52,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2025,2]]}},"alternative-id":["1604"],"URL":"https:\/\/doi.org\/10.1007\/s00530-024-01604-5","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,12,14]]},"assertion":[{"value":"30 July 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 November 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 December 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"This work was also supported by the Excellent Youth Program of State Key Laboratory of Multimodal Artificial Intelligence Systems. The authors have no conflict of interest to declare that are relevant to the content of this article.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Not applicable.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethics approval"}}],"article-number":"13"}}