{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T18:22:27Z","timestamp":1784658147513,"version":"3.55.0"},"reference-count":43,"publisher":"Springer Science and Business Media LLC","issue":"9","license":[{"start":{"date-parts":[[2023,8,22]],"date-time":"2023-08-22T00:00:00Z","timestamp":1692662400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,8,22]],"date-time":"2023-08-22T00:00:00Z","timestamp":1692662400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["No. 61972267"],"award-info":[{"award-number":["No. 61972267"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003787","name":"Natural Science Foundation of Hebei Province","doi-asserted-by":"publisher","award":["No. F2019210306"],"award-info":[{"award-number":["No. F2019210306"]}],"id":[{"id":"10.13039\/501100003787","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/natural","name":"Natural Science Foundation of Hebei Province","doi-asserted-by":"publisher","id":[{"id":"10.13039\/natural","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"DOI":"10.1007\/s11042-023-16041-1","type":"journal-article","created":{"date-parts":[[2023,8,22]],"date-time":"2023-08-22T05:01:55Z","timestamp":1692680515000},"page":"25643-25656","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":15,"title":["Cross-enhancement transformer for action segmentation"],"prefix":"10.1007","volume":"83","author":[{"given":"Jiahui","family":"Wang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9976-8583","authenticated-orcid":false,"given":"Zhengyou","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shanna","family":"Zhuang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yaqian","family":"Hao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hui","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,8,22]]},"reference":[{"key":"16041_CR1","doi-asserted-by":"crossref","unstructured":"Ahn H, Lee D (2021) Refining action segmentation with hierarchical video representations. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp 16302\u201316310","DOI":"10.1109\/ICCV48922.2021.01599"},{"key":"16041_CR2","doi-asserted-by":"crossref","unstructured":"Arnab A, Dehghani M, Heigold G, Sun C, Lu\u010di\u0107 M, Schmid C (2021) Vivit: a video vision transformer. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp 6836\u20136846","DOI":"10.1109\/ICCV48922.2021.00676"},{"key":"16041_CR3","doi-asserted-by":"crossref","unstructured":"Carreira J, Zisserman A (2017) Quo vadis, action recognition? A new model and the kinetics dataset. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp 6299\u20136308","DOI":"10.1109\/CVPR.2017.502"},{"key":"16041_CR4","doi-asserted-by":"crossref","unstructured":"Chen C-FR, Fan Q, Panda R (2021) Crossvit: cross-attention multi-scale vision transformer for image classification. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp 357\u2013366","DOI":"10.1109\/ICCV48922.2021.00041"},{"key":"16041_CR5","doi-asserted-by":"crossref","unstructured":"Chen M-H, Li B, Bao Y, AlRegib G, Kira Z (2020) Action segmentation with joint self-supervised temporal domain adaptation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp 9454\u20139463","DOI":"10.1109\/CVPR42600.2020.00947"},{"key":"16041_CR6","doi-asserted-by":"crossref","unstructured":"Chen W, Chai Y, Qi M, Sun H, Pu Q, Kong J, Zheng C (2022) Bottom-up improved multistage temporal convolutional network for action segmentation. Applied Intelligence 1\u201317","DOI":"10.1007\/s10489-022-03382-x"},{"issue":"8","key":"16041_CR7","doi-asserted-by":"publisher","first-page":"745","DOI":"10.1109\/TPAMI.2000.868676","volume":"22","author":"RT Collins","year":"2000","unstructured":"Collins RT, Lipton AJ, Kanade T (2000) Introduction to the special section on video surveillance. IEEE Trans Pattern Anal Mach Intell 22(8):745\u2013746","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"issue":"2","key":"16041_CR8","doi-asserted-by":"publisher","first-page":"690","DOI":"10.1007\/s10489-020-01823-z","volume":"51","author":"O Elharrouss","year":"2021","unstructured":"Elharrouss O, Almaadeed N, Al-Maadeed S, Bouridane A, Beghdadi A (2021) A combined multiple action recognition and summarization for surveillance video sequences. Appl Intell 51(2):690\u2013712","journal-title":"Appl Intell"},{"key":"16041_CR9","doi-asserted-by":"crossref","unstructured":"Farha YA, Gall J (2019) MS-TCN: Multi-stage temporal convolutional network for action segmentation. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp 3575\u20133584","DOI":"10.1109\/CVPR.2019.00369"},{"key":"16041_CR10","doi-asserted-by":"crossref","unstructured":"Fathi A, Ren X, Rehg JM (2011) Learning to recognize objects in egocentric activities. In: CVPR 2011. IEEE, pp 3281\u20133288","DOI":"10.1109\/CVPR.2011.5995444"},{"key":"16041_CR11","doi-asserted-by":"crossref","unstructured":"Fayyaz M, Gall J (2020) Sct: Set constrained temporal transformer for set supervised action segmentation. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp 501\u2013510","DOI":"10.1109\/CVPR42600.2020.00058"},{"key":"16041_CR12","doi-asserted-by":"crossref","unstructured":"Feichtenhofer C, Fan H, itendra Malik J, He K (2019) Slowfast networks for video recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp 6202\u20136211","DOI":"10.1109\/ICCV.2019.00630"},{"key":"16041_CR13","doi-asserted-by":"crossref","unstructured":"Ishikawa Y, Kasai S, Aoki Y, Kataoka H (2021) Alleviating over-segmentation errors by detecting action boundaries. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision. pp 2322\u20132331","DOI":"10.1109\/WACV48630.2021.00237"},{"issue":"10","key":"16041_CR14","doi-asserted-by":"publisher","first-page":"7043","DOI":"10.1007\/s10489-021-02195-8","volume":"51","author":"G Jiang","year":"2021","unstructured":"Jiang G, Jiang X, Fang Z, Chen S (2021) An efficient attention module for 3D convolutional neural networks in action recognition. Appl Intell 51(10):7043\u20137057","journal-title":"Appl Intell"},{"key":"16041_CR15","unstructured":"Karaman S, Seidenari L, Del Bimbo A (2014) Fast saliency based pooling of fisher encoded dense trajectories. In: ECCV THUMOS Workshop, vol 1. p 5"},{"key":"16041_CR16","doi-asserted-by":"crossref","unstructured":"Karpathy A, Toderici G, Shetty S, Leung T, Sukthankar R, Fei-Fei L (2014) Large-scale video classification with convolutional neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp 1725\u20131732","DOI":"10.1109\/CVPR.2014.223"},{"key":"16041_CR17","doi-asserted-by":"crossref","unstructured":"Kuehne H, Arslan A, Serre T (2014) The language of actions: recovering the syntax and semantics of goal-directed human activities. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp 780\u2013787","DOI":"10.1109\/CVPR.2014.105"},{"key":"16041_CR18","doi-asserted-by":"crossref","unstructured":"Kuehne H, Gall J, Serre T (2016) An end-to-end generative framework for video segmentation and recognition. In: 2016 IEEE Winter Conference on Applications of Computer Vision (WACV). IEEE, pp 1\u20138","DOI":"10.1109\/WACV.2016.7477701"},{"key":"16041_CR19","doi-asserted-by":"crossref","unstructured":"Lea C, Flynn MD, Vidal R, Reiter A, Hager GD (2017) Temporal convolutional networks for action segmentation and detection. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp 156\u2013165","DOI":"10.1109\/CVPR.2017.113"},{"key":"16041_CR20","doi-asserted-by":"crossref","unstructured":"Lei P, Todorovic S (2018) Temporal deformable residual networks for action segmentation in videos. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp 6742\u20136751","DOI":"10.1109\/CVPR.2018.00705"},{"key":"16041_CR21","doi-asserted-by":"crossref","unstructured":"Li S-J, AbuFarha Y, Liu Y, Cheng M-M, Gall J (2020) MS-TCN++: multi-stage temporal convolutional network for action segmentation. IEEE Trans Pattern Anal Mach Intell","DOI":"10.1109\/CVPR.2019.00369"},{"key":"16041_CR22","doi-asserted-by":"crossref","unstructured":"Li X, Hou Y, Wang P, Gao Z, Xu M, Li W (2021) Trear: transformer-based RGB-D egocentric action recognition. IEEE Trans Cogn Dev Syst","DOI":"10.1109\/TCDS.2020.3048883"},{"key":"16041_CR23","doi-asserted-by":"publisher","first-page":"373","DOI":"10.1016\/j.neucom.2021.04.121","volume":"454","author":"Y Li","year":"2021","unstructured":"Li Y, Dong Z, Liu K, Feng L, Hu L, Zhu J, Xu L, Liu S et al (2021) Efficient two-step networks for temporal action segmentation. Neurocomputing 454:373\u2013381","journal-title":"Neurocomputing"},{"key":"16041_CR24","doi-asserted-by":"crossref","unstructured":"Liu Z, Lin Y, Cao Y, Hu H, Wei Y, Zhang Z, Lin S, Guo B (2021) Swin transformer: hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp 10012\u201310022","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"16041_CR25","doi-asserted-by":"crossref","unstructured":"Ma M, Xia H, Tan Y, Li H, Song S (2022) HT-NET: hierarchical context-attention transformer network for medical CT image segmentation. Applied Intelligence. pp 1\u201314","DOI":"10.1007\/s10489-021-03010-0"},{"key":"16041_CR26","doi-asserted-by":"crossref","unstructured":"Rohrbach M, Amin S, Andriluka M, Schiele B (2012) A database for fine grained activity detection of cooking activities. In: 2012 IEEE Conference on Computer Vision and Pattern Recognition. IEEE, pp 1194\u20131201","DOI":"10.1109\/CVPR.2012.6247801"},{"key":"16041_CR27","unstructured":"Singhania D, Rahaman R, Yao A (2021) Coarse to fine multi-resolution temporal convolutional network. Preprint at http:\/\/arxiv.org\/abs\/2105.10859"},{"key":"16041_CR28","doi-asserted-by":"crossref","unstructured":"Stein S, McKenna SJ (2013) Combining embedded accelerometers with computer vision for recognizing food preparation activities. In: Proceedings of the 2013 ACM International Joint Conference on Pervasive and Ubiquitous Computing. pp 729\u2013738","DOI":"10.1145\/2493432.2493482"},{"key":"16041_CR29","doi-asserted-by":"crossref","unstructured":"Strudel R, Garcia R, Laptev I, Schmid C (2021) Segmenter: transformer for semantic segmentation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp 7262\u20137272","DOI":"10.1109\/ICCV48922.2021.00717"},{"key":"16041_CR30","doi-asserted-by":"crossref","unstructured":"Sun Y, Cheng C, Zhang Y, Zhang C, Zheng L, Wang Z, Wei Y (2020) Circle loss: A unified perspective of pair similarity optimization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp 6398\u20136407","DOI":"10.1109\/CVPR42600.2020.00643"},{"key":"16041_CR31","doi-asserted-by":"crossref","unstructured":"Tran D, Bourdev L, Fergus R, Torresani L, Paluri M (2015) Learning spatiotemporal features with 3D convolutional networks. In: Proceedings of the IEEE international conference on computer vision. pp 4489\u20134497","DOI":"10.1109\/ICCV.2015.510"},{"key":"16041_CR32","unstructured":"Touvron H, Cord M, Douze M, Massa F, Sablayrolles A, J\u00e9gou H (2021) Training data-efficient image transformers & distillation through attention. In: International Conference on Machine Learning. PMLR. pp 10347\u201310357"},{"key":"16041_CR33","first-page":"2","volume":"125","author":"A Van Den Oord","year":"2016","unstructured":"Van Den Oord A, Dieleman S, Zen H, Simonyan K, Vinyals O, Graves A, Kalchbrenner N, Senior AW, Kavukcuoglu K (2016) Wavenet: a generative model for raw audio. SSW 125:2","journal-title":"SSW"},{"key":"16041_CR34","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser \u0141, Polosukhin I (2017) Attention is all you need. Adv Neural Inf Proces Syst 30"},{"key":"16041_CR35","doi-asserted-by":"crossref","unstructured":"Vo NN, Bobick AF (2014) From stochastic grammar to bayes network: probabilistic parsing of complex activity. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp 2641\u20132648","DOI":"10.1109\/CVPR.2014.338"},{"key":"16041_CR36","doi-asserted-by":"crossref","unstructured":"Wang L, Li W, Li W, Van\u00a0Gool L (2018) Appearance-and-relation networks for video classification. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp 1430\u20131439","DOI":"10.1109\/CVPR.2018.00155"},{"key":"16041_CR37","doi-asserted-by":"crossref","unstructured":"Wang W, Xie E, Li X, Fan D-P, Song K, Liang D, Lu T, Luo P, Shao L (2021) Pyramid vision transformer: a versatile backbone for dense prediction without convolutions. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp 568\u2013578","DOI":"10.1109\/ICCV48922.2021.00061"},{"key":"16041_CR38","doi-asserted-by":"crossref","unstructured":"Wang Z, Gao Z, Wang L, Li Z, Wu G (2020) Boundary-aware cascade networks for temporal action segmentation. In: European Conference on Computer Vision. Springer, pp 34\u201351","DOI":"10.1007\/978-3-030-58595-2_3"},{"key":"16041_CR39","unstructured":"Yang D, Cao Z, Mao L, Zhang R (2022) A temporal and channel-combined attention block for action segmentation. Applied Intelligence 1\u201313"},{"key":"16041_CR40","doi-asserted-by":"crossref","unstructured":"Yang J, Ge H, Su S, Liu G (2022) Transformer-based two-source motion model for multi-object tracking. Appl Intell 1\u201313","DOI":"10.1007\/s10489-021-03012-y"},{"key":"16041_CR41","unstructured":"Yi F, Wen H, Jiang T (2021) Asformer: transformer for action segmentation. Preprint at http:\/\/arxiv.org\/abs\/2110.08568"},{"key":"16041_CR42","doi-asserted-by":"crossref","unstructured":"Yue-Hei\u00a0Ng J, Hausknecht M, Vijayanarasimhan S, Vinyals O, Monga R, Toderici G (2015) Beyond short snippets: Deep networks for video classification. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. pp 4694\u20134702","DOI":"10.1109\/CVPR.2015.7299101"},{"key":"16041_CR43","doi-asserted-by":"crossref","unstructured":"Zheng S, Lu J, Zhao H, Zhu X, Luo Z, Wang Y, Fu Y, Feng J, Xiang T, Torr PHS et\u00a0al (2021) Rethinking semantic segmentation from a sequence-to-sequence perspective with transformers. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp 6881\u20136890","DOI":"10.1109\/CVPR46437.2021.00681"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-16041-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-023-16041-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-023-16041-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,2,29]],"date-time":"2024-02-29T10:26:48Z","timestamp":1709202408000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-023-16041-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,8,22]]},"references-count":43,"journal-issue":{"issue":"9","published-online":{"date-parts":[[2024,3]]}},"alternative-id":["16041"],"URL":"https:\/\/doi.org\/10.1007\/s11042-023-16041-1","relation":{},"ISSN":["1573-7721"],"issn-type":[{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,8,22]]},"assertion":[{"value":"15 September 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"20 April 2023","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 June 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 August 2023","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}