{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,5]],"date-time":"2026-06-05T13:03:45Z","timestamp":1780664625027,"version":"3.54.1"},"reference-count":58,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2026,3,21]],"date-time":"2026-03-21T00:00:00Z","timestamp":1774051200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,3,21]],"date-time":"2026-03-21T00:00:00Z","timestamp":1774051200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["61872227"],"award-info":[{"award-number":["61872227"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["12126420"],"award-info":[{"award-number":["12126420"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int. J. Mach. Learn. &amp; Cyber."],"published-print":{"date-parts":[[2026,5]]},"DOI":"10.1007\/s13042-026-03039-y","type":"journal-article","created":{"date-parts":[[2026,3,21]],"date-time":"2026-03-21T05:21:54Z","timestamp":1774070514000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["GAttenRNN: a recurrent neural network for spati-temporal prediction learning based on gated transformer"],"prefix":"10.1007","volume":"17","author":[{"given":"Gaihui","family":"Guo","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiaxin","family":"Wan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Weichuan","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,3,21]]},"reference":[{"key":"3039_CR1","first-page":"1","volume":"28","author":"X Shi","year":"2015","unstructured":"Shi X, Chen Z, Wang H et al (2015) Convolutional LSTM network: a machine learning approach for precipitation nowcasting. Adv Neural Inf Process Syst 28:1","journal-title":"Adv Neural Inf Process Syst"},{"key":"3039_CR2","unstructured":"Shi X, Gao Z, Lausen L et al (2017) Deep learning for precipitation nowcasting: a benchmark and a new model. Adv Neural Inf Process Syst 30"},{"key":"3039_CR3","unstructured":"Tang Y, Zhou J, Pan X et al (2023) PostRainBench: a comprehensive benchmark and a new model for precipitation forecasting. arXiv:2310.02676"},{"key":"3039_CR4","unstructured":"Wang Y, Long M, Wang J et al (2017) PredRNN: recurrent neural networks for predictive learning using spatiotemporal LSTMS. Adv Neural Inf Process Syst 30"},{"key":"3039_CR5","doi-asserted-by":"crossref","unstructured":"Bhattacharyya A, Fritz M, Schiele B (2018) Long-term on-board prediction of people in traffic scenes under uncertainty. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 4194\u20134202","DOI":"10.1109\/CVPR.2018.00441"},{"key":"3039_CR6","doi-asserted-by":"crossref","unstructured":"Kwon Y-H, Park M-G (2019) Predicting future frames using retrospective cycle GAN. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 1811\u20131820","DOI":"10.1109\/CVPR.2019.00191"},{"key":"3039_CR7","doi-asserted-by":"crossref","unstructured":"Li R, Li S J, Chen XY et al (2024) TFNet: exploiting temporal cues for fast and accurate lidar semantic segmentation. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 4547\u20134556","DOI":"10.1109\/CVPRW63382.2024.00457"},{"key":"3039_CR8","doi-asserted-by":"crossref","unstructured":"Xu ZR, Wang YB, Long MS et al (2018) PredCNN: predictive learning with cascade convolutions. In: International joint conference on artificial intelligence, pp 2940\u20132947","DOI":"10.24963\/ijcai.2018\/408"},{"key":"3039_CR9","doi-asserted-by":"crossref","unstructured":"Zhang JB, Zheng Y, Qi DK (2017) Deep spatio-temporal residual networks for citywide crowd flows prediction. Proc AAAI Conf Artif Intell 31(1)","DOI":"10.1609\/aaai.v31i1.10735"},{"key":"3039_CR10","doi-asserted-by":"publisher","first-page":"118","DOI":"10.1016\/j.cviu.2018.04.007","volume":"171","author":"P Wang","year":"2018","unstructured":"Wang P, Li W, Ogunbona P et al (2018) RGB-D-based human motion recognition with deep learning: a survey. Comput Vis Image Underst 171:118\u2013139","journal-title":"Comput Vis Image Underst"},{"key":"3039_CR11","doi-asserted-by":"crossref","unstructured":"Zhang L, Zhu GM, Shen PY et al (2017) Learning spatiotemporal features using 3DCNN and convolutional LSTM for gesture recognition. In: Proceedings of the IEEE international conference on computer vision workshops, pp 3120\u20133128","DOI":"10.1109\/ICCVW.2017.369"},{"key":"3039_CR12","doi-asserted-by":"crossref","unstructured":"Jenni S, Meishvili G, Favaro P (2020) Video representation learning by recognizing temporal transformations. In: European conference on computer vision. Springer, pp 425\u2013442","DOI":"10.1007\/978-3-030-58604-1_26"},{"key":"3039_CR13","doi-asserted-by":"crossref","unstructured":"Qian R, Meng TJ, Gong BQ et al (2021) Spatiotemporal contrastive video representation learning. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 6964\u20136974","DOI":"10.1109\/CVPR46437.2021.00689"},{"key":"3039_CR14","unstructured":"Wu J, Lu E, Kohli P et al (2017) Learning to see physics via visual de-animation. Adv Neural Inf Process Syst 30"},{"key":"3039_CR15","unstructured":"Van Steenkiste S, Chang M, Greff K et al (2018) Relational neural expectation maximization: unsupervised discovery of objects and their interactions. arXiv:1802.10353"},{"key":"3039_CR16","unstructured":"Kipf T, Fetaya E, Wang KC et al (2018) Neural relational inference for interacting systems. In: International conference on machine learning, pp 2688\u20132697"},{"key":"3039_CR17","unstructured":"Xu ZJ, Liu ZJ, Sun C et al (2019) Unsupervised discovery of parts, structure, and dynamics. arXiv:1903.05136"},{"key":"3039_CR18","doi-asserted-by":"crossref","unstructured":"Wang YB, Zhang JJ, Zhu HY et al (2019) Memory in memory: a predictive neural network for learning higher-order non-stationarity from spatiotemporal dynamics. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 9154\u20139162","DOI":"10.1109\/CVPR.2019.00937"},{"key":"3039_CR19","unstructured":"Ha D, Schmidhuber J (2018) Recurrent world models facilitate policy evolution. Adv Neural Inf Process Syst 31"},{"key":"3039_CR20","unstructured":"Hafner D, Lillicrap T, Fischer I et al (2019) Learning latent dynamics for planning from pixels. In: International conference on machine learning, pp 2555\u20132565"},{"key":"3039_CR21","doi-asserted-by":"crossref","unstructured":"Finn C, Levine S (2017) Deep visual foresight for planning robot motion. In: International conference on robotics and automation. IEEE, pp 2786\u20132793","DOI":"10.1109\/ICRA.2017.7989324"},{"issue":"16","key":"3039_CR22","first-page":"23","volume":"12","author":"F Ebert","year":"2017","unstructured":"Ebert F, Finn C, Lee AX et al (2017) Self-supervised visual planning with temporal skip connections. Conf Robot Learn 12(16):23","journal-title":"Conf Robot Learn"},{"key":"3039_CR23","unstructured":"Wang YB, Jiang L, Yang MH et al (2018) Eidetic 3D LSTM: a model for video prediction and beyond. In: International conference on learning representations"},{"key":"3039_CR24","first-page":"26950","volume":"34","author":"Z Chang","year":"2021","unstructured":"Chang Z, Zhang XF, Wang SS et al (2021) MAU: a motion-aware unit for video prediction and beyond. Adv Neural Inf Process Syst 34:26950\u201326962","journal-title":"Adv Neural Inf Process Syst"},{"key":"3039_CR25","doi-asserted-by":"crossref","unstructured":"Tang YJ, Dong PJ, Tang ZH et al (2024) VmRNN: integrating vision mamba and LSTM for efficient and accurate spatiotemporal forecasting. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 5663\u20135673","DOI":"10.1109\/CVPRW63382.2024.00575"},{"key":"3039_CR26","doi-asserted-by":"crossref","unstructured":"Gao ZY, Tan C, Wu LR et al (2022) SimVP: simpler yet better video prediction. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 3170\u20133180","DOI":"10.1109\/CVPR52688.2022.00317"},{"key":"3039_CR27","unstructured":"Tan C, Gao ZY, Li SY et al (2022) SimVP: towards simple yet powerful spatiotemporal predictive learning. arXiv:2211.12509"},{"key":"3039_CR28","unstructured":"Dosovitskiy A, Beyer L, Kolesnikov A et al (2020) An image is worth $$16\\times 16$$ words: transformers for image recognition at scale. arXiv:2010.11929"},{"key":"3039_CR29","first-page":"103031","volume":"37","author":"Y Liu","year":"2024","unstructured":"Liu Y, Tian YJ, Zhao YZ et al (2024) VMamba: visual state space model. Adv Neural Inf Process Syst 37:103031\u2013103063","journal-title":"Adv Neural Inf Process Syst"},{"key":"3039_CR30","unstructured":"Srivastava N, Mansimov E, Salakhudinov R (2015) Unsupervised learning of video representations using LSTMs. In: International conference on machine learning, pp 843\u2013852"},{"key":"3039_CR31","unstructured":"Wang YB, Gao ZF, Long MS et al (2018) PredRNN++: towards a resolution of the deep-in-time dilemma in spatiotemporal predictive learning. In: International conference on machine learning, pp 5123\u20135132"},{"key":"3039_CR32","unstructured":"Yu W, Lu YC, Easterbrook S et al (2020) Efficient and information-preserving future frame prediction and beyond. In: International conference on learning representations"},{"issue":"2","key":"3039_CR33","doi-asserted-by":"publisher","first-page":"2208","DOI":"10.1109\/TPAMI.2022.3165153","volume":"45","author":"YB Wang","year":"2022","unstructured":"Wang YB, Wu HX, Zhang JJ et al (2022) PredRNN: a recurrent neural network for spatiotemporal predictive learning. IEEE Trans Pattern Anal Mach Intell 45(2):2208\u20132225","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"3039_CR34","doi-asserted-by":"crossref","unstructured":"Le Guen V, Thome N (2020) Disentangling physical dynamics from unknown factors for unsupervised video prediction. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 11474\u201311484","DOI":"10.1109\/CVPR42600.2020.01149"},{"key":"3039_CR35","doi-asserted-by":"crossref","unstructured":"Tang S, Li C, Zhang P et al (2023) SwinLSTM: improving spatiotemporal prediction accuracy using swin transformer and LSTM. In: Proceedings of the IEEE international conference on computer vision, pp 13470\u201313479","DOI":"10.1109\/ICCV51070.2023.01239"},{"key":"3039_CR36","doi-asserted-by":"crossref","unstructured":"Liu Z, Lin YT, Cao Y et al (2021) Swin transformer: hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE international conference on computer vision, pp 10012\u201310022","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"3039_CR37","doi-asserted-by":"crossref","unstructured":"Tan C, Gao ZY, Wu LR et al (2023) Temporal attention unit: towards efficient spatiotemporal predictive learning. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 18770\u201318782","DOI":"10.1109\/CVPR52729.2023.01800"},{"key":"3039_CR38","doi-asserted-by":"crossref","unstructured":"Liu ZW, Yeh RA, Tang XO et al (2017) Video frame synthesis using deep voxel flow. In: Proceedings of the IEEE international conference on computer vision, pp 4463\u20134471","DOI":"10.1109\/ICCV.2017.478"},{"key":"3039_CR39","first-page":"69819","volume":"36","author":"C Tan","year":"2023","unstructured":"Tan C, Li SY, Gao ZY et al (2023) OpenSTL: a comprehensive benchmark of spatio-temporal predictive learning. Adv Neural Inf Process Syst 36:69819\u201369831","journal-title":"Adv Neural Inf Process Syst"},{"key":"3039_CR40","doi-asserted-by":"crossref","unstructured":"Hu XT, Huang ZW, Huang AL et al (2023) A dynamic multi-scale voxel flow network for video prediction. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 6121\u20136131","DOI":"10.1109\/CVPR52729.2023.00593"},{"key":"3039_CR41","doi-asserted-by":"crossref","unstructured":"Gao H, Xu HZ, Cai QZ et al (2019) Disentangling propagation and generation for video prediction. In: Proceedings of the IEEE international conference on computer vision, pp 9006\u20139015","DOI":"10.1109\/ICCV.2019.00910"},{"key":"3039_CR42","unstructured":"Shouno O (2020) Photo-realistic video prediction on natural videos of largely changing frames. arXiv:2003.08635"},{"key":"3039_CR43","doi-asserted-by":"publisher","first-page":"107871","DOI":"10.1016\/j.neunet.2025.107871","volume":"191","author":"S Xiang","year":"2025","unstructured":"Xiang S, Chongqing C, Dezhi H et al (2025) A triple-branch hybrid dynamic-static alignment strategy for vision-language tasks. Neural Netw 191:107871. https:\/\/doi.org\/10.1016\/j.neunet.2025.107871","journal-title":"Neural Netw"},{"key":"3039_CR44","doi-asserted-by":"publisher","first-page":"128857","DOI":"10.1016\/j.eswa.2025.128857","volume":"295","author":"S Xiang","year":"2026","unstructured":"Xiang S, Dezhi H, Chin-Chen C et al (2026) Multimodal context-aware consistency alignment for vision-language tasks. Expert Syst Appl 295:128857. https:\/\/doi.org\/10.1016\/j.eswa.2025.128857","journal-title":"Expert Syst Appl"},{"key":"3039_CR45","unstructured":"Shazeer N (2020) GLU variants improve transformer. arXiv:2002.05202"},{"key":"3039_CR46","unstructured":"Hendrycks D, Gimpel K (2016) Gaussian error linear units (GELUs). arXiv:1606.08415"},{"key":"3039_CR47","unstructured":"Finn C, Goodfellow I, Levine S (2016) Unsupervised learning for physical interaction through video prediction. Adv Neural Inf Process Syst 29"},{"key":"3039_CR48","unstructured":"Denton E, Fergus R (2018) Stochastic video generation with a learned prior. In: International conference on machine learning, pp 1174\u20131183"},{"key":"3039_CR49","unstructured":"Bengio S, Vinyals O, Jaitly N et al (2015) Scheduled sampling for sequence prediction with recurrent neural networks. Adv Neural Inf Process Syst 28"},{"issue":"4","key":"3039_CR50","doi-asserted-by":"publisher","first-page":"600","DOI":"10.1109\/TIP.2003.819861","volume":"13","author":"Z Wang","year":"2004","unstructured":"Wang Z, Bovik AC, Sheikh HR et al (2004) Image quality assessment: from error visibility to structural similarity. IEEE Trans Image Process 13(4):600\u2013612","journal-title":"IEEE Trans Image Process"},{"key":"3039_CR51","doi-asserted-by":"crossref","unstructured":"Schuldt C, Laptev I, Caputo B (2004) Recognizing human actions: a local SVM approach. In: Proceedings of the 17th international conference on pattern recognition, pp 32\u201336","DOI":"10.1109\/ICPR.2004.1334462"},{"key":"3039_CR52","doi-asserted-by":"publisher","unstructured":"Villegas R, Yang J, Hong S et al (2017) Decomposing motion and content for natural video sequence prediction. https:\/\/doi.org\/10.48550\/arXiv.1706.08033","DOI":"10.48550\/arXiv.1706.08033"},{"key":"3039_CR53","unstructured":"Yujin T, Lu Q, Fei X et al (2024) Video prediction transformers without recurrence or convolution. arXiv:2410.04733 [1 Oct 2025]"},{"key":"3039_CR54","doi-asserted-by":"crossref","unstructured":"Gao H, Xu HZ Cai QZ et al (2019) Disentangling propagation and generation for video prediction. In: Proceedings of the IEEE international conference on computer vision, pp 9006\u20139015","DOI":"10.1109\/ICCV.2019.00910"},{"key":"3039_CR55","doi-asserted-by":"crossref","unstructured":"Oliu M, Selva J, Escalera S (2018) Folded recurrent neural networks for future video prediction. In: Proceedings of the European conference on computer vision, pp 716\u2013731","DOI":"10.1007\/978-3-030-01264-9_44"},{"key":"3039_CR56","unstructured":"Jia X, De Brabandere B, Tuytelaars T et al (2016) Dynamic filter networks. Adv Neural Inf Process Syst 29"},{"key":"3039_CR57","doi-asserted-by":"crossref","unstructured":"Jin BB, Hu Y, Zeng YM et al (2018) VarNet: exploring variations for unsupervised video prediction. In: IEEE international conference on intelligent robots and systems, pp 5801\u20135806","DOI":"10.1109\/IROS.2018.8594264"},{"key":"3039_CR58","unstructured":"Lee AX, Zhang R, Ebert F et al (2018) Stochastic adversarial video prediction. arXiv:1804.01523"}],"container-title":["International Journal of Machine Learning and Cybernetics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13042-026-03039-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s13042-026-03039-y","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s13042-026-03039-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,5]],"date-time":"2026-06-05T12:09:22Z","timestamp":1780661362000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s13042-026-03039-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,21]]},"references-count":58,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2026,5]]}},"alternative-id":["3039"],"URL":"https:\/\/doi.org\/10.1007\/s13042-026-03039-y","relation":{},"ISSN":["1868-8071","1868-808X"],"issn-type":[{"value":"1868-8071","type":"print"},{"value":"1868-808X","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,3,21]]},"assertion":[{"value":"1 September 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"10 February 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 March 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"207"}}