{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T17:30:06Z","timestamp":1783013406801,"version":"3.54.6"},"reference-count":96,"publisher":"Springer Science and Business Media LLC","issue":"21","license":[{"start":{"date-parts":[[2024,4,23]],"date-time":"2024-04-23T00:00:00Z","timestamp":1713830400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,4,23]],"date-time":"2024-04-23T00:00:00Z","timestamp":1713830400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62072334"],"award-info":[{"award-number":["62072334"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Neural Comput &amp; Applic"],"published-print":{"date-parts":[[2024,7]]},"DOI":"10.1007\/s00521-024-09763-2","type":"journal-article","created":{"date-parts":[[2024,4,23]],"date-time":"2024-04-23T10:02:05Z","timestamp":1713866525000},"page":"12951-12976","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":5,"title":["Sign language translation with hierarchical memorized context in question answering scenarios"],"prefix":"10.1007","volume":"36","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-4518-2154","authenticated-orcid":false,"given":"Liqing","family":"Gao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wei","family":"Feng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Peng","family":"Shi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ruize","family":"Han","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Di","family":"Lin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Liang","family":"Wan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,4,23]]},"reference":[{"key":"9763_CR1","doi-asserted-by":"crossref","unstructured":"Cheng KL, Yang Z, Chen Q, Tai Y-W (2020) Fully convolutional networks for continuous sign language recognition. arXiv preprint arXiv:2007.12402","DOI":"10.1007\/978-3-030-58586-0_41"},{"key":"9763_CR2","doi-asserted-by":"crossref","unstructured":"Guo D, Tang S, Wang M (2019) Connectionist temporal modeling of video and language: a joint model for translation and sign labeling. In: IJCAI","DOI":"10.24963\/ijcai.2019\/106"},{"key":"9763_CR3","doi-asserted-by":"crossref","unstructured":"Shi L, Zhang Y, Cheng J, Lu H (2018) Two-stream adaptive graph convolutional networks for skeleton-based action recognition. In: CVPR","DOI":"10.1109\/CVPR.2019.01230"},{"key":"9763_CR4","doi-asserted-by":"crossref","unstructured":"Guo D, Wang S, Tian Q, Wang M (2019) Dense temporal convolution network for sign language translation. In: IJCAI","DOI":"10.24963\/ijcai.2019\/105"},{"key":"9763_CR5","unstructured":"Hu H, Zhou W, Pu J, Li H (2020) Global-local enhancement network for nmfs-aware sign language recognition. arXiv preprint arXiv:2008.10428"},{"key":"9763_CR6","unstructured":"Camgoz NC, Koller O, Hadfield S, Bowden R (2020) Sign language transformers: Joint end-to-end sign language recognition and translation. In: CVPR"},{"key":"9763_CR7","doi-asserted-by":"crossref","unstructured":"Li D, Yu X, Xu C, Petersson L, Li H (2020) Transferring cross-domain knowledge for video sign language recognition. In: CVPR","DOI":"10.1109\/CVPR42600.2020.00624"},{"issue":"9","key":"9763_CR8","doi-asserted-by":"publisher","first-page":"2306","DOI":"10.1109\/TPAMI.2019.2911077","volume":"42","author":"O Koller","year":"2020","unstructured":"Koller O, Camgoz C, Ney H, Bowden R (2020) Weakly supervised learning with multi-stream CNN-LSTM-HMMs to discover sequential parallelism in sign language videos. IEEE TPAMI 42(9):2306\u20132320","journal-title":"IEEE TPAMI"},{"key":"9763_CR9","doi-asserted-by":"crossref","unstructured":"Yin K, Read J (2020) Better sign language translation with STMC-transformer. In: COLING","DOI":"10.18653\/v1\/2020.coling-main.525"},{"key":"9763_CR10","doi-asserted-by":"crossref","unstructured":"Guo D, Zhou W, Li H, Wang M (2018) Hierarchical LSTM for sign language translation. In: AAAI","DOI":"10.1609\/aaai.v32i1.12235"},{"key":"9763_CR11","first-page":"108","volume":"141","author":"O Koller","year":"2015","unstructured":"Koller O, Forster J, Ney H (2015) Continuous sign language recognition: towards large vocabulary statistical recognition systems handling multiple signers. CVIU 141:108\u2013125","journal-title":"CVIU"},{"key":"9763_CR12","doi-asserted-by":"crossref","unstructured":"Zhou H, Zhou W, Qi W, Pu J, Li H (2021) Improving sign language translation with monolingual data by sign back-translation. In: CVPR, pp 1316\u20131325","DOI":"10.1109\/CVPR46437.2021.00137"},{"key":"9763_CR13","doi-asserted-by":"crossref","unstructured":"Camgoz NC, Hadfield S, Koller O, Ney H, Bowden R (2018) Neural sign language translation. In: CVPR","DOI":"10.1109\/CVPR.2018.00812"},{"key":"9763_CR14","doi-asserted-by":"crossref","unstructured":"Wang S, Guo D, Zhou W-G, Zha Z-J, Wang M (2018) Connectionist temporal fusion for sign language translation. In: ACM multimedia","DOI":"10.1145\/3240508.3240671"},{"key":"9763_CR15","doi-asserted-by":"crossref","unstructured":"Duarte AC (2019) Cross-modal neural sign language translation. In: ACM MM","DOI":"10.1145\/3343031.3352587"},{"key":"9763_CR16","doi-asserted-by":"crossref","unstructured":"Song P, Guo D, Xin H, Wang M (2019) Parallel temporal encoder for sign language translation. In: ICIP. IEEE, pp 1915\u20131919","DOI":"10.1109\/ICIP.2019.8803123"},{"key":"9763_CR17","doi-asserted-by":"crossref","unstructured":"Orbay A, Akarun L (2020) Neural sign language translation by learning tokenization. In: FG, pp 222\u2013228","DOI":"10.1109\/FG47880.2020.00002"},{"key":"9763_CR18","first-page":"1575","volume":"29","author":"D Guo","year":"2020","unstructured":"Guo D, Zhou W, Li A, Li H, Wang M (2020) Hierarchical recurrent deep fusion using adaptive clip summarization for sign language translation. IEEE TIP 29:1575\u20131590","journal-title":"IEEE TIP"},{"key":"9763_CR19","doi-asserted-by":"crossref","unstructured":"Camgoz NC, Koller O, Hadfield S, Bowden R (2020) Multi-channel transformers for multi-articulatory sign language translation. arXiv preprint arXiv:2009.00299","DOI":"10.1007\/978-3-030-66823-5_18"},{"key":"9763_CR20","unstructured":"Li D, Xu C, Yu X, Zhang K, Swift B, Suominen H, Li H (2020) Tspnet: Hierarchical feature learning via temporal semantic pyramid for sign language translation. In: NeurIPS"},{"key":"9763_CR21","doi-asserted-by":"crossref","unstructured":"Zhou H, Zhou W, Qi W, Pu J, Li H (2021) Improving sign language translation with monolingual data by sign back-translation. In: CVPR","DOI":"10.1109\/CVPR46437.2021.00137"},{"key":"9763_CR22","first-page":"4433","volume":"24","author":"S Tang","year":"2021","unstructured":"Tang S, Guo D, Hong R, Wang M (2021) Graph-based multimodal sequential embedding for sign language translation. IEEE TMM 24:4433\u20134445","journal-title":"IEEE TMM"},{"key":"9763_CR23","first-page":"768","volume":"24","author":"H Zhou","year":"2021","unstructured":"Zhou H, Zhou W, Zhou Y, Li H (2021) Spatial-temporal multi-cue network for sign language recognition and translation. IEEE TMM 24:768\u2013779","journal-title":"IEEE TMM"},{"key":"9763_CR24","doi-asserted-by":"crossref","unstructured":"Chen Y, Wei F, Sun X, Wu Z, Lin S (2022) A simple multi-modality transfer learning baseline for sign language translation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 5120\u20135130","DOI":"10.1109\/CVPR52688.2022.00506"},{"key":"9763_CR25","doi-asserted-by":"crossref","unstructured":"Kan J, Hu K, Hagenbuchner M, Tsoi AC, Bennamoun M, Wang Z (2022) Sign language translation with hierarchical spatio-temporal graph neural network. In: Proceedings of the IEEE\/CVF winter conference on applications of computer vision, pp 3367\u20133376","DOI":"10.1109\/WACV51458.2022.00219"},{"key":"9763_CR26","doi-asserted-by":"crossref","unstructured":"Ye J, Jiao W, Wang X, Tu Z, Xiong H (2023) Cross-modality data augmentation for end-to-end sign language translation. arXiv preprint arXiv:2305.11096","DOI":"10.18653\/v1\/2023.findings-emnlp.904"},{"key":"9763_CR27","unstructured":"Zhang B, M\u00fcller M, Sennrich R (2023) Sltunet: A simple unified model for sign language translation. arXiv preprint arXiv:2305.01778"},{"key":"9763_CR28","doi-asserted-by":"crossref","unstructured":"Yin A, Zhong T, Tang L, Jin W, Jin T, Zhao Z (2023) Gloss attention for gloss-free sign language translation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 2551\u20132562","DOI":"10.1109\/CVPR52729.2023.00251"},{"issue":"15","key":"9763_CR29","doi-asserted-by":"publisher","first-page":"23483","DOI":"10.1007\/s11042-022-14172-5","volume":"82","author":"W Xu","year":"2023","unstructured":"Xu W, Ying J, Yang H, Liu J, Hu X (2023) Residual spatial graph convolution and temporal sequence attention network for sign language translation. Multimed Tools Appl 82(15):23483\u201323507","journal-title":"Multimed Tools Appl"},{"key":"9763_CR30","doi-asserted-by":"crossref","unstructured":"Fu B, Ye P, Zhang L, Yu P, Hu C, Shi X, Chen Y (2023) A token-level contrastive framework for sign language translation. In: IEEE international conference on acoustics, speech and signal processing, pp 1\u20135","DOI":"10.1109\/ICASSP49357.2023.10095466"},{"key":"9763_CR31","doi-asserted-by":"crossref","unstructured":"Zheng J, Wang Y, Tan C, Li S, Wang G, Xia J, Chen Y, Li SZ (2023) Cvt-slr: Contrastive visual-textual transformation for sign language recognition with variational alignment. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 23141\u201323150","DOI":"10.1109\/CVPR52729.2023.02216"},{"key":"9763_CR32","doi-asserted-by":"publisher","first-page":"7957","DOI":"10.1007\/s00521-019-04691-y","volume":"32","author":"A Wadhawan","year":"2020","unstructured":"Wadhawan A, Kumar P (2020) Deep learning-based sign language recognition system for static signs. Neural Comput Appl 32:7957\u20137968","journal-title":"Neural Comput Appl"},{"issue":"7","key":"9763_CR33","doi-asserted-by":"publisher","first-page":"9627","DOI":"10.1007\/s11042-021-11595-4","volume":"82","author":"U Nandi","year":"2023","unstructured":"Nandi U, Ghorai A, Singh MM, Changdar C, Bhakta S, Kumar Pal R (2023) Indian sign language alphabet recognition system using CNN with DIFFGRAD optimizer and stochastic pooling. Multimed Tools Appl 82(7):9627\u20139648","journal-title":"Multimed Tools Appl"},{"key":"9763_CR34","doi-asserted-by":"crossref","unstructured":"Boh\u00e1\u010dek M, Hr\u00faz M (2022) Sign pose-based transformer for word-level sign language recognition. In: Proceedings of the IEEE\/CVF winter conference on applications of computer vision, pp 182\u2013191","DOI":"10.1109\/WACVW54805.2022.00024"},{"key":"9763_CR35","doi-asserted-by":"crossref","unstructured":"Cui R, Liu H, Zhang C (2019) A deep neural framework for continuous sign language recognition by iterative training. TMM","DOI":"10.1109\/TMM.2018.2889563"},{"key":"9763_CR36","doi-asserted-by":"crossref","unstructured":"Jang Y, Oh Y, Cho JW, Kim M, Kim D-J, Kweon IS, Chung JS (2023) Self-sufficient framework for continuous sign language recognition. In: IEEE international conference on acoustics, speech and signal processing, pp 1\u20135","DOI":"10.1109\/ICASSP49357.2023.10095732"},{"key":"9763_CR37","doi-asserted-by":"publisher","first-page":"19917","DOI":"10.1007\/s11042-019-7263-7","volume":"78","author":"KM Lim","year":"2019","unstructured":"Lim KM, Tan AWC, Lee CP, Tan SC (2019) Isolated sign language recognition using convolutional neural network hand modelling and hand energy image. Multimed Tools Appl 78:19917\u201319944","journal-title":"Multimed Tools Appl"},{"key":"9763_CR38","doi-asserted-by":"crossref","unstructured":"V\u00e1zquez-Enr\u00edquez M, Alba-Castro JL, Doc\u00edo-Fern\u00e1ndez L, Rodr\u00edguez-Banga E (2021) Isolated sign language recognition with multi-scale spatial-temporal graph convolutional networks. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 3462\u20133471","DOI":"10.1109\/CVPRW53098.2021.00385"},{"key":"9763_CR39","doi-asserted-by":"crossref","unstructured":"Pu J, Zhou W, Li H (2019) Iterative alignment network for continuous sign language recognition. In: CVPR","DOI":"10.1109\/CVPR.2019.00429"},{"key":"9763_CR40","doi-asserted-by":"crossref","unstructured":"Min Y, Hao A, Chai X, Chen X (2021) Visual alignment constraint for continuous sign language recognition. In: Proceedings of the IEEE\/CVF international conference on computer vision, pp 11542\u201311551","DOI":"10.1109\/ICCV48922.2021.01134"},{"key":"9763_CR41","unstructured":"Simonyan K, Zisserman A (2014) Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556"},{"key":"9763_CR42","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: CVPR","DOI":"10.1109\/CVPR.2016.90"},{"key":"9763_CR43","first-page":"1","volume":"25","author":"A Krizhevsky","year":"2012","unstructured":"Krizhevsky A, Sutskever I, Hinton GE (2012) Imagenet classification with deep convolutional neural networks. Adv Neural Inf Process Syst 25:1","journal-title":"Adv Neural Inf Process Syst"},{"key":"9763_CR44","doi-asserted-by":"crossref","unstructured":"Hara K, Kataoka H, Satoh Y (2018) Can spatiotemporal 3d cnns retrace the history of 2d cnns and imagenet? In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 6546\u20136555","DOI":"10.1109\/CVPR.2018.00685"},{"key":"9763_CR45","doi-asserted-by":"crossref","unstructured":"Carreira J, Zisserman A (2017) Quo vadis, action recognition? a new model and the kinetics dataset. In: Proceedings of the IEEE conference on computer vision and pattern recognition, pp 6299\u20136308","DOI":"10.1109\/CVPR.2017.502"},{"key":"9763_CR46","doi-asserted-by":"crossref","unstructured":"Qiu Z, Yao T, Mei T (2017) Learning spatio-temporal representation with pseudo-3d residual networks. In: Proceedings of the IEEE international conference on computer vision, pp 5533\u20135541","DOI":"10.1109\/ICCV.2017.590"},{"issue":"8","key":"9763_CR47","first-page":"1735","volume":"9","author":"LS-T Memory","year":"2010","unstructured":"Memory LS-T (2010) Long short-term memory. Neural Comput 9(8):1735\u20131780","journal-title":"Neural Comput"},{"key":"9763_CR48","unstructured":"Chung J, Gulcehre C, Cho K, Bengio Y (2014) Empirical evaluation of gated recurrent neural networks on sequence modeling. arXiv preprint arXiv:1412.3555"},{"key":"9763_CR49","doi-asserted-by":"crossref","unstructured":"Graves A, Fern\u00e1ndez S, Gomez F, Schmidhuber J (2006) Connectionist temporal classification: labelling unsegmented sequence data with recurrent neural networks. In: ICML","DOI":"10.1145\/1143844.1143891"},{"key":"9763_CR50","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2022.109233","volume":"136","author":"P Xie","year":"2023","unstructured":"Xie P, Cui Z, Du Y, Zhao M, Cui J, Wang B, Hu X (2023) Multi-scale local-temporal similarity fusion for continuous sign language recognition. Pattern Recogn 136:109233","journal-title":"Pattern Recogn"},{"issue":"4","key":"9763_CR51","doi-asserted-by":"publisher","first-page":"541","DOI":"10.1162\/neco.1989.1.4.541","volume":"1","author":"Y LeCun","year":"1989","unstructured":"LeCun Y, Boser B, Denker JS, Henderson D, Howard RE, Hubbard W, Jackel LD (1989) Backpropagation applied to handwritten zip code recognition. Neural Comput 1(4):541\u2013551","journal-title":"Neural Comput"},{"key":"9763_CR52","doi-asserted-by":"crossref","unstructured":"Molchanov P, Yang X, Gupta S, Kim K, Tyree S, Kautz J (2016) Online detection and classification of dynamic hand gestures with recurrent 3d convolutional neural network. In: CVPR","DOI":"10.1109\/CVPR.2016.456"},{"key":"9763_CR53","doi-asserted-by":"crossref","unstructured":"Pu J, Zhou W, Li H (2018) Dilated convolutional network with iterative optimization for continuous sign language recognition. In: IJCAI","DOI":"10.24963\/ijcai.2018\/123"},{"key":"9763_CR54","doi-asserted-by":"crossref","unstructured":"Zhou H, Zhou W, Li H (2019) Dynamic pseudo label decoding for continuous sign language recognition. In: ICME","DOI":"10.1109\/ICME.2019.00223"},{"key":"9763_CR55","doi-asserted-by":"crossref","unstructured":"Li H, Gao L, Han R, Wan L, Feng W (2020) Key action and joint ctc-attention based sign language recognition. In: IEEE international conference on acoustics, speech and signal processing, pp 2348\u20132352","DOI":"10.1109\/ICASSP40776.2020.9054316"},{"key":"9763_CR56","doi-asserted-by":"crossref","unstructured":"Cui R, Liu H, Zhang C (2017) Recurrent convolutional neural networks for continuous sign language recognition by staged optimization. In: CVPR","DOI":"10.1109\/CVPR.2017.175"},{"key":"9763_CR57","unstructured":"Sutskever I, Vinyals O, Le QV (2014) Sequence to sequence learning with neural networks. In: NIPS"},{"key":"9763_CR58","unstructured":"Bahdanau D, Cho K, Bengio Y (2014) Neural machine translation by jointly learning to align and translate. arXiv preprint arXiv:1409.0473"},{"key":"9763_CR59","first-page":"15908","volume":"34","author":"K Han","year":"2021","unstructured":"Han K, Xiao A, Wu E, Guo J, Xu C, Wang Y (2021) Transformer in transformer. Adv Neural Inf Process Syst 34:15908\u201315919","journal-title":"Adv Neural Inf Process Syst"},{"key":"9763_CR60","doi-asserted-by":"crossref","unstructured":"Dai Z, Yang Z, Yang Y, Carbonell J, Le QV, Salakhutdinov R (2019) Transformer-xl: Attentive language models beyond a fixed-length context. arXiv preprint arXiv:1901.02860","DOI":"10.18653\/v1\/P19-1285"},{"key":"9763_CR61","doi-asserted-by":"crossref","unstructured":"Tsai Y-HH, Bai S, Liang PP, Kolter JZ, Morency L-P, Salakhutdinov R (2019) Multimodal transformer for unaligned multimodal language sequences. In: ACL","DOI":"10.18653\/v1\/P19-1656"},{"key":"9763_CR62","doi-asserted-by":"crossref","unstructured":"Zhou H, Zhou W, Zhou Y, Li H (2020) Spatial-temporal multi-cue network for continuous sign language recognition. In: AAAI","DOI":"10.1109\/ICME.2019.00223"},{"key":"9763_CR63","doi-asserted-by":"crossref","unstructured":"Luong MT, Pham H, Manning CD (2015) Effective approaches to attention-based neural machine translation. arXiv preprint arXiv:1508.04025","DOI":"10.18653\/v1\/D15-1166"},{"key":"9763_CR64","doi-asserted-by":"crossref","unstructured":"Wang X, Girshick R, Gupta A, He K (2017) Non-local neural networks. In: CVPR","DOI":"10.1109\/CVPR.2018.00813"},{"key":"9763_CR65","unstructured":"Zhang H, Goodfellow I, Metaxas D, Odena A (2019) Self-attention generative adversarial networks. In: ICML"},{"key":"9763_CR66","unstructured":"Devlin J, Chang M-W, Lee K, Toutanova K (2018) Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805"},{"key":"9763_CR67","unstructured":"Liu Y, Ott M, Goyal N, Du J, Joshi M, Chen D, Levy O, Lewis M, Zettlemoyer L, Stoyanov V (2019) Roberta: a robustly optimized bert pretraining approach. arXiv preprint arXiv:1907.11692"},{"key":"9763_CR68","doi-asserted-by":"crossref","unstructured":"Lin K, Li L, Lin C-C, Ahmed F, Gan Z, Liu Z, Lu Y, Wang L (2022) Swinbert: End-to-end transformers with sparse attention for video captioning. In: CVPR","DOI":"10.1109\/CVPR52688.2022.01742"},{"key":"9763_CR69","unstructured":"Weston J, Chopra S, Bordes A (2014) Memory networks. arXiv preprint arXiv:1410.3916"},{"key":"9763_CR70","doi-asserted-by":"crossref","unstructured":"Cai Q, Pan Y, Yao T, Yan C, Mei T (2018) Memory matching networks for one-shot image recognition. In: CVPR","DOI":"10.1109\/CVPR.2018.00429"},{"key":"9763_CR71","unstructured":"Kumar A, Irsoy O, Ondruska P, Iyyer M, Bradbury J, Gulrajani I, Zhong V, Paulus R, Socher R (2015) Ask me anything: dynamic memory networks for natural language processing. In: ICML"},{"key":"9763_CR72","doi-asserted-by":"crossref","unstructured":"Liu D, Zhou P (2023) Jointly visual-and semantic-aware graph memory networks for temporal sentence localization in videos. In: IEEE international conference on acoustics, speech and signal processing, pp 1\u20135","DOI":"10.1109\/ICASSP49357.2023.10096382"},{"key":"9763_CR73","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2022.109202","volume":"136","author":"T-Z Niu","year":"2023","unstructured":"Niu T-Z, Dong S-S, Chen Z-D, Luo X, Huang Z, Guo S, Xu X-S (2023) A multi-layer memory sharing network for video captioning. Pattern Recogn 136:109202","journal-title":"Pattern Recogn"},{"key":"9763_CR74","unstructured":"Santoro A, Bartunov S, Botvinick M, Wierstra D, Lillicrap T (2016) Meta-learning with memory-augmented neural networks. In: International conference on machine learning. PMLR, pp 1842\u20131850"},{"key":"9763_CR75","unstructured":"Weston JE, Szlam AD, Fergus RD, Sukhbaatar S (2017) End-to-end memory networks. In: NeurIPS"},{"key":"9763_CR76","doi-asserted-by":"crossref","unstructured":"Ma C, Shen C, Dick A, Wu Q, Wang P, Hengel AVD, Reid I (2018) Visual question answering with memory-augmented networks. In: CVPR","DOI":"10.1109\/CVPR.2018.00729"},{"key":"9763_CR77","doi-asserted-by":"crossref","unstructured":"Ravi S, Chinchure A, Sigal L, Liao R, Shwartz V (2023) Vlc-bert: visual question answering with contextualized commonsense knowledge. In: Proceedings of the IEEE\/CVF winter conference on applications of computer vision, pp 1155\u20131165","DOI":"10.1109\/WACV56688.2023.00121"},{"key":"9763_CR78","doi-asserted-by":"crossref","unstructured":"Shao Z, Yu Z, Wang M, Yu J (2023) Prompting large language models with answer heuristics for knowledge-based visual question answering. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 14974\u201314983","DOI":"10.1109\/CVPR52729.2023.01438"},{"key":"9763_CR79","doi-asserted-by":"crossref","unstructured":"Liu Y, Li G, Lin L (2023) Cross-modal causal relational reasoning for event-level visual question answering. IEEE Trans Pattern Anal Mach Intell","DOI":"10.1109\/TPAMI.2023.3284038"},{"key":"9763_CR80","unstructured":"Mikolov T, Chen K, Corrado G, Dean J (2013) Efficient estimation of word representations in vector space. In: ICLR"},{"key":"9763_CR81","unstructured":"Jang E, Gu S, Poole B (2017) Categorical reparameterization with gumbel-softmax. In: ICLR"},{"key":"9763_CR82","doi-asserted-by":"crossref","unstructured":"He K, Fan H, Wu Y, Xie S, Girshick R (2020) Momentum contrast for unsupervised visual representation learning. In: CVPR","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"9763_CR83","unstructured":"Ma S, Zeng Z, McDuff D, Song Y (2020) Learning audio-visual representations with active contrastive coding. arXiv preprint arXiv:2009.09805"},{"key":"9763_CR84","unstructured":"Bahdanau D, Cho K, Bengio Y (2014) Neural machine translation by jointly learning to align and translate. arXiv preprint arXiv:1409.0473"},{"key":"9763_CR85","doi-asserted-by":"crossref","unstructured":"Yin A, Zhao Z, Jin W, Zhang M, Zeng X, He X (2022) Mlslt: Towards multilingual sign language translation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition, pp 5109\u20135119","DOI":"10.1109\/CVPR52688.2022.00505"},{"key":"9763_CR86","unstructured":"Paszke A, Gross S, Massa F, Lerer A, Bradbury J, Chanan G, Killeen T, Lin Z, Gimelshein N, Antiga L et al (2019) Pytorch: an imperative style, high-performance deep learning library. In: NeurIPS"},{"key":"9763_CR87","doi-asserted-by":"crossref","unstructured":"Deng J, Dong W, Socher R, Li L-J, Li K, Fei-Fei L (2009) Imagenet: a large-scale hierarchical image database. In: CVPR","DOI":"10.1109\/CVPR.2009.5206848"},{"issue":"8","key":"9763_CR88","doi-asserted-by":"publisher","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter S, Schmidhuber J (1997) Long short-term memory. Neural Comput 9(8):1735\u20131780","journal-title":"Neural Comput"},{"key":"9763_CR89","first-page":"1","volume":"30","author":"A Vaswani","year":"2017","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser \u0141, Polosukhin I (2017) Attention is all you need. Adv Neural Inf Process Syst 30:1","journal-title":"Adv Neural Inf Process Syst"},{"key":"9763_CR90","unstructured":"Kingma DP, Ba, J (2014) Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980"},{"issue":"11","key":"9763_CR91","doi-asserted-by":"publisher","first-page":"4500","DOI":"10.1109\/TNNLS.2019.2955777","volume":"31","author":"SR Dubey","year":"2019","unstructured":"Dubey SR, Chakraborty S, Roy SK, Mukherjee S, Singh SK, Chaudhuri BB (2019) diffgrad: an optimization method for convolutional neural networks. IEEE Trans Neural Networks Learn Syst 31(11):4500\u20134511","journal-title":"IEEE Trans Neural Networks Learn Syst"},{"key":"9763_CR92","unstructured":"Liu L, Jiang H, He P, Chen W, Liu X, Gao J, Han J (2019) On the variance of the adaptive learning rate and beyond. arXiv preprint arXiv:1908.03265"},{"key":"9763_CR93","unstructured":"Tieleman T, Hinton G (2017) Divide the gradient by a running average of its recent magnitude. coursera: Neural networks for machine learning. Technical report"},{"key":"9763_CR94","unstructured":"Loshchilov I, Hutter F (2017) Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101"},{"key":"9763_CR95","unstructured":"Liu H, Li Z, Hall D, Liang P, Ma T (2023) Sophia: A scalable stochastic second-order optimizer for language model pre-training. arXiv preprint arXiv:2305.14342"},{"key":"9763_CR96","unstructured":"Haibo L, Li H, Gao L, Han R, Wan L, Feng W (2020) Key action and joint ctc-attention based sign language recognition. In: ICASSP"}],"container-title":["Neural Computing and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-024-09763-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00521-024-09763-2\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-024-09763-2.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,7,12]],"date-time":"2024-07-12T10:08:25Z","timestamp":1720778905000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00521-024-09763-2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,4,23]]},"references-count":96,"journal-issue":{"issue":"21","published-print":{"date-parts":[[2024,7]]}},"alternative-id":["9763"],"URL":"https:\/\/doi.org\/10.1007\/s00521-024-09763-2","relation":{},"ISSN":["0941-0643","1433-3058"],"issn-type":[{"value":"0941-0643","type":"print"},{"value":"1433-3058","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,4,23]]},"assertion":[{"value":"19 May 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"25 March 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"23 April 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no conflict of interest to declare that is relevant to the content of this article.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}},{"value":"Ethical approval is not applicable for this article. Verbal informed consent was obtained from the human subjects for their anonymized information to be published in this article.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Ethical approval and informed consent"}}]}}