{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,17]],"date-time":"2026-08-17T13:24:14Z","timestamp":1786973054536,"version":"build-2736575974"},"reference-count":50,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,12,1]],"date-time":"2026-12-01T00:00:00Z","timestamp":1796083200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62276098"],"award-info":[{"award-number":["62276098"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition"],"published-print":{"date-parts":[[2026,12]]},"DOI":"10.1016\/j.patcog.2026.114591","type":"journal-article","created":{"date-parts":[[2026,8,8]],"date-time":"2026-08-08T15:04:20Z","timestamp":1786201460000},"page":"114591","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PF","title":["TCFnet: Temporal-frequency hybrid MetaFormer with multi-stage training for neural tracking of speech"],"prefix":"10.1016","volume":"180","author":[{"given":"Dongdong","family":"Li","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shengyao","family":"Huang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhongliang","family":"Zeng","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhishuo","family":"Jin","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3759-2041","authenticated-orcid":false,"given":"Zhe","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hai","family":"Yang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"issue":"7","key":"10.1016\/j.patcog.2026.114591_b1","doi-asserted-by":"crossref","first-page":"4410","DOI":"10.1109\/TCYB.2022.3178370","article-title":"Deep EEG superresolution via correlating brain structural and functional connectivities","volume":"53","author":"Tang","year":"2023","journal-title":"IEEE Trans. Cybern."},{"key":"10.1016\/j.patcog.2026.114591_b2","series-title":"ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"1","article-title":"EEG2image: Image reconstruction from EEG brain signals","author":"Singh","year":"2023"},{"issue":"9","key":"10.1016\/j.patcog.2026.114591_b3","doi-asserted-by":"crossref","first-page":"10760","DOI":"10.1109\/TPAMI.2023.3263181","article-title":"Decoding visual neural representations by multimodal learning of brain-visual-linguistic features","volume":"45","author":"Du","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.patcog.2026.114591_b4","unstructured":"Y. Benchetrit, H. Banville, J.-R. King, Brain Decoding: Toward Real-Time Reconstruction of Visual Perception, in: The Twelfth International Conference on Learning Representations, 2023."},{"issue":"4","key":"10.1016\/j.patcog.2026.114591_b5","doi-asserted-by":"crossref","DOI":"10.1088\/1741-2552\/ace73f","article-title":"Relating EEG to continuous speech using deep neural networks: a review","volume":"20","author":"Puffay","year":"2023","journal-title":"J. Neural Eng."},{"issue":"5","key":"10.1016\/j.patcog.2026.114591_b6","doi-asserted-by":"crossref","first-page":"980","DOI":"10.1016\/j.neuron.2012.12.037","article-title":"Mechanisms underlying selective neuronal tracking of attended speech at a \u201ccocktail party\u201d","volume":"77","author":"Golumbic","year":"2013","journal-title":"Neuron"},{"issue":"7397","key":"10.1016\/j.patcog.2026.114591_b7","doi-asserted-by":"crossref","first-page":"233","DOI":"10.1038\/nature11020","article-title":"Selective cortical representation of attended speaker in multi-talker speech perception","volume":"485","author":"Mesgarani","year":"2012","journal-title":"Nature"},{"issue":"2","key":"10.1016\/j.patcog.2026.114591_b8","doi-asserted-by":"crossref","first-page":"181","DOI":"10.1007\/s10162-018-0654-z","article-title":"Speech intelligibility predicted from neural entrainment of the speech envelope","volume":"19","author":"Vanthornhout","year":"2018","journal-title":"J. Assoc. Res. Otolaryngol."},{"issue":"5","key":"10.1016\/j.patcog.2026.114591_b9","doi-asserted-by":"crossref","first-page":"402","DOI":"10.1109\/TNSRE.2016.2571900","article-title":"Auditory-inspired speech envelope extraction methods for improved EEG-based auditory attention detection in a cocktail party scenario","volume":"25","author":"Biesmans","year":"2016","journal-title":"IEEE Trans. Neural Syst. Rehabil. Eng."},{"issue":"4","key":"10.1016\/j.patcog.2026.114591_b10","doi-asserted-by":"crossref","DOI":"10.1088\/1741-2552\/ac7976","article-title":"Robust decoding of the speech envelope from EEG recordings through deep neural networks","volume":"19","author":"Thornton","year":"2022","journal-title":"J. Neural Eng."},{"issue":"1","key":"10.1016\/j.patcog.2026.114591_b11","doi-asserted-by":"crossref","first-page":"812","DOI":"10.1038\/s41598-022-27332-2","article-title":"Decoding of the speech envelope from EEG using the VLAAI deep neural network","volume":"13","author":"Accou","year":"2023","journal-title":"Sci. Rep."},{"key":"10.1016\/j.patcog.2026.114591_b12","series-title":"ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"1","article-title":"Happyquokka system for ICASSP 2023 auditory EEG challenge","author":"Piao","year":"2023"},{"key":"10.1016\/j.patcog.2026.114591_b13","series-title":"ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"1","article-title":"Decoding auditory EEG responses using an adapted wavenet","author":"Van Dyck","year":"2023"},{"key":"10.1016\/j.patcog.2026.114591_b14","doi-asserted-by":"crossref","DOI":"10.1177\/23312165241282872","article-title":"ADT network: A novel nonlinear method for decoding speech envelopes from EEG signals","author":"Liu","year":"2024","journal-title":"Trends Hear."},{"key":"10.1016\/j.patcog.2026.114591_b15","unstructured":"Y. Ren, C. Hu, X. Tan, T. Qin, S. Zhao, Z. Zhao, T.-Y. Liu, FastSpeech 2: Fast and High-Quality End-to-End Text to Speech, in: International Conference on Learning Representations, 2020."},{"key":"10.1016\/j.patcog.2026.114591_b16","unstructured":"Z. Ju, Y. Wang, K. Shen, X. Tan, D. Xin, D. Yang, E. Liu, Y. Leng, K. Song, S. Tang, Z. Wu, T. Qin, X. Li, W. Ye, S. Zhang, J. Bian, L. He, J. Li, S. Zhao, NaturalSpeech 3: Zero-Shot Speech Synthesis with Factorized Codec and Diffusion Models, in: Forty-First International Conference on Machine Learning, 2024."},{"key":"10.1016\/j.patcog.2026.114591_b17","series-title":"2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"10809","article-title":"Metaformer is actually what you need for vision","author":"Yu","year":"2022"},{"key":"10.1016\/j.patcog.2026.114591_b18","series-title":"Proceedings of the 31st International Conference on Neural Information Processing Systems","first-page":"6000","article-title":"Attention is all you need","author":"Vaswani","year":"2017"},{"key":"10.1016\/j.patcog.2026.114591_b19","series-title":"2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"15979","article-title":"Masked autoencoders are scalable vision learners","author":"He","year":"2022"},{"key":"10.1016\/j.patcog.2026.114591_b20","first-page":"12449","article-title":"Wav2vec 2.0: A framework for self-supervised learning of speech representations","volume":"vol. 33","author":"Baevski","year":"2020"},{"issue":"2","key":"10.1016\/j.patcog.2026.114591_b21","doi-asserted-by":"crossref","first-page":"777","DOI":"10.1109\/JBHI.2023.3335854","article-title":"ST-SCGNN: A spatio-temporal self-constructing graph neural network for cross-subject EEG-based emotion recognition and consciousness detection","volume":"28","author":"Pan","year":"2024","journal-title":"IEEE J. Biomed. Health Inform."},{"issue":"25","key":"10.1016\/j.patcog.2026.114591_b22","doi-asserted-by":"crossref","first-page":"15843","DOI":"10.1007\/s00521-024-09731-w","article-title":"TD-LSTM: A time distributed and deep-learning-based architecture for classification of motor imagery and execution in EEG signals","volume":"36","author":"Karimian-Kelishadrokhi","year":"2024","journal-title":"Neural Comput. Appl."},{"key":"10.1016\/j.patcog.2026.114591_b23","article-title":"Graph-regularized geometric deep learning for motor imagery EEG decoding via multi-domain fusion","author":"Chu","year":"2025","journal-title":"Pattern Recognit."},{"issue":"1","key":"10.1016\/j.patcog.2026.114591_b24","doi-asserted-by":"crossref","first-page":"20237","DOI":"10.1038\/s41598-024-71118-7","article-title":"CTNet: A convolutional transformer network for EEG-based motor imagery classification","volume":"14","author":"Zhao","year":"2024","journal-title":"Sci. Rep."},{"key":"10.1016\/j.patcog.2026.114591_b25","doi-asserted-by":"crossref","first-page":"710","DOI":"10.1109\/TNSRE.2022.3230250","article-title":"EEG conformer: Convolutional transformer for EEG decoding and visualization","volume":"31","author":"Song","year":"2023","journal-title":"IEEE Trans. Neural Syst. Rehabil. Eng."},{"key":"10.1016\/j.patcog.2026.114591_b26","doi-asserted-by":"crossref","DOI":"10.3389\/fnhum.2021.653659","article-title":"BENDR: Using transformers and a contrastive self-supervised learning task to learn from massive amounts of EEG data","volume":"15","author":"Kostas","year":"2021","journal-title":"Front. Hum. Neurosci."},{"key":"10.1016\/j.patcog.2026.114591_b27","unstructured":"W. Jiang, L. Zhao, B.-l. Lu, Large Brain Model for Learning Generic Representations with Tremendous EEG Data in BCI, in: The Twelfth International Conference on Learning Representations, 2023."},{"issue":"1","key":"10.1016\/j.patcog.2026.114591_b28","doi-asserted-by":"crossref","first-page":"48","DOI":"10.1038\/s41467-021-27725-3","article-title":"Imagined speech can be decoded from low- and cross-frequency intracranial EEG features","volume":"13","author":"Proix","year":"2022","journal-title":"Nat. Commun."},{"issue":"1","key":"10.1016\/j.patcog.2026.114591_b29","doi-asserted-by":"crossref","first-page":"6510","DOI":"10.1038\/s41467-022-33611-3","article-title":"Generalizable spelling using a speech neuroprosthesis in an individual with severe limb and vocal paralysis","volume":"13","author":"Metzger","year":"2022","journal-title":"Nat. Commun."},{"issue":"5","key":"10.1016\/j.patcog.2026.114591_b30","doi-asserted-by":"crossref","DOI":"10.1088\/1741-2552\/aace8c","article-title":"EEGNet: A compact convolutional network for EEG-based brain-computer interfaces","volume":"15","author":"Lawhern","year":"2018","journal-title":"J. Neural Eng."},{"key":"10.1016\/j.patcog.2026.114591_b31","series-title":"Proceedings of the 38th International Conference on Machine Learning","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford","year":"2021"},{"key":"10.1016\/j.patcog.2026.114591_b32","doi-asserted-by":"crossref","DOI":"10.1038\/s42256-023-00714-5","article-title":"Decoding speech perception from non-invasive brain recordings","author":"D\u00e9fossez","year":"2023","journal-title":"Nat. Mach. Intell."},{"issue":"1","key":"10.1016\/j.patcog.2026.114591_b33","doi-asserted-by":"crossref","first-page":"28744","DOI":"10.1038\/s41598-025-13646-4","article-title":"Contrastive representation learning with transformers for robust auditory EEG decoding","volume":"15","author":"Bollens","year":"2025","journal-title":"Sci. Rep."},{"key":"10.1016\/j.patcog.2026.114591_b34","doi-asserted-by":"crossref","unstructured":"C. Fan, S. Zhang, J. Zhang, E. Liu, X. Li, G. Zhao, Z. Lv, Dmf2mel: A dynamic multiscale fusion network for eeg-driven mel spectrogram reconstruction, in: Proceedings of the 33rd ACM International Conference on Multimedia, 2025, pp. 6977\u20136985.","DOI":"10.1145\/3746027.3755501"},{"key":"10.1016\/j.patcog.2026.114591_b35","series-title":"ICASSP 2025-2025 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"1","article-title":"SSM2Mel: State space model to reconstruct mel spectrogram from the EEG","author":"Fan","year":"2025"},{"key":"10.1016\/j.patcog.2026.114591_b36","series-title":"2024 IEEE International Conference on Acoustics, Speech, and Signal Processing Workshops","first-page":"113","article-title":"ConvConcatNet: A deep convolutional neural network to reconstruct mel spectrogram from the EEG","author":"Xu","year":"2024"},{"key":"10.1016\/j.patcog.2026.114591_b37","series-title":"2023 IEEE EMBS International Conference on Biomedical and Health Informatics","first-page":"1","article-title":"BrainTalker: Low-resource brain-to-speech synthesis with transfer learning using Wav2Vec 2.0","author":"Kim","year":"2023"},{"key":"10.1016\/j.patcog.2026.114591_b38","article-title":"Fastspeech: Fast, robust and controllable text to speech","volume":"vol. 32","author":"Ren","year":"2019"},{"key":"10.1016\/j.patcog.2026.114591_b39","series-title":"Proc. Interspeech 2020","first-page":"5036","article-title":"Conformer: Convolution-augmented transformer for speech recognition","author":"Gulati","year":"2020"},{"key":"10.1016\/j.patcog.2026.114591_b40","doi-asserted-by":"crossref","first-page":"3","DOI":"10.1016\/j.neunet.2017.12.012","article-title":"Sigmoid-weighted linear units for neural network function approximation in reinforcement learning","volume":"107","author":"Elfwing","year":"2018","journal-title":"Neural Netw."},{"key":"10.1016\/j.patcog.2026.114591_b41","article-title":"Root mean square layer normalization","volume":"vol. 32","author":"Zhang","year":"2019"},{"issue":"9","key":"10.1016\/j.patcog.2026.114591_b42","doi-asserted-by":"crossref","first-page":"10960","DOI":"10.1109\/TPAMI.2023.3263824","article-title":"GFNet: Global filter networks for visual recognition","volume":"45","author":"Rao","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.patcog.2026.114591_b43","doi-asserted-by":"crossref","unstructured":"Y. Liu, S. Zhang, J. Chen, Z. Yu, K. Chen, D. Lin, Improving Pixel-Based MIM by Reducing Wasted Modeling Capability, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 5361\u20135372.","DOI":"10.1109\/ICCV51070.2023.00494"},{"key":"10.1016\/j.patcog.2026.114591_b44","doi-asserted-by":"crossref","first-page":"35946","DOI":"10.52202\/068431-2605","article-title":"Masked autoencoders as spatiotemporal learners","volume":"35","author":"Feichtenhofer","year":"2022","journal-title":"Adv. Neural Inf. Process. Syst."},{"issue":"29","key":"10.1016\/j.patcog.2026.114591_b45","doi-asserted-by":"crossref","first-page":"5750","DOI":"10.1523\/JNEUROSCI.1828-18.2019","article-title":"Neural speech tracking in the theta and in the delta frequency band differentially encode clarity and comprehension of speech in noise","volume":"39","author":"Etard","year":"2019","journal-title":"J. Neurosci."},{"issue":"1","key":"10.1016\/j.patcog.2026.114591_b46","doi-asserted-by":"crossref","first-page":"155","DOI":"10.1162\/jocn_a_01467","article-title":"Cortical tracking of surprisal during continuous speech comprehension","volume":"32","author":"Weissbart","year":"2020","journal-title":"J. Cogn. Neurosci."},{"key":"10.1016\/j.patcog.2026.114591_b47","series-title":"SparrKULee: A speech-evoked auditory response repository of the KU leuven, containing EEG of 85 participants","author":"Accou","year":"2023"},{"key":"10.1016\/j.patcog.2026.114591_b48","series-title":"2024 IEEE International Conference on Acoustics, Speech, and Signal Processing Workshops","first-page":"127","article-title":"ICASSP 2024 auditory EEG decoding challenge","author":"Bollens","year":"2024"},{"key":"10.1016\/j.patcog.2026.114591_b49","doi-asserted-by":"crossref","DOI":"10.3389\/fnins.2021.705621","article-title":"Linear modeling of neurophysiological responses to speech and other continuous stimuli: methodological considerations for applied research","volume":"15","author":"Crosse","year":"2021","journal-title":"Front. Neurosci."},{"key":"10.1016\/j.patcog.2026.114591_b50","series-title":"Proceedings of the 2022 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies","first-page":"4296","article-title":"FNet: Mixing tokens with Fourier transforms","author":"Lee-Thorp","year":"2022"}],"container-title":["Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326015554?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0031320326015554?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,8,17]],"date-time":"2026-08-17T12:25:12Z","timestamp":1786969512000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0031320326015554"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,12]]},"references-count":50,"alternative-id":["S0031320326015554"],"URL":"https:\/\/doi.org\/10.1016\/j.patcog.2026.114591","relation":{},"ISSN":["0031-3203"],"issn-type":[{"value":"0031-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,12]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"TCFnet: Temporal-frequency hybrid MetaFormer with multi-stage training for neural tracking of speech","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patcog.2026.114591","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"114591"}}