{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,29]],"date-time":"2026-06-29T13:57:09Z","timestamp":1782741429853,"version":"3.54.5"},"reference-count":34,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100016974","name":"Quanzhou Normal University","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100016974","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003392","name":"Fujian Provincial Natural Science Foundation","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100003392","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100018624","name":"Science and Technology Bureau of Quanzhou","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100018624","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Neurocomputing"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.neucom.2026.134349","type":"journal-article","created":{"date-parts":[[2026,6,25]],"date-time":"2026-06-25T15:20:09Z","timestamp":1782400809000},"page":"134349","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Multi-band residual fusion with per-head gating for enhanced transformer language models"],"prefix":"10.1016","volume":"699","author":[{"given":"Zhigao","family":"Huang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Miao","family":"Pan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Quanfa","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.neucom.2026.134349_bib0005","series-title":"Advances in Neural Information Processing Systems","first-page":"5998","article-title":"Attention is all you need","volume":"vol. 30","author":"Vaswani","year":"2017"},{"key":"10.1016\/j.neucom.2026.134349_bib0010","series-title":"Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long and Short Papers)","first-page":"4171","article-title":"BERT: pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2019"},{"key":"10.1016\/j.neucom.2026.134349_bib0015","series-title":"Advances in Neural Information Processing Systems","first-page":"1877","article-title":"Language models are few-shot learners","volume":"vol. 33","author":"Brown","year":"2020"},{"key":"10.1016\/j.neucom.2026.134349_bib0020","series-title":"Discrete-Time Signal Processing","author":"Oppenheim","year":"1999"},{"key":"10.1016\/j.neucom.2026.134349_bib0025","author":"Wang"},{"key":"10.1016\/j.neucom.2026.134349_bib0030","series-title":"International Conference on Learning Representations (ICLR)","article-title":"Rethinking attention with performers","author":"Choromanski","year":"2021"},{"key":"10.1016\/j.neucom.2026.134349_bib0035","series-title":"Advances in Neural Information Processing Systems","first-page":"17283","article-title":"Big bird: transformers for longer sequences","volume":"vol. 33","author":"Zaheer","year":"2020"},{"key":"10.1016\/j.neucom.2026.134349_bib0040","series-title":"Advances in Neural Information Processing Systems","first-page":"16344","article-title":"FlashAttention: fast and memory-efficient exact attention with IO-awareness","volume":"vol. 35","author":"Dao","year":"2022"},{"key":"10.1016\/j.neucom.2026.134349_bib0045","series-title":"International Conference on Learning Representations (ICLR)","article-title":"FlashAttention-2: faster attention with better parallelism and work partitioning","author":"Dao","year":"2024"},{"issue":"120","key":"10.1016\/j.neucom.2026.134349_bib0050","first-page":"1","article-title":"Switch transformers: scaling to trillion parameter models with simple and efficient sparsity","volume":"23","author":"Fedus","year":"2022","journal-title":"J. Mach. Learn. Res."},{"key":"10.1016\/j.neucom.2026.134349_bib0055","author":"He"},{"issue":"140","key":"10.1016\/j.neucom.2026.134349_bib0060","first-page":"1","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","volume":"21","author":"Raffel","year":"2020","journal-title":"J. Mach. Learn. Res."},{"key":"10.1016\/j.neucom.2026.134349_bib0065","author":"Touvron"},{"key":"10.1016\/j.neucom.2026.134349_bib0070","author":"Touvron"},{"key":"10.1016\/j.neucom.2026.134349_bib0075","author":"Achiam"},{"key":"10.1016\/j.neucom.2026.134349_bib0080","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2023.127063","article-title":"RoFormer: enhanced transformer with rotary position embedding","volume":"568","author":"Su","year":"2024","journal-title":"Neurocomputing"},{"key":"10.1016\/j.neucom.2026.134349_bib0085","series-title":"International Conference on Learning Representations (ICLR)","article-title":"Fourier neural operator for parametric partial differential equations","author":"Li","year":"2021"},{"key":"10.1016\/j.neucom.2026.134349_bib0090","series-title":"International Conference on Learning Representations (ICLR)","article-title":"Spectral networks and locally connected networks on graphs","author":"Bruna","year":"2014"},{"key":"10.1016\/j.neucom.2026.134349_bib0095","series-title":"9th ISCA Speech Synthesis Workshop","first-page":"125","article-title":"WaveNet: a generative model for raw audio","author":"Oord","year":"2016"},{"key":"10.1016\/j.neucom.2026.134349_bib0100","series-title":"International Conference on Learning Representations","article-title":"Exploring sparsity in recurrent neural networks","author":"Narang","year":"2017"},{"key":"10.1016\/j.neucom.2026.134349_bib0105","series-title":"Proceedings of the 2022 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Volume 1: Long Papers)","first-page":"4296","article-title":"FNet: mixing tokens with Fourier transforms","author":"Lee-Thorp","year":"2022"},{"key":"10.1016\/j.neucom.2026.134349_bib0110","series-title":"International Conference on Learning Representations (ICLR)","article-title":"Efficiently modeling long sequences with structured state spaces","author":"Gu","year":"2022"},{"key":"10.1016\/j.neucom.2026.134349_bib0115","author":"Gu"},{"key":"10.1016\/j.neucom.2026.134349_bib0120","series-title":"Proceedings of the 40th International Conference on Machine Learning","first-page":"28043","article-title":"Hyena hierarchy: towards larger convolutional language models","volume":"vol. 202","author":"Poli","year":"2023"},{"key":"10.1016\/j.neucom.2026.134349_bib0125","series-title":"Advances in Neural Information Processing Systems","first-page":"103031","article-title":"VMamba: visual state space model","volume":"vol. 37","author":"Liu","year":"2024"},{"issue":"8","key":"10.1016\/j.neucom.2026.134349_bib0130","doi-asserted-by":"crossref","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","article-title":"Long short-term memory","volume":"9","author":"Hochreiter","year":"1997","journal-title":"Neural Comput."},{"key":"10.1016\/j.neucom.2026.134349_bib0135","series-title":"Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing (EMNLP)","first-page":"1724","article-title":"Learning phrase representations using RNN encoder-decoder for statistical machine translation","author":"Cho","year":"2014"},{"key":"10.1016\/j.neucom.2026.134349_bib0140","series-title":"Proceedings of the 34th International Conference on Machine Learning","first-page":"933","article-title":"Language modeling with gated convolutional networks","volume":"vol. 70","author":"Dauphin","year":"2017"},{"key":"10.1016\/j.neucom.2026.134349_bib0145","author":"Shazeer"},{"key":"10.1016\/j.neucom.2026.134349_bib0150","series-title":"International Conference on Learning Representations (ICLR)","article-title":"Outrageously large neural networks: the sparsely-gated mixture-of-experts layer","author":"Shazeer","year":"2017"},{"issue":"2","key":"10.1016\/j.neucom.2026.134349_bib0155","first-page":"313","article-title":"Building a large annotated corpus of english: the penn treebank","volume":"19","author":"Marcus","year":"1993","journal-title":"Comput. Linguist."},{"key":"10.1016\/j.neucom.2026.134349_bib0160","series-title":"International Conference on Learning Representations (ICLR)","article-title":"Pointer sentinel mixture models","author":"Merity","year":"2017"},{"key":"10.1016\/j.neucom.2026.134349_bib0165","series-title":"Language models are unsupervised multitask learners","author":"Radford","year":"2019"},{"key":"10.1016\/j.neucom.2026.134349_bib0170","series-title":"International Conference on Learning Representations (ICLR)","article-title":"Decoupled weight decay regularization","author":"Loshchilov","year":"2019"}],"container-title":["Neurocomputing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0925231226017479?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0925231226017479?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,6,29]],"date-time":"2026-06-29T13:24:14Z","timestamp":1782739454000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0925231226017479"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":34,"alternative-id":["S0925231226017479"],"URL":"https:\/\/doi.org\/10.1016\/j.neucom.2026.134349","relation":{},"ISSN":["0925-2312"],"issn-type":[{"value":"0925-2312","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Multi-band residual fusion with per-head gating for enhanced transformer language models","name":"articletitle","label":"Article Title"},{"value":"Neurocomputing","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.neucom.2026.134349","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"134349"}}