{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,14]],"date-time":"2026-07-14T16:05:02Z","timestamp":1784045102283,"version":"3.55.0"},"reference-count":55,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2026,8]]},"DOI":"10.1016\/j.eswa.2026.132634","type":"journal-article","created":{"date-parts":[[2026,4,27]],"date-time":"2026-04-27T23:44:54Z","timestamp":1777333494000},"page":"132634","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Expert Transformer: Fine-Grained Mixture-of-Experts for Time-Series forecasting"],"prefix":"10.1016","volume":"324","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-4222-603X","authenticated-orcid":false,"given":"Janghoon","family":"Yang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.132634_b0005","series-title":"Advances in Neural Information Processing Systems","first-page":"30","article-title":"Attention is all you need","author":"Vaswani","year":"2017"},{"issue":"1","key":"10.1016\/j.eswa.2026.132634_b0010","doi-asserted-by":"crossref","first-page":"79","DOI":"10.1162\/neco.1991.3.1.79","article-title":"Adaptive mixtures of local experts","volume":"3","author":"Jacobs","year":"1991","journal-title":"Neural Computation"},{"issue":"8","key":"10.1016\/j.eswa.2026.132634_b0015","doi-asserted-by":"crossref","first-page":"1177","DOI":"10.1109\/TNNLS.2012.2200299","article-title":"Twenty years of mixture of experts","volume":"23","author":"Yuksel","year":"2012","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"issue":"2","key":"10.1016\/j.eswa.2026.132634_b0020","doi-asserted-by":"crossref","first-page":"181","DOI":"10.1162\/neco.1994.6.2.181","article-title":"Hierarchical mixtures of experts and the EM algorithm","volume":"6","author":"Jordan","year":"1994","journal-title":"Neural Computation"},{"key":"10.1016\/j.eswa.2026.132634_b0025","unstructured":"Shazeer, N., Mirhoseini, A., Maziarz, K., Davis, A., Le, Q., Hinton, G., & Dean, J. (2017). Outrageously large neural networks: The sparsely-gated mixture-of-experts layer. arXiv preprint arXiv:1701.06538."},{"issue":"120","key":"10.1016\/j.eswa.2026.132634_b0030","first-page":"1","article-title":"Switch transformers: Scaling to trillion parameter models with simple and efficient sparsity","volume":"23","author":"Fedus","year":"2022","journal-title":"Journal of Machine Learning Research"},{"key":"10.1016\/j.eswa.2026.132634_b0035","unstructured":"Rajbhandari, S., Li, C., Yao, Z., Zhang, M., Aminabadi, R. Y., Awan, A. A., Rasley, J., & He, Y. (2022). DeepSpeed-MoE: Advancing mixture-of-experts inference and training to power next-generation AI scale. Proceedings of the 39th International Conference on Machine Learning, 162, 18332\u201318346."},{"key":"10.1016\/j.eswa.2026.132634_b0040","unstructured":"Box, G. E. P., & Jenkins, G. M. (1970). Time series analysis: Forecasting and control. Holden-Day."},{"key":"10.1016\/j.eswa.2026.132634_b0045","unstructured":"Hyndman, R. J., & Athanasopoulos, G. (2021). Forecasting: Principles and practice (3rd ed.). OTexts."},{"key":"10.1016\/j.eswa.2026.132634_b0050","series-title":"International Conference on Learning Representations","article-title":"A time series is worth 64 words: Long-term forecasting with transformers","author":"Nie","year":"2023"},{"key":"10.1016\/j.eswa.2026.132634_b0055","unstructured":"Jin, M., Wang, S., Ma, L., Chu, Z., Zhang, J. Y., Shi, X., Chen, P.-Y., Liang, Y., Li, Y.-F., Pan, S., & Wen, Q. (2024). Time-LLM: Time series forecasting by reprogramming large language models. International Conference on Learning Representations."},{"issue":"8065","key":"10.1016\/j.eswa.2026.132634_b0060","doi-asserted-by":"crossref","first-page":"1172","DOI":"10.1038\/s41586-025-08897-0","article-title":"End-to-end data-driven weather prediction","volume":"641","author":"Allen","year":"2025","journal-title":"Nature"},{"key":"10.1016\/j.eswa.2026.132634_b0065","doi-asserted-by":"crossref","first-page":"8699","DOI":"10.1038\/s41598-025-93227-7","article-title":"Developing a seasonal-adjusted machine-learning-based hybrid time-series model to forecast heatwave warning","volume":"15","author":"Qureshi","year":"2025","journal-title":"Scientific Reports"},{"issue":"7","key":"10.1016\/j.eswa.2026.132634_b0070","doi-asserted-by":"crossref","first-page":"695","DOI":"10.3390\/e27070695","article-title":"A hybrid framework integrating traditional models and deep learning for multi-scale time series forecasting","volume":"27","author":"Liu","year":"2025","journal-title":"Entropy"},{"key":"10.1016\/j.eswa.2026.132634_b0075","doi-asserted-by":"crossref","first-page":"194","DOI":"10.1007\/s44196-025-00930-4","article-title":"A residual-corrected hybrid ARIMA\u2013CNN\u2013LSTM framework for high-accuracy tobacco sales forecasting in regulated markets","volume":"18","author":"Huang","year":"2025","journal-title":"International Journal of Computational Intelligence Systems"},{"key":"10.1016\/j.eswa.2026.132634_b0080","series-title":"International Conference on Learning Representations","article-title":"Time-MoE: Billion-scale time series foundation models with mixture of experts","author":"Shi","year":"2025"},{"key":"10.1016\/j.eswa.2026.132634_b0085","unstructured":"Eigen, D., Ranzato, M., & Sutskever, I. (2013). Learning factored representations in a deep mixture of experts. arXiv preprint arXiv:1312.4314."},{"key":"10.1016\/j.eswa.2026.132634_b0090","unstructured":"Du, N., Huang, Y., Dai, A. M., Tong, S., Lepikhin, D., Xu, Y., Krikun, M., Zhou, Y., Yu, A. W., Firat, O., Zoph, B., Fedus, L., Bosma, M., Zhou, Z., Wang, T., Wang, E., Webster, K., Pellat, M., Robinson, K., Meier-Hellstern, K., Duke, T., Dixon, L., Zhang, K., Le, Q., Wu, Y., Chen, Z., & Cui, C. (2022). GLaM: Efficient scaling of language models with mixture-of-experts. Proceedings of the 39th International Conference on Machine Learning, 162, 5547\u20135569."},{"key":"10.1016\/j.eswa.2026.132634_b0095","unstructured":"Bengio, Y., L\u00e9onard, N., & Courville, A. (2013). Estimating or propagating gradients through stochastic neurons for conditional computation. arXiv preprint arXiv:1308.3432."},{"key":"10.1016\/j.eswa.2026.132634_b0100","unstructured":"Ren, J., Li, Y., Ding, Z., Pan, W., & Dong, H. (2021). Probabilistic mixture-of-experts for efficient deep reinforcement learning. arXiv preprint arXiv:2104.09122."},{"key":"10.1016\/j.eswa.2026.132634_b0105","unstructured":"Omi, N., Sen, S., & Farhadi, A. (2025). Load balancing mixture of experts with similarity preserving routers. arXiv preprint arXiv:2506.14038."},{"key":"10.1016\/j.eswa.2026.132634_b0110","article-title":"Categorical reparameterization with Gumbel-Softmax","author":"Jang","year":"2017","journal-title":"International Conference on Learning"},{"key":"10.1016\/j.eswa.2026.132634_b0115","unstructured":"Zhang, D., Song, J., Bi, Z., Yuan, Y., Wang, T., Yeong, J., & Hao, J. (2025). Mixture of experts in large language models. arXiv preprint arXiv:2507.11181."},{"key":"10.1016\/j.eswa.2026.132634_b0120","series-title":"Advances in Neural Information Processing Systems","first-page":"37","article-title":"Spiking transformer with experts mixture","author":"Zhou","year":"2024"},{"key":"10.1016\/j.eswa.2026.132634_b0125","unstructured":"Badjie, B., Cec\u00edlio, J., & Casimiro, A. (2025). Double-stage feature-level clustering-based mixture of experts framework. arXiv preprint arXiv:2503.09504."},{"key":"10.1016\/j.eswa.2026.132634_b0130","doi-asserted-by":"crossref","unstructured":"Zhou, Y., Karamanolakis, G., Soto, V., Rumshisky, A., Kulkarni, M., Huang, F., Ai, W., & Lu, J. (2025). MergeME: Model merging techniques for homogeneous and heterogeneous MoEs. arXiv preprint arXiv:2502.00997.","DOI":"10.18653\/v1\/2025.naacl-long.117"},{"key":"10.1016\/j.eswa.2026.132634_b0135","unstructured":"Shen, L., Tang, A., Yang, E., Guo, G., Luo, Y., Zhang, L., Cao, X., Du, B., & Tao, D. (2024). Efficient and effective weight-ensembling mixture of experts for multi-task model merging. arXiv preprint arXiv:2410.21804."},{"key":"10.1016\/j.eswa.2026.132634_b0140","series-title":"International Conference on Learning Representations","article-title":"Dynamic mixture of experts: An auto-tuning approach for efficient transformer models","author":"Guo","year":"2025"},{"key":"10.1016\/j.eswa.2026.132634_b0145","article-title":"From sparse to soft mixtures of experts","author":"Puigcerver","year":"2024","journal-title":"International Conference on Learning"},{"key":"10.1016\/j.eswa.2026.132634_b0150","unstructured":"Mu, S., & Lin, S. (2025). Mixture-of-experts: Algorithms, theory, and applications. arXiv preprint arXiv:2503.07137."},{"issue":"636\u2013649","key":"10.1016\/j.eswa.2026.132634_b0155","first-page":"267","article-title":"On a method of investigating periodicities in disturbed series, with special reference to Wolfer's sunspot numbers. Philosophical transactions of the Royal Society of London","volume":"226","author":"Yule","year":"1927","journal-title":"Series A, Containing Papers of a Mathematical or Physical Character"},{"key":"10.1016\/j.eswa.2026.132634_b0160","unstructured":"Whittle, P. (1951). Hypothesis testing in time series analysis. Ph.D. thesis, Uppsala University."},{"issue":"4","key":"10.1016\/j.eswa.2026.132634_b0165","doi-asserted-by":"crossref","first-page":"987","DOI":"10.2307\/1912773","article-title":"Autoregressive conditional heteroscedasticity with estimates of the variance of United Kingdom inflation","volume":"50","author":"Engle","year":"1982","journal-title":"Econometrica"},{"issue":"2","key":"10.1016\/j.eswa.2026.132634_b0170","first-page":"897","article-title":"A SARIMAX coupled modelling applied to individual load curves intraday forecasting","volume":"4","author":"Bercu","year":"2013","journal-title":"IEEE Transactions on Smart Grid"},{"issue":"6088","key":"10.1016\/j.eswa.2026.132634_b0175","doi-asserted-by":"crossref","first-page":"533","DOI":"10.1038\/323533a0","article-title":"Learning representations by back-propagating errors","volume":"323","author":"Rumelhart","year":"1986","journal-title":"Nature"},{"issue":"8","key":"10.1016\/j.eswa.2026.132634_b0180","doi-asserted-by":"crossref","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","article-title":"Long short-term memory","volume":"9","author":"Hochreiter","year":"1997","journal-title":"Neural Computation"},{"key":"10.1016\/j.eswa.2026.132634_b0185","article-title":"Time series forecasting model based on the adapted transformer neural network and FFT-based features extraction","volume":"238","author":"Yemets","year":"2024","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.132634_b0190","unstructured":"Gu, A., & Dao, T. (2023). Mamba: Linear-time sequence modeling with selective state spaces. arXiv preprint arXiv:2312.00752."},{"key":"10.1016\/j.eswa.2026.132634_b0195","doi-asserted-by":"crossref","unstructured":"Wu, Y., Meng, X., Hu, H., Zhang, J., Dong, Y., & Lu, D. (2025). Affirm: Interactive Mamba with adaptive Fourier filters for long-term time series forecasting. In Proceedings of the Thirty-Ninth AAAI Conference on Artificial Intelligence and Thirty-Seventh Conference on Innovative Applications of Artificial Intelligence and Fifteenth Symposium on Educational Advances in Artificial Intelligence (AAAI'25\/IAAI'25\/EAAI'25) (Vol. 39, pp. 21599\u201321607). AAAI Press. https:\/\/doi.org\/10.1609\/aaai.v39i20.35463.","DOI":"10.1609\/aaai.v39i20.35463"},{"key":"10.1016\/j.eswa.2026.132634_b0200","unstructured":"Zeng, C., Liu, Z., Zheng, G., Kong, L., & Pan, S. (2024). CMamba: Channel correlation enhanced state space models for multivariate time series forecasting. arXiv preprint arXiv:2407.13258."},{"key":"10.1016\/j.eswa.2026.132634_b0205","series-title":"Advances in Neural Information Processing Systems","first-page":"36","article-title":"Large language models are zero-shot time series forecasters","author":"Gruver","year":"2023"},{"key":"10.1016\/j.eswa.2026.132634_b0210","unstructured":"Requeima, J., Allingham, J. U., & Turner, R. E. (2024). Lag-Llama: Towards foundation models for probabilistic time series forecasting. arXiv preprint arXiv:2406.08268."},{"key":"10.1016\/j.eswa.2026.132634_b0215","unstructured":"Liu, X., Wang, S., Chu, Z., Zhang, J. Y., Jin, M., Li, Y., Pan, S., Liang, Y., & Wen, Q. (2024). TsLLM: Augmenting LLMs for general time series understanding and forecasting. International Conference on Learning Representations."},{"key":"10.1016\/j.eswa.2026.132634_b0220","unstructured":"Zhou, H., Zhang, S., Peng, J., Zhang, S., Li, J., Xiong, H., & Zhang, W. (2023). GPT4TS: Forecasting time series with LLMs via patch-based prompting. arXiv preprint arXiv:2311.14561."},{"key":"10.1016\/j.eswa.2026.132634_b0225","series-title":"International Conference on Learning Representations","article-title":"Moirai-MoE: Empowering time series foundation models with sparse mixture of experts","author":"Liu","year":"2025"},{"key":"10.1016\/j.eswa.2026.132634_b0230","unstructured":"Liu, Z. (2024). FreqMoE: Enhancing time series forecasting through frequency decomposition mixture of experts. arXiv preprint arXiv:2408.07508."},{"key":"10.1016\/j.eswa.2026.132634_b0235","unstructured":"Ni, R., Lin, Z., Wang, S., & Fanti, G. (2024). Mixture-of-linear-experts for long-term time series forecasting. Proceedings of the 27th International Conference on Artificial Intelligence and Statistics, 238, 4672\u20134680."},{"key":"10.1016\/j.eswa.2026.132634_b0240","unstructured":"Ni, Z., Ma, X., Wu, Z., Xiao, S., Shu, H., & Chen, X. (2024). Ada-MoGE: Adaptive mixture of Gaussian expert model for time series forecasting. arXiv preprint arXiv:2406.02932."},{"key":"10.1016\/j.eswa.2026.132634_b0245","unstructured":"Liu, X., Wang, S., Chu, Z., Zhang, J. Y., Jin, M., Li, Y., Pan, S., Liang, Y., & Wen, Q. (2024). TimeTracker: Mixture-of-experts-enhanced foundation time series forecasting model with decoupled training pipelines. International Conference on Learning Representations."},{"issue":"12","key":"10.1016\/j.eswa.2026.132634_b0250","doi-asserted-by":"crossref","first-page":"666","DOI":"10.1016\/j.tics.2013.09.016","article-title":"Functional connectomics from resting-state fMRI","volume":"17","author":"Smith","year":"2013","journal-title":"Trends in Cognitive Sciences"},{"issue":"11","key":"10.1016\/j.eswa.2026.132634_b0255","first-page":"3238","article-title":"An average sliding window correlation method for dynamic functional connectivity","volume":"40","author":"Vergara","year":"2019","journal-title":"Human Brain Mapping"},{"key":"10.1016\/j.eswa.2026.132634_b0260","doi-asserted-by":"crossref","unstructured":"Wu, Y., Meng, X., Hu, H., Zhang, J., Dong, Y., & Lu, D. (2025). EMTSF: Extraordinary mixture of SOTA models for time series forecasting. In: Proceedings of the Thirty-Ninth AAAI Conference on Artificial Intelligence (AAAI 2025), Vol. 39 (pp. 21599\u201321607). AAAI Press. arXiv:2510.23396. https:\/\/doi.org\/10.48550\/arXiv.2510.23396.","DOI":"10.1609\/aaai.v39i20.35463"},{"key":"10.1016\/j.eswa.2026.132634_b0265","unstructured":"Wang, Y., Cui, T., Ma, C., Wang, X., & Yao, H. (2024). DeformableTST: A novel time series forecasting model with deformable convolution and global attention mechanisms. In: Proceedings of the 38th Conference on Neural Information Processing Systems (NeurIPS 2024), Vancouver, Canada. https:\/\/neurips.cc\/virtual\/2024\/poster\/96221."},{"key":"10.1016\/j.eswa.2026.132634_b0270","unstructured":"Jin, M., Wang, S., Ma, L., Chu, Z., Zhang, J.Y., Shi, X., Chen, P.-Y., Liang, Y., Li, Y.-F., Pan, S., & Wen, Q. (2024). Time-LLM: Time series forecasting by reprogramming large language models. In: The Twelfth International Conference on Learning Representations (ICLR 2024), Vienna, Austria. arXiv:2310.01728. https:\/\/doi.org\/10.48550\/arXiv.2310.01728."},{"key":"10.1016\/j.eswa.2026.132634_b0275","unstructured":"Wang, Y., Cui, T., Ma, C., Wang, X., & Yao, H. (2025). TFB: Towards comprehensive and fair benchmarking of time series forecasting methods. arXiv:2403.20150. https:\/\/doi.org\/10.48550\/arXiv.2403.20150."}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426015472?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426015472?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,14]],"date-time":"2026-07-14T15:21:30Z","timestamp":1784042490000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426015472"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8]]},"references-count":55,"alternative-id":["S0957417426015472"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132634","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2026,8]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Expert Transformer: Fine-Grained Mixture-of-Experts for Time-Series forecasting","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.132634","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"132634"}}