{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T12:04:37Z","timestamp":1784203477226,"version":"3.55.0"},"reference-count":56,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Neural Networks"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.neunet.2026.109035","type":"journal-article","created":{"date-parts":[[2026,4,25]],"date-time":"2026-04-25T06:12:40Z","timestamp":1777097560000},"page":"109035","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["TWT: Textual white-box transformer for natural language understanding"],"prefix":"10.1016","volume":"202","author":[{"given":"Shu-Xun","family":"Yang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yanzhe","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.neunet.2026.109035_bib0001","series-title":"2009 IEEE international conference on acoustics, speech and signal processing","first-page":"693","article-title":"A fast iterative shrinkage-thresholding algorithm with application to wavelet-based image deblurring","author":"Beck","year":"2009"},{"key":"10.1016\/j.neunet.2026.109035_bib0002","series-title":"Proceedings of the 2025 conference of the nations of the americas chapter of the association for computational linguistics: Human language technologies (volume 1: Long papers)","first-page":"12119","article-title":"In-context learning with long-context models: An in-depth exploration","author":"Bertsch","year":"2025"},{"key":"10.1016\/j.neunet.2026.109035_bib0003","unstructured":"Brown, T.B., Mann, B., Ryder, N., Subbiah, M., Kaplan, J., Dhariwal, P., Neelakantan, A., Shyam, P., Sastry, G., Askell, A., Agarwal, S., Herbert-Voss, A., Krueger, G., Henighan, T., Child, R., Ramesh, A., Ziegler, D.M., Wu, J., Winter, C., Hesse, C., Chen, M., Sigler, E., Litwin, M., Gray, S., Chess, B., Clark, J., Berner, C., McCandlish, S., Radford, A., Sutskever, I., & Amodei, D. (2020). Language models are few-shot learners. https:\/\/arxiv.org\/abs\/2005.14165."},{"issue":"8","key":"10.1016\/j.neunet.2026.109035_bib0004","doi-asserted-by":"crossref","first-page":"1872","DOI":"10.1109\/TPAMI.2012.230","article-title":"Invariant scattering convolution networks","volume":"35","author":"Bruna","year":"2013","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.neunet.2026.109035_bib0005","unstructured":"Caron, M., Bojanowski, P., Joulin, A., & Douze, M. (2019). Deep clustering for unsupervised learning of visual features. https:\/\/arxiv.org\/abs\/1807.05520."},{"issue":"1","key":"10.1016\/j.neunet.2026.109035_bib0006","first-page":"4907","article-title":"ReduNet: A white-box deep network from the principle of maximizing rate reduction","volume":"23","author":"Chan","year":"2022","journal-title":"The Journal of Machine Learning Research"},{"key":"10.1016\/j.neunet.2026.109035_bib0007","first-page":"10534","article-title":"The sparse manifold transform","volume":"31","author":"Chen","year":"2018","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.neunet.2026.109035_bib0008","series-title":"Proceedings of the 15th conference of the European chapter of the association for computational linguistics: Volume 1, long papers","first-page":"1107","article-title":"Very deep convolutional networks for text classification","author":"Conneau","year":"2017"},{"key":"10.1016\/j.neunet.2026.109035_bib0009","series-title":"Proceedings of the 2019 conference of the north American chapter of the association for computational linguistics: human language technologies, volume 1 (long and short papers)","first-page":"4171","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2019"},{"key":"10.1016\/j.neunet.2026.109035_bib0010","series-title":"Proceedings of the 20th ACM SIGKDD international conference on knowledge discovery and data mining","first-page":"193","article-title":"Jointly modeling aspects, ratings and sentiments for movie recommendation (JMARS)","author":"Diao","year":"2014"},{"key":"10.1016\/j.neunet.2026.109035_bib0011","doi-asserted-by":"crossref","unstructured":"Ghoshal, S., & Al-Bustami, A. (2026). When do tools and planning help LLMs think? a cost- and latency-aware benchmark. https:\/\/arxiv.org\/abs\/2601.02663.","DOI":"10.1109\/SoutheastCon63549.2026.11475987"},{"key":"10.1016\/j.neunet.2026.109035_bib0012","unstructured":"Hallam, M.A., & Tseng, K.-K. (2026). Fusion matters: Length-aware analysis of positional-encoding fusion in transformers. https:\/\/arxiv.org\/abs\/2601.05807."},{"key":"10.1016\/j.neunet.2026.109035_bib0013","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"770","article-title":"Deep residual learning for image recognition","author":"He","year":"2016"},{"key":"10.1016\/j.neunet.2026.109035_bib0014","first-page":"6840","article-title":"Denoising diffusion probabilistic models","volume":"33","author":"Ho","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"issue":"8","key":"10.1016\/j.neunet.2026.109035_bib0015","doi-asserted-by":"crossref","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","article-title":"Long short-term memory","volume":"9","author":"Hochreiter","year":"1997","journal-title":"Neural Computation"},{"key":"10.1016\/j.neunet.2026.109035_bib0016","unstructured":"Hu, E.J., Shen, Y., Wallis, P., Allen-Zhu, Z., Li, Y., Wang, S., Wang, L., & Chen, W. (2021). LoRA: Low-rank adaptation of large language models, arXiv: 2106.09685."},{"key":"10.1016\/j.neunet.2026.109035_bib0017","series-title":"Proceedings of the 15th conference of the European chapter of the association for computational linguistics: Volume 2, short papers","first-page":"427","article-title":"Bag of tricks for efficient text classification","author":"Joulin","year":"2017"},{"key":"10.1016\/j.neunet.2026.109035_bib0018","series-title":"Proceedings of the 2014 conference on empirical methods in natural language processing (EMNLP)","first-page":"1746","article-title":"Convolutional neural networks for sentence classification","author":"Kim","year":"2014"},{"key":"10.1016\/j.neunet.2026.109035_bib0019","series-title":"3rd international conference on learning representations, ICLR 2015, San Diego, CA, USA, May 7\u20139, 2015, conference track proceedings","article-title":"Adam: A method for stochastic optimization","author":"Kingma","year":"2015"},{"key":"10.1016\/j.neunet.2026.109035_bib0020","series-title":"Proceedings of the twenty-ninth AAAI conference on artificial intelligence","first-page":"2267","article-title":"Recurrent convolutional neural networks for text classification","author":"Lai","year":"2015"},{"issue":"2","key":"10.1016\/j.neunet.2026.109035_bib0021","first-page":"167","article-title":"DBpedia\u2013a large-scale, multilingual knowledge base extracted from wikipedia","volume":"6","author":"Lehmann","year":"2015","journal-title":"Semantic Web"},{"key":"10.1016\/j.neunet.2026.109035_bib0022","unstructured":"Li, H., Wang, M., Liu, S., & Chen, P.-Y. (2023). A theoretical understanding of shallow vision transformers: Learning, generalization, and sample complexity. arXiv preprint arXiv: 2302.06015."},{"key":"10.1016\/j.neunet.2026.109035_bib0023","series-title":"Coling 2002: The 19th international conference on computational linguistics","article-title":"Learning question classifiers","author":"Li","year":"2002"},{"key":"10.1016\/j.neunet.2026.109035_bib0024","unstructured":"Liu, Y., Ott, M., Goyal, N., Du, J., Joshi, M., Chen, D., Levy, O., Lewis, M., Zettlemoyer, L., & Stoyanov, V. (2019). RoBERTa: A robustly optimized BERT pretraining approach, arXiv: 1907.11692."},{"key":"10.1016\/j.neunet.2026.109035_bib0025","unstructured":"Loshchilov, I., & Hutter, F. (2017). SGDR: Stochastic gradient descent with warm restarts, arXiv: 1608.03983."},{"issue":"9","key":"10.1016\/j.neunet.2026.109035_bib0026","doi-asserted-by":"crossref","first-page":"1546","DOI":"10.1109\/TPAMI.2007.1085","article-title":"Segmentation of multivariate mixed data via lossy data coding and compression","volume":"29","author":"Ma","year":"2007","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.neunet.2026.109035_bib0027","unstructured":"Mikolov, T., Chen, K., Corrado, G., & Dean, J. (2013). Efficient estimation of word representations in vector space. arXiv preprint arXiv: 1301.3781."},{"issue":"23","key":"10.1016\/j.neunet.2026.109035_bib0028","doi-asserted-by":"crossref","first-page":"3311","DOI":"10.1016\/S0042-6989(97)00169-7","article-title":"Sparse coding with an overcomplete basis set: A strategy employed by V1?","volume":"37","author":"Olshausen","year":"1997","journal-title":"Vision Research"},{"issue":"4","key":"10.1016\/j.neunet.2026.109035_bib0029","doi-asserted-by":"crossref","first-page":"72","DOI":"10.1109\/MSP.2018.2820224","article-title":"Theoretical foundations of deep learning via sparse representations: A multilayer sparse model and its connection to convolutional neural networks","volume":"35","author":"Papyan","year":"2018","journal-title":"IEEE Signal Processing Magazine"},{"key":"10.1016\/j.neunet.2026.109035_bib0030","series-title":"Proceedings of the 2014 conference on empirical methods in natural language processing (EMNLP)","first-page":"1532","article-title":"Glove: Global vectors for word representation","author":"Pennington","year":"2014"},{"key":"10.1016\/j.neunet.2026.109035_bib0031","unstructured":"Qwen, :., Yang, A., Yang, B., Zhang, B., Hui, B., Zheng, B., Yu, B., Li, C., Liu, D., Huang, F., Wei, H., Lin, H., Yang, J., Tu, J., Zhang, J., Yang, J., Yang, J., Zhou, J., Lin, J., Dang, K., Lu, K., Bao, K., Yang, K., Yu, L., Li, M., Xue, M., Zhang, P., Zhu, Q., Men, R., Lin, R., Li, T., Tang, T., Xia, T., Ren, X., Ren, X., Fan, Y., Su, Y., Zhang, Y., Wan, Y., Liu, Y., Cui, Z., Zhang, Z., & Qiu, Z. (2025). Qwen2.5 technical report. https:\/\/arxiv.org\/abs\/2412.15115."},{"key":"10.1016\/j.neunet.2026.109035_bib0032","unstructured":"Radford, A., Narasimhan, K., Salimans, T., Sutskever, I. et al. (2018). Improving language understanding by generative pre-training. https:\/\/api.semanticscholar.org\/CorpusID:49313245."},{"key":"10.1016\/j.neunet.2026.109035_bib0033","unstructured":"Rahman, M.W.U., Nevarez, R., Mim, L.T., & Hariri, S. (2025). SDEC: Semantic deep embedded clustering. https:\/\/arxiv.org\/abs\/2508.15823."},{"key":"10.1016\/j.neunet.2026.109035_bib0034","unstructured":"Rajeev, A., Avadhanam, U., Tulapurkar, H., & Sundar, S. (2025). Small sample-based adaptive text classification through iterative and contrastive description refinement. https:\/\/arxiv.org\/abs\/2508.00957."},{"key":"10.1016\/j.neunet.2026.109035_bib0035","series-title":"Conference on learning theory","article-title":"Exact recovery of sparsely-used dictionaries","author":"Spielman","year":"2012"},{"key":"10.1016\/j.neunet.2026.109035_bib0036","series-title":"2018 25th IEEE international conference on image processing (ICIP)","first-page":"346","article-title":"Supervised deep sparse coding networks","author":"Sun","year":"2018"},{"key":"10.1016\/j.neunet.2026.109035_bib0037","first-page":"6827","article-title":"What makes for good views for contrastive learning?","volume":"33","author":"Tian","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.neunet.2026.109035_bib0038","unstructured":"Touvron, H., Lavril, T., Izacard, G., Martinet, X., Lachaux, M.-A., Lacroix, T., Rozi\u00e8re, B., Goyal, N., Hambro, E., Azhar, F., Rodriguez, A., Joulin, A., Grave, E., & Lample, G. (2023). Llama: Open and efficient foundation language models, arXiv: 2302.13971."},{"key":"10.1016\/j.neunet.2026.109035_bib0039","first-page":"6000","article-title":"Attention is all you need","volume":"30","author":"Vaswani","year":"2017","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.neunet.2026.109035_bib0040","series-title":"2025 IEEE 10th International Workshop on Computational Advances in Multi-Sensor Adaptive Processing (CAMSAP)","first-page":"352","article-title":"Attention: Self-expression is all you need","author":"Vidal","year":"2025"},{"key":"10.1016\/j.neunet.2026.109035_bib0041","series-title":"Proceedings of the 56th annual meeting of the association for computational linguistics (volume 1: Long papers)","first-page":"2321","article-title":"Joint embedding of words and labels for text classification","author":"Wang","year":"2018"},{"key":"10.1016\/j.neunet.2026.109035_bib0042","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"16041","article-title":"Rethinking minimal sufficient representation in contrastive learning","author":"Wang","year":"2022"},{"key":"10.1016\/j.neunet.2026.109035_bib0043","series-title":"High-dimensional data analysis with low-dimensional models: Principles, computation, and applications","author":"Wright","year":"2022"},{"issue":"9","key":"10.1016\/j.neunet.2026.109035_bib0044","doi-asserted-by":"crossref","first-page":"6189","DOI":"10.1109\/TSMC.2025.3578348","article-title":"FMvPCI: A multiview fusion neural network for identifying protein complex via fuzzy clustering","volume":"55","author":"Yang","year":"2025","journal-title":"IEEE Transactions on Systems, Man, and Cybernetics: Systems"},{"issue":"8","key":"10.1016\/j.neunet.2026.109035_bib0045","doi-asserted-by":"crossref","first-page":"5730","DOI":"10.1109\/TSMC.2025.3572738","article-title":"Link-based attributed graph clustering via approximate generative bayesian learning","volume":"55","author":"Yang","year":"2025","journal-title":"IEEE Transactions on Systems, Man, and Cybernetics: Systems"},{"key":"10.1016\/j.neunet.2026.109035_bib0046","first-page":"5753","article-title":"XLNet: Generalized autoregressive pretraining for language understanding","volume":"32","author":"Yang","year":"2019","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.neunet.2026.109035_bib0047","series-title":"Proceedings of the 2016 conference of the North American chapter of the association for computational linguistics: human language technologies","first-page":"1480","article-title":"Hierarchical attention networks for document classification","author":"Yang","year":"2016"},{"key":"10.1016\/j.neunet.2026.109035_bib0048","doi-asserted-by":"crossref","unstructured":"Yu, Y., Buchanan, S., Pai, D., Chu, T., Wu, Z., Tong, S., Haeffele, B.D., & Ma, Y. (2023). White-box transformers via sparse rate reduction. arXiv preprint arXiv: 2306.01129.","DOI":"10.52202\/075280-0413"},{"key":"10.1016\/j.neunet.2026.109035_bib0049","first-page":"9422","article-title":"Learning diverse and discriminative representations via the principle of maximal coding rate reduction","volume":"33","author":"Yu","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.neunet.2026.109035_bib0050","unstructured":"Zarka, J., Thiry, L., Angles, T., & Mallat, S. (2020). Deep network classification by scattering and homotopy dictionary learning, arXiv: 1910.03561."},{"issue":"1","key":"10.1016\/j.neunet.2026.109035_bib0051","first-page":"6622","article-title":"Complete dictionary learning via l 4-norm maximization over the orthogonal group","volume":"21","author":"Zhai","year":"2020","journal-title":"The Journal of Machine Learning Research"},{"key":"10.1016\/j.neunet.2026.109035_bib0052","series-title":"Findings of the association for computational linguistics: NAACL 2024","first-page":"3881","article-title":"Sentiment analysis in the era of large language models: A reality check","author":"Zhang","year":"2024"},{"key":"10.1016\/j.neunet.2026.109035_bib0053","unstructured":"Zhang, X., Zhao, J., & LeCun, Y. (2015). Character-level convolutional networks for text classification. Advances in Neural Information Processing Systems, 28, 649\u2013657."},{"key":"10.1016\/j.neunet.2026.109035_bib0054","doi-asserted-by":"crossref","first-page":"43","DOI":"10.1007\/s13042-010-0001-0","article-title":"Understanding bag-of-words model: A statistical framework","volume":"1","author":"Zhang","year":"2010","journal-title":"International Journal of Machine Learning and Cybernetics"},{"key":"10.1016\/j.neunet.2026.109035_bib0055","series-title":"Proceedings of the 2018 conference on empirical methods in natural language processing","first-page":"3110","article-title":"Investigating capsule networks with dynamic routing for text classification","author":"Zhao","year":"2018"},{"key":"10.1016\/j.neunet.2026.109035_bib0056","unstructured":"Zhao, X., Chen, X., Liu, B., Gao, H., Zhao, Z., & Chen, Y. (2025). Who speaks for the trigger? dynamic expert routing in backdoored mixture-of-experts transformers. https:\/\/arxiv.org\/abs\/2510.13462."}],"container-title":["Neural Networks"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0893608026004958?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0893608026004958?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,16]],"date-time":"2026-07-16T11:20:45Z","timestamp":1784200845000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0893608026004958"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":56,"alternative-id":["S0893608026004958"],"URL":"https:\/\/doi.org\/10.1016\/j.neunet.2026.109035","relation":{},"ISSN":["0893-6080"],"issn-type":[{"value":"0893-6080","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"TWT: Textual white-box transformer for natural language understanding","name":"articletitle","label":"Article Title"},{"value":"Neural Networks","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.neunet.2026.109035","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"109035"}}