{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T20:16:44Z","timestamp":1783628204341,"version":"3.55.0"},"reference-count":50,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2027,1]]},"DOI":"10.1016\/j.eswa.2026.133551","type":"journal-article","created":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T21:58:50Z","timestamp":1783288730000},"page":"133551","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PA","title":["TAMLU: A learnable activation function that autonomously discovers architectural structure"],"prefix":"10.1016","volume":"332","author":[{"given":"Patrick Kwabena","family":"Mensah","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5105-9625","authenticated-orcid":false,"given":"Mighty Abra","family":"Ayidzoe","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.133551_b0005","unstructured":"Abai, Z., & Rajmalwar, N. (2020). DenseNet Models for Tiny ImageNet Classification (arXiv:1904.10429). arXiv. https:\/\/doi.org\/10.48550\/arXiv.1904.10429."},{"key":"10.1016\/j.eswa.2026.133551_b0010","unstructured":"Abdullah, A. N., & Aydin, T. (2025). Activator: GLU Activation Function as the Core Component of a Vision Transformer (arXiv:2405.15953). arXiv. https:\/\/doi.org\/10.48550\/arXiv.2405.15953."},{"key":"10.1016\/j.eswa.2026.133551_b0015","unstructured":"Acosta, G. V. (2021). Using precision reduction to efficiently improve mixed-precision GPUs reliability. https:\/\/lume.ufrgs.br\/handle\/10183\/224241."},{"key":"10.1016\/j.eswa.2026.133551_bib251","unstructured":"Agarap, A. F. (2018). Deep learning using rectified linear units (relu). arXiv preprint arXiv:1803.08375."},{"key":"10.1016\/j.eswa.2026.133551_b0020","unstructured":"Agostinelli, F., Hoffman, M., Sadowski, P., & Baldi, P. (2015). Learning Activation Functions to Improve Deep Neural Networks (arXiv:1412.6830). arXiv. https:\/\/doi.org\/10.48550\/arXiv.1412.6830."},{"key":"10.1016\/j.eswa.2026.133551_bib252","unstructured":"Amatriain, X., Sankar, A., Bing, J., Bodigutla, P. K., Hazen, T. J., & Kazi, M. (2023). Transformer models: an introduction and catalog. arXiv preprint arXiv:2302.07730."},{"key":"10.1016\/j.eswa.2026.133551_b0030","doi-asserted-by":"crossref","first-page":"14","DOI":"10.1016\/j.neunet.2021.01.026","article-title":"A survey on modern trainable activation functions","volume":"138","author":"Apicella","year":"2021","journal-title":"Neural Networks"},{"key":"10.1016\/j.eswa.2026.133551_b0035","unstructured":"Barron, J. T. (2021). Squareplus: A Softplus-Like Algebraic Rectifier (arXiv:2112.11687). arXiv. https:\/\/doi.org\/10.48550\/arXiv.2112.11687."},{"key":"10.1016\/j.eswa.2026.133551_b0040","unstructured":"Chakraborti, A., & Chaudhuri, B. B. (2024). DSReLU: A Novel Dynamic Slope Function for Superior Model Training (arXiv:2408.09156). arXiv. https:\/\/doi.org\/10.48550\/arXiv.2408.09156."},{"key":"10.1016\/j.eswa.2026.133551_b0045","unstructured":"Chang, J.-R., & Chen, Y.-S. (2015). Batch-normalized Maxout Network in Network (arXiv:1511.02583). arXiv. https:\/\/doi.org\/10.48550\/arXiv.1511.02583."},{"key":"10.1016\/j.eswa.2026.133551_b0050","unstructured":"Chen, Y., Dai, X., Liu, M., Chen, D., Yuan, L., & Liu, Z. (2020). Dynamic ReLU (arXiv:2003.10027). arXiv. https:\/\/doi.org\/10.48550\/arXiv.2003.10027."},{"key":"10.1016\/j.eswa.2026.133551_b0055","unstructured":"Clevert, D.-A., Unterthiner, T., & Hochreiter, S. (2016). Fast and Accurate Deep Network Learning by Exponential Linear Units (ELUs) (arXiv:1511.07289). arXiv. https:\/\/doi.org\/10.48550\/arXiv.1511.07289."},{"key":"10.1016\/j.eswa.2026.133551_b0060","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S., Uszkoreit, J., & Houlsby, N. (2021). An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale (arXiv:2010.11929). arXiv. https:\/\/doi.org\/10.48550\/arXiv.2010.11929."},{"key":"10.1016\/j.eswa.2026.133551_b0065","doi-asserted-by":"crossref","first-page":"92","DOI":"10.1016\/j.neucom.2022.06.111","article-title":"Activation functions in deep learning: a comprehensive survey and benchmark","volume":"503","author":"Dubey","year":"2022","journal-title":"Neurocomputing"},{"key":"10.1016\/j.eswa.2026.133551_b0070","doi-asserted-by":"crossref","unstructured":"Eriksson, K., Estep, D., & Johnson, C. (2004). Lipschitz Continuity. In K. Eriksson, D. Estep, & C. Johnson (Eds.), Applied Mathematics: Body and Soul: Volume 1: Derivatives and Geometry in IR3 (pp. 149\u2013164). Springer. https:\/\/doi.org\/10.1007\/978-3-662-05796-4_12.","DOI":"10.1007\/978-3-662-05796-4_12"},{"key":"10.1016\/j.eswa.2026.133551_b0200","unstructured":"Qiufeng Fan, Fanbo Hou, & Feng Shi. (2020). Bent Identity-based CNN for Image Denoising. \u6de1\u6c5f\u7406\u5de5\u5b78\u520a, 23(3). https:\/\/doi.org\/10.6180\/jase.202009_23(3).0019."},{"key":"10.1016\/j.eswa.2026.133551_b0075","unstructured":"Fernandez, A., & Mali, A. (2024). TeLU Activation Function for Fast and Stable Deep Learning (arXiv:2412.20269; Version 1). arXiv. https:\/\/doi.org\/10.48550\/arXiv.2412.20269."},{"key":"10.1016\/j.eswa.2026.133551_bib253","series-title":"International conference on machine learning","first-page":"3059","article-title":"Noisy activation functions","author":"Gulcehre","year":"2016"},{"key":"10.1016\/j.eswa.2026.133551_b0085","doi-asserted-by":"crossref","unstructured":"Gustineli, M. (2022). A survey on recently proposed activation functions for Deep Learning (arXiv:2204.02921). arXiv. https:\/\/doi.org\/10.48550\/arXiv.2204.02921.","DOI":"10.31224\/2245"},{"key":"10.1016\/j.eswa.2026.133551_bib257","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2015a). Deep Residual Learning for Image Recognition (arXiv:1512.03385). arXiv. https:\/\/doi.org\/10.48550\/arXiv.1512.03385."},{"key":"10.1016\/j.eswa.2026.133551_bib256","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2015b). Delving deep into rectifiers: Surpassing human-level performance on imagenet classification. In Proceedings of the IEEE international conference on computer vision (pp. 1026-1034).","DOI":"10.1109\/ICCV.2015.123"},{"key":"10.1016\/j.eswa.2026.133551_b0100","unstructured":"Hendrycks, D., & Gimpel, K. (2023). Gaussian Error Linear Units (GELUs) (arXiv:1606.08415). arXiv. https:\/\/doi.org\/10.48550\/arXiv.1606.08415."},{"key":"10.1016\/j.eswa.2026.133551_b0115","first-page":"1314","article-title":"Searching for MobileNetV3","volume":"2019","author":"Howard","year":"2019","journal-title":"IEEE\/CVF International Conference on Computer Vision (ICCV)"},{"key":"10.1016\/j.eswa.2026.133551_b0125","unstructured":"Huang, G., Liu, Z., Maaten, L. van der, & Weinberger, K. Q. (2018). Densely Connected Convolutional Networks (arXiv:1608.06993; Version 5). arXiv. https:\/\/doi.org\/10.48550\/arXiv.1608.06993."},{"key":"10.1016\/j.eswa.2026.133551_b0120","unstructured":"Hu, Y., Huber, A., Anumula, J., & Liu, S.-C. (2019). Overcoming the vanishing gradient problem in plain recurrent networks (arXiv:1801.06105). arXiv. https:\/\/doi.org\/10.48550\/arXiv.1801.06105."},{"key":"10.1016\/j.eswa.2026.133551_b0130","doi-asserted-by":"crossref","unstructured":"Kavun, S. (2025). Hybrid activation functions for deep neural networks: S3 and S4 -- a novel approach to gradient flow optimization (arXiv:2507.22090). arXiv. https:\/\/doi.org\/10.48550\/arXiv.2507.22090.","DOI":"10.21203\/rs.3.rs-7684932\/v1"},{"key":"10.1016\/j.eswa.2026.133551_b0135","unstructured":"Klambauer, G., Unterthiner, T., Mayr, A., & Hochreiter, S. (2017). Self-Normalizing Neural Networks (arXiv:1706.02515). arXiv. https:\/\/doi.org\/10.48550\/arXiv.1706.02515."},{"key":"10.1016\/j.eswa.2026.133551_b0105","unstructured":"Lee, Diogo Almeida, Kihyuk Sohn, W. S. (2016). Understanding and Improving Convolutional Neural Networks via Concatenated Rectified Linear Units. Proceedings of the 33 Rd International Conference on Machine Learning, New York, NY, USA, 2016. International Conference on Machine Learning."},{"key":"10.1016\/j.eswa.2026.133551_b0140","unstructured":"Liu, Q., Chen, Y., & Furber, S. (2017). Noisy Softplus: An activation function that enables SNNs to be trained as ANNs (arXiv:1706.03609). arXiv. https:\/\/doi.org\/10.48550\/arXiv.1706.03609."},{"key":"10.1016\/j.eswa.2026.133551_b0145","unstructured":"Manavazhahan, M. (2017). A Study of Activation Functions for Neural Networks."},{"key":"10.1016\/j.eswa.2026.133551_b0150","doi-asserted-by":"crossref","DOI":"10.1016\/j.dib.2023.109306","article-title":"CCMT: dataset for crop pest and disease detection","volume":"49","author":"Mensah","year":"2023","journal-title":"Data in Brief"},{"key":"10.1016\/j.eswa.2026.133551_b0155","doi-asserted-by":"crossref","unstructured":"Misra, D. (2020). Mish: A Self Regularized Non-Monotonic Activation Function (arXiv:1908.08681). arXiv. https:\/\/doi.org\/10.48550\/arXiv.1908.08681.","DOI":"10.5244\/C.34.191"},{"key":"10.1016\/j.eswa.2026.133551_b0160","unstructured":"Nadeau, C., & Bengio, Y. (1999). Inference for the Generalization Error. Advances in Neural Information Processing Systems, 12. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/1999\/hash\/7d12b66d3df6af8d429c1a357d8b9e1a-Abstract.html."},{"key":"10.1016\/j.eswa.2026.133551_b0165","unstructured":"Nader, A., & Azar, D. (2021). Evolution of Activation Functions: An Empirical Investigation (arXiv:2105.14614). arXiv. https:\/\/doi.org\/10.48550\/arXiv.2105.14614."},{"key":"10.1016\/j.eswa.2026.133551_bib254","first-page":"807","author":"Nair","year":"2010","journal-title":"Rectified linear units improve restricted boltzmann machines (ICML-10)"},{"issue":"1","key":"10.1016\/j.eswa.2026.133551_b0175","doi-asserted-by":"crossref","first-page":"69","DOI":"10.1016\/S0020-0255(96)00200-9","article-title":"The generalized sigmoid activation function: Competitive supervised learning","volume":"99","author":"Narayan","year":"1997","journal-title":"Information Sciences"},{"key":"10.1016\/j.eswa.2026.133551_b0180","doi-asserted-by":"crossref","unstructured":"Naveen, P. (2021). Phish: A Novel Hyper-Optimizable Activation Function.","DOI":"10.36227\/techrxiv.17283824.v1"},{"key":"10.1016\/j.eswa.2026.133551_b0185","unstructured":"Nwankpa, C., Ijomah, W., Gachagan, A., & Marshall, S. (2018). Activation Functions: Comparison of trends in Practice and Research for Deep Learning (arXiv:1811.03378). arXiv. https:\/\/doi.org\/10.48550\/arXiv.1811.03378."},{"key":"10.1016\/j.eswa.2026.133551_b0190","doi-asserted-by":"crossref","unstructured":"Pydimarry, S. A., Khairnar, S. M., Palacios, S. G., Sankaranarayanan, G., Hoagland, D., Nepomnayshy, D., & Nguyen, H. P. (2024). Evaluating Model Performance with Hard-Swish Activation Function Adjustments (arXiv:2410.06879; Version 1). arXiv. https:\/\/doi.org\/10.48550\/arXiv.2410.06879.","DOI":"10.1101\/2024.12.18.24319237"},{"key":"10.1016\/j.eswa.2026.133551_b0195","doi-asserted-by":"crossref","unstructured":"Qiu, S., Xu, X., & Cai, B. (2018). FReLU: Flexible Rectified Linear Units for Improving Convolutional Neural Networks (arXiv:1706.08098). arXiv. https:\/\/doi.org\/10.48550\/arXiv.1706.08098.","DOI":"10.1109\/ICPR.2018.8546022"},{"key":"10.1016\/j.eswa.2026.133551_b0205","unstructured":"Ramachandran, P., Zoph, B., & Le, Q. V. (2017). Swish: A Self-Gated Activation Function (arXiv:1710.05941; Version 1). arXiv. https:\/\/doi.org\/10.48550\/arXiv.1710.05941."},{"key":"10.1016\/j.eswa.2026.133551_b0210","unstructured":"Salimans, T., & Kingma, D. P. (2016). Weight Normalization: A Simple Reparameterization to Accelerate Training of Deep Neural Networks. Advances in Neural Information Processing Systems, 29. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2016\/hash\/ed265bc903a5a097f61d3ec064d96d2e-Abstract.html."},{"key":"10.1016\/j.eswa.2026.133551_b0215","unstructured":"Sarkar, G., Gala, J., & Tripathi, S. (2025). SG-Blend: Learning an Interpolation Between Improved Swish and GELU for Robust Neural Representations (arXiv:2505.23942). arXiv. https:\/\/doi.org\/10.48550\/arXiv.2505.23942."},{"key":"10.1016\/j.eswa.2026.133551_b0220","unstructured":"Simonyan, K., & Zisserman, A. (2015). Very Deep Convolutional Networks for Large-Scale Image Recognition (arXiv:1409.1556). arXiv. https:\/\/doi.org\/10.48550\/arXiv.1409.1556."},{"key":"10.1016\/j.eswa.2026.133551_b0225","unstructured":"Sitzmann, V., Martel, J. N. P., Bergman, A. W., Lindell, D. B., & Wetzstein, G. (2020). Implicit Neural Representations with Periodic Activation Functions (arXiv:2006.09661). arXiv. https:\/\/doi.org\/10.48550\/arXiv.2006.09661."},{"key":"10.1016\/j.eswa.2026.133551_b0230","unstructured":"Tan, M., & Le, Q. (2019). EfficientNet: Rethinking Model Scaling for Convolutional Neural Networks. Proceedings of the 36th International Conference on Machine Learning, 6105\u20136114. https:\/\/proceedings.mlr.press\/v97\/tan19a.html."},{"key":"10.1016\/j.eswa.2026.133551_b0235","first-page":"7537","article-title":"Fourier features let networks learn high frequency functions in low dimensional domains","volume":"33","author":"Tancik","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.133551_b0240","unstructured":"Xiao, H., Rasul, K., & Vollgraf, R. (2017). Fashion-MNIST: A Novel Image Dataset for Benchmarking Machine Learning Algorithms (arXiv:1708.07747). arXiv. https:\/\/doi.org\/10.48550\/arXiv.1708.07747."},{"key":"10.1016\/j.eswa.2026.133551_b0245","unstructured":"Xu, B., Wang, N., Chen, T., & Li, M. (2015). Empirical Evaluation of Rectified Activations in Convolutional Network (arXiv:1505.00853). arXiv. https:\/\/doi.org\/10.48550\/arXiv.1505.00853."},{"key":"10.1016\/j.eswa.2026.133551_b0250","unstructured":"Zhou, Y., Li, D., Huo, S., & Kung, S.-Y. (2020). Soft-Root-Sign Activation Function (arXiv:2003.00547). arXiv. https:\/\/doi.org\/10.48550\/arXiv.2003.00547."}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426024607?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426024607?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T19:23:15Z","timestamp":1783624995000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426024607"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2027,1]]},"references-count":50,"alternative-id":["S0957417426024607"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.133551","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2027,1]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"TAMLU: A learnable activation function that autonomously discovers architectural structure","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.133551","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"133551"}}