{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,5]],"date-time":"2026-07-05T00:18:26Z","timestamp":1783210706671,"version":"3.54.6"},"reference-count":67,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T00:00:00Z","timestamp":1785542400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100008628","name":"Ministry of Electronics and Information technology","doi-asserted-by":"publisher","award":["EE-9\/2\/2021-R"],"award-info":[{"award-number":["EE-9\/2\/2021-R"]}],"id":[{"id":"10.13039\/501100008628","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Applied Soft Computing"],"published-print":{"date-parts":[[2026,8]]},"DOI":"10.1016\/j.asoc.2026.115319","type":"journal-article","created":{"date-parts":[[2026,4,25]],"date-time":"2026-04-25T15:45:23Z","timestamp":1777131923000},"page":"115319","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Optimized lightweight CNN for speech recognition with multi-feature spectrograms"],"prefix":"10.1016","volume":"199","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-1298-6399","authenticated-orcid":false,"given":"Sridhar","family":"C.","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3674-2968","authenticated-orcid":false,"given":"Aniruddha","family":"Kanhe","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.asoc.2026.115319_bib0005","doi-asserted-by":"crossref","first-page":"79236","DOI":"10.1109\/ACCESS.2021.3084299","article-title":"A survey of speaker recognition: fundamental theories, recognition methods and opportunities","volume":"9","author":"Kabir","year":"2021","journal-title":"IEEE Access"},{"key":"10.1016\/j.asoc.2026.115319_bib0010","doi-asserted-by":"crossref","first-page":"1221","DOI":"10.1007\/s12559-024-10288-y","article-title":"Towards efficient recurrent architectures: a deep LSTM neural network applied to speech enhancement and recognition","volume":"16","author":"Wang","year":"2024","journal-title":"Cogn. Comput."},{"key":"10.1016\/j.asoc.2026.115319_bib0015","series-title":"ICASSP 2022 - 2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","first-page":"646","article-title":"Hts-at: a hierarchical token-semantic audio transformer for sound classification and detection","author":"Chen","year":"2022"},{"key":"10.1016\/j.asoc.2026.115319_bib0020","doi-asserted-by":"crossref","DOI":"10.1038\/s41598-025-28766-0","article-title":"Stacked convolutional neural network for emotion recognition using multi feature speech analysis","volume":"15","author":"Roy","year":"2025","journal-title":"Sci. Rep."},{"issue":"1","key":"10.1016\/j.asoc.2026.115319_bib0025","article-title":"Searching for effective preprocessing method and cnn-based architecture with efficient channel attention on speech emotion recognition","volume":"15","author":"Kim","year":"2025","journal-title":"Sci. Rep."},{"issue":"10","key":"10.1016\/j.asoc.2026.115319_bib0030","first-page":"10699","article-title":"Ssast: self-supervised audio spectrogram transformer","volume":"36","author":"Gong","year":"2022","journal-title":"Proc. AAAI Conf. Artif. Intell."},{"issue":"5","key":"10.1016\/j.asoc.2026.115319_bib0035","doi-asserted-by":"crossref","first-page":"67","DOI":"10.1109\/MSP.2021.3090678","article-title":"Sound event detection: a tutorial","volume":"38","author":"Mesaros","year":"2021","journal-title":"IEEE Signal Process. Mag."},{"key":"10.1016\/j.asoc.2026.115319_bib0040","doi-asserted-by":"crossref","first-page":"1720","DOI":"10.1109\/TASLP.2023.3268730","article-title":"Diffsound: discrete diffusion model for text-to-sound generation","volume":"31","author":"Yang","year":"2023","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.asoc.2026.115319_bib0045","series-title":"Interspeech 2017","article-title":"Voxceleb: a large-scale speaker identification dataset","author":"Nagrani","year":"2017"},{"key":"10.1016\/j.asoc.2026.115319_bib0050","doi-asserted-by":"crossref","first-page":"462","DOI":"10.1109\/TASLP.2022.3225649","article-title":"A time-frequency attention module for neural speech enhancement","volume":"31","author":"Zhang","year":"2023","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.asoc.2026.115319_bib0055","series-title":"Advances in Neural Information Processing Systems (NeurIPS), NeurIPS \u201920","article-title":"Wav2vec 2.0: a framework for self-supervised learning of speech representations","author":"Baevski","year":"2020"},{"key":"10.1016\/j.asoc.2026.115319_bib0060","series-title":"2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"770","article-title":"Deep residual learning for image recognition","author":"He","year":"2016"},{"key":"10.1016\/j.asoc.2026.115319_bib0065","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2024.124119","article-title":"End-to-end automated speech recognition using a character based small scale transformer architecture","volume":"252","author":"Loubser","year":"2024","journal-title":"Expert Syst. Appl."},{"key":"10.1016\/j.asoc.2026.115319_bib0070","author":"Howard"},{"key":"10.1016\/j.asoc.2026.115319_bib0075","series-title":"2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"6848","article-title":"Shufflenet: an extremely efficient convolutional neural network for mobile devices","author":"Zhang","year":"2018"},{"key":"10.1016\/j.asoc.2026.115319_bib0080","series-title":"Proc. Interspeech","first-page":"1686","article-title":"Lighthubert: lightweight and configurable speech representation learning with once-for-all hidden-unit BERT","author":"Wang","year":"2022"},{"key":"10.1016\/j.asoc.2026.115319_bib0085","series-title":"ICASSP 2020 - IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"7034","article-title":"Attention-based ASR with lightweight and dynamic convolutions","author":"Fujita","year":"2020"},{"issue":"19","key":"10.1016\/j.asoc.2026.115319_bib0090","doi-asserted-by":"crossref","first-page":"3032","DOI":"10.3390\/math12193032","article-title":"Optimizing convolutional neural network architectures","volume":"12","author":"Balderas","year":"2024","journal-title":"Mathematics"},{"issue":"14","key":"10.1016\/j.asoc.2026.115319_bib0095","doi-asserted-by":"crossref","first-page":"2862","DOI":"10.3390\/electronics13142862","article-title":"Rotating kernel CNN optimization for efficient IOT surveillance on low-power devices","volume":"13","author":"Wang","year":"2024","journal-title":"Electronics"},{"issue":"2","key":"10.1016\/j.asoc.2026.115319_bib0100","doi-asserted-by":"crossref","first-page":"257","DOI":"10.1109\/5.18626","article-title":"A tutorial on hidden markov models and selected applications in speech recognition","volume":"77","author":"Rabiner","year":"1989","journal-title":"Proc. IEEE"},{"key":"10.1016\/j.asoc.2026.115319_bib0105","series-title":"Connectionist Speech Recognition: A Hybrid Approach","author":"Bourlard","year":"1993"},{"issue":"6","key":"10.1016\/j.asoc.2026.115319_bib0110","doi-asserted-by":"crossref","first-page":"82","DOI":"10.1109\/MSP.2012.2205597","article-title":"Deep neural networks for acoustic modeling in speech recognition: the shared views of four research groups","volume":"29","author":"Hinton","year":"2012","journal-title":"IEEE Signal Process. Mag."},{"key":"10.1016\/j.asoc.2026.115319_bib0115","series-title":"IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","first-page":"4688","article-title":"Large vocabulary continuous speech recognition with context-dependent dbn-hmms","author":"Dahl","year":"2011"},{"key":"10.1016\/j.asoc.2026.115319_bib0120","series-title":"2022 21st IEEE International Conference on Machine Learning and Applications (ICMLA)","first-page":"699","article-title":"Cnn-n-gru: end-to-end speech emotion recognition from raw waveform signal using CNNS and gated recurrent unit networks","author":"Nfissi","year":"2022"},{"key":"10.1016\/j.asoc.2026.115319_bib0125","series-title":"IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","article-title":"High-accuracy and low-latency speech recognition with two-head contextual layer trajectory LSTM model","author":"Li","year":"2020"},{"key":"10.1016\/j.asoc.2026.115319_bib0130","series-title":"2017 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","first-page":"131","article-title":"CNN architectures for large-scale audio classification","author":"Hershey","year":"2017"},{"key":"10.1016\/j.asoc.2026.115319_bib0135","series-title":"2015 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","first-page":"4295","article-title":"Convolutional neural networks-based continuous speech recognition using raw speech signal","author":"Palaz","year":"2015"},{"key":"10.1016\/j.asoc.2026.115319_bib0140","series-title":"International Conference on Computer Science and Software Engineering (CSASE)","first-page":"88","article-title":"The impact of filter size and number of filters on classification accuracy in CNN","author":"Ahmed","year":"2020"},{"issue":"18","key":"10.1016\/j.asoc.2026.115319_bib0145","doi-asserted-by":"crossref","DOI":"10.3390\/s20185212","article-title":"Deep-net: a lightweight cnn-based speech emotion recognition system using deep frequency features","volume":"20","author":"Anvarjon","year":"2020","journal-title":"Sensors"},{"issue":"4","key":"10.1016\/j.asoc.2026.115319_bib0150","doi-asserted-by":"crossref","first-page":"2773","DOI":"10.1121\/10.0010257","article-title":"Lightweight deep convolutional neural network for background sound classification in speech signals","volume":"151","author":"Dayal","year":"2022","journal-title":"J. Acoust. Soc. Am."},{"key":"10.1016\/j.asoc.2026.115319_bib0155","author":"Yang"},{"key":"10.1016\/j.asoc.2026.115319_bib0160","article-title":"A lightweight eca-based DCNN approach for speech command recognition","volume":"197","author":"Karthikeyan","year":"2025","journal-title":"Comput. Biol. Med."},{"issue":"2","key":"10.1016\/j.asoc.2026.115319_bib0165","doi-asserted-by":"crossref","first-page":"27","DOI":"10.1007\/s12530-026-09797-y","article-title":"Multidimensional acoustic feature fusion with attention-guided light weighted CNN for improved speaker recognition","volume":"17","author":"Karthikeyan","year":"2026","journal-title":"Evolving Systems"},{"key":"10.1016\/j.asoc.2026.115319_bib0170","doi-asserted-by":"crossref","DOI":"10.1016\/j.compeleceng.2025.110271","article-title":"Embedded deep learning models for multilingual speech recognition","volume":"123","author":"Rahmouni","year":"2025","journal-title":"Comput. Electr. Eng."},{"issue":"8","key":"10.1016\/j.asoc.2026.115319_bib0175","doi-asserted-by":"crossref","first-page":"10731","DOI":"10.1007\/s13369-022-06649-0","article-title":"Spoken utterance classification task of arabic numerals and selected isolated words","volume":"47","author":"Dabbabi","year":"2022","journal-title":"Arab. J. Sci. Eng."},{"key":"10.1016\/j.asoc.2026.115319_bib0180","doi-asserted-by":"crossref","first-page":"23498","DOI":"10.1109\/ACCESS.2025.3536470","article-title":"Dligru-x: efficient x-vector-based embeddings for small-footprint keyword spotting system","volume":"13","author":"Wu","year":"2025","journal-title":"IEEE Access"},{"key":"10.1016\/j.asoc.2026.115319_bib0185","doi-asserted-by":"crossref","first-page":"85473","DOI":"10.1109\/ACCESS.2025.3568026","article-title":"A multi-scale deep learning framework combining mobilevit-eca and LSTM for accurate ECG analysis","volume":"13","author":"Ba Mahel","year":"2025","journal-title":"IEEE Access"},{"key":"10.1016\/j.asoc.2026.115319_bib0190","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2023.120378","article-title":"Automatic speech recognition of Portuguese phonemes using neural networks ensemble","volume":"229","author":"Nedjah","year":"2023","journal-title":"Expert Syst. Appl."},{"issue":"6","key":"10.1016\/j.asoc.2026.115319_bib0195","doi-asserted-by":"crossref","first-page":"737","DOI":"10.1007\/s42979-025-04209-5","article-title":"Optimizing retail supply chain sales forecasting with a mayfly algorithm-enhanced bidirectional gated recurrent unit","volume":"6","author":"Sajja","year":"2025","journal-title":"SN Comput. Sci."},{"issue":"3","key":"10.1016\/j.asoc.2026.115319_bib0200","doi-asserted-by":"crossref","DOI":"10.1016\/j.jnlest.2025.100322","article-title":"Xabh-cnn-gru: explainable attention-based hybrid cnn-gru model for accurate identification of common arrhythmias","volume":"23","author":"Ba Mahel","year":"2025","journal-title":"J. Electron. Sci. Technol."},{"key":"10.1016\/j.asoc.2026.115319_bib0205","series-title":"Recent Challenges in Intelligent Information and Database Systems","first-page":"221","article-title":"A lightweight transformer model for real-time object detection on edge devices in surveillance camera systems","author":"Nguyen","year":"2025"},{"key":"10.1016\/j.asoc.2026.115319_bib0210","series-title":"2024 International Joint Conference on Neural Networks (IJCNN)","first-page":"1","article-title":"Efficientasr: speech recognition network compression via attention redundancy and chunk-level ffn optimization","author":"Wang","year":"2024"},{"issue":"2","key":"10.1016\/j.asoc.2026.115319_bib0215","doi-asserted-by":"crossref","first-page":"1973","DOI":"10.1007\/s13369-022-07086-9","article-title":"Accent recognition using a spectrogram image feature-based convolutional neural network","volume":"48","author":"Cetin","year":"2023","journal-title":"Arab. J. Sci. Eng."},{"issue":"2","key":"10.1016\/j.asoc.2026.115319_bib0220","doi-asserted-by":"crossref","first-page":"52","DOI":"10.1017\/S002510030000431X","article-title":"Douglas o\u2019shaughnessy, speech communication: human and machine","volume":"20","author":"Poser","year":"1990","journal-title":"J. Int. Phon. Assoc."},{"issue":"2","key":"10.1016\/j.asoc.2026.115319_bib0225","doi-asserted-by":"crossref","first-page":"209","DOI":"10.1016\/0378-5955(87)90050-5","article-title":"Formulae describing frequency selectivity as a function of frequency and level, and their use in calculating excitation patterns","volume":"28","author":"Moore","year":"1987","journal-title":"Hear. Res."},{"issue":"2","key":"10.1016\/j.asoc.2026.115319_bib0230","doi-asserted-by":"crossref","first-page":"248","DOI":"10.1121\/1.1908630","article-title":"Subdivision of the audible frequency range into critical bands","volume":"33","author":"Zwicker","year":"1961","journal-title":"J. Acoust. Soc. Am."},{"issue":"3","key":"10.1016\/j.asoc.2026.115319_bib0235","first-page":"37","article-title":"A comparative study of performance of FPGA based mel filter bank and bark filter bank","volume":"3","author":"Ghosh","year":"2012","journal-title":"Int. J. Artif. Intell. Appl."},{"key":"10.1016\/j.asoc.2026.115319_bib0240","series-title":"2016 Conference on Advances in Signal Processing (CASP)","first-page":"191","article-title":"A novel approach for marathi numeral recognition using bark scale and discrete sine transform method","author":"Ghule","year":"2016"},{"key":"10.1016\/j.asoc.2026.115319_bib0245","series-title":"Digital Processing of Speech Signals","author":"Rabiner","year":"1978"},{"issue":"5","key":"10.1016\/j.asoc.2026.115319_bib0250","doi-asserted-by":"crossref","first-page":"8401","DOI":"10.1109\/TNNLS.2024.3415068","article-title":"Filter pruning based on information capacity and Independence","volume":"36","author":"Tang","year":"2025","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"10.1016\/j.asoc.2026.115319_bib0255","first-page":"1929","article-title":"Dropout: a simple way to prevent neural networks from overfitting","volume":"15","author":"Srivastava","year":"2014","journal-title":"J. Mach. Learn. Res."},{"key":"10.1016\/j.asoc.2026.115319_bib0260","series-title":"Free Spoken Digit Dataset (Fsdd)","author":"Jackson","year":"2016"},{"key":"10.1016\/j.asoc.2026.115319_bib0265","author":"Warden"},{"issue":"1","key":"10.1016\/j.asoc.2026.115319_bib0270","doi-asserted-by":"crossref","first-page":"418","DOI":"10.1016\/j.jfranklin.2023.11.038","article-title":"Audiomnist: exploring explainable artificial intelligence for audio analysis on a simple benchmark","volume":"361","author":"Becker","year":"2024","journal-title":"J. Frankl. Inst."},{"key":"10.1016\/j.asoc.2026.115319_bib0275","author":"Zhang"},{"key":"10.1016\/j.asoc.2026.115319_bib0280","series-title":"IEEE International Symposium on Smart Electronic Systems (iSES)","first-page":"225","article-title":"Isolated word recognition based on convolutional recurrent neural network","author":"Rajani","year":"2022"},{"key":"10.1016\/j.asoc.2026.115319_bib0285","author":"Rai"},{"issue":"1","key":"10.1016\/j.asoc.2026.115319_bib0290","doi-asserted-by":"crossref","first-page":"61","DOI":"10.1007\/s44196-024-00448-1","article-title":"Speech keyword spotting method based on swin-transformer model","volume":"17","author":"Sun","year":"2024","journal-title":"Int. J. Comput. Intell. Syst."},{"key":"10.1016\/j.asoc.2026.115319_bib0295","author":"Dosovitskiy"},{"key":"10.1016\/j.asoc.2026.115319_bib0300","author":"Gulati"},{"key":"10.1016\/j.asoc.2026.115319_bib0305","series-title":"2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition","first-page":"4510","article-title":"Mobilenetv2: inverted residuals and linear bottlenecks","author":"Sandler","year":"2018"},{"key":"10.1016\/j.asoc.2026.115319_bib0310","series-title":"2018 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"8697","article-title":"Learning transferable architectures for scalable image recognition","author":"Zoph","year":"2018"},{"key":"10.1016\/j.asoc.2026.115319_bib0315","author":"Iandola"},{"key":"10.1016\/j.asoc.2026.115319_bib0320","series-title":"Roceedings of the 36th International Conference on Machine Learning, Vol. 97 of Proceedings of Machine Learning Research","first-page":"6105","article-title":"Efficientnet: rethinking model scaling for convolutional neural networks","author":"Tan","year":"2019"},{"key":"10.1016\/j.asoc.2026.115319_bib0325","series-title":"2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)","first-page":"2261","article-title":"Densely connected convolutional networks","author":"Huang","year":"2017"},{"issue":"6","key":"10.1016\/j.asoc.2026.115319_bib0330","doi-asserted-by":"crossref","first-page":"84","DOI":"10.1145\/3065386","article-title":"Imagenet classification with deep convolutional neural networks","volume":"60","author":"Krizhevsky","year":"2017","journal-title":"Commun. ACM"},{"key":"10.1016\/j.asoc.2026.115319_bib0335","series-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR)","article-title":"Xception: deep learning with depthwise separable convolutions","author":"Chollet","year":"2017"}],"container-title":["Applied Soft Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1568494626007672?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S1568494626007672?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T23:57:27Z","timestamp":1783209447000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S1568494626007672"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8]]},"references-count":67,"alternative-id":["S1568494626007672"],"URL":"https:\/\/doi.org\/10.1016\/j.asoc.2026.115319","relation":{},"ISSN":["1568-4946"],"issn-type":[{"value":"1568-4946","type":"print"}],"subject":[],"published":{"date-parts":[[2026,8]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Optimized lightweight CNN for speech recognition with multi-feature spectrograms","name":"articletitle","label":"Article Title"},{"value":"Applied Soft Computing","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.asoc.2026.115319","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier B.V. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"115319"}}