{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T08:51:32Z","timestamp":1781772692445,"version":"3.54.5"},"reference-count":51,"publisher":"China Science Publishing & Media Ltd.","issue":"2","content-domain":{"domain":["engine.scichina.com"],"crossmark-restriction":false},"short-container-title":["DI"],"published-print":{"date-parts":[[2026,6,1]]},"DOI":"10.3724\/2096-7004.di.2026.0249","type":"journal-article","created":{"date-parts":[[2026,5,21]],"date-time":"2026-05-21T03:25:38Z","timestamp":1779333938000},"page":"20260249","update-policy":"https:\/\/doi.org\/10.1360\/scp-crossmark-policy-page","source":"Crossref","is-referenced-by-count":0,"title":["Bidirectional Dynamic Convolutional Dense Network for Speech Emotion Recognition"],"prefix":"10.3724","volume":"8","author":[{"given":"Yungang","family":"Xiao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Feilong","family":"Bao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"2026","published-online":{"date-parts":[[2026,5,21]]},"reference":[{"key":"null","unstructured":"Li R. et al., \u201cTowards Discriminative Representation Learning for Speech Emotion Recognition,\u201d in Proc. 28th Int. Joint Conf. Artif. Intell., Jul. 2019, pp. 5060\u20135066."},{"key":"null","unstructured":"Schuller B. W., \u201cSpeech Emotion Recognition: Two Decades in a Nutshell, Benchmarks, and Ongoing Trends,\u201d Commun. ACM, vol. 61, no. 5, pp. 90\u201399, Apr. 2018."},{"key":"null","unstructured":"Zheng Z. et al., \u201cIntelligent Customer Service System Based on Speech Emotion Recognition,\u201d in Proc. 7th Int. Conf. Electron. Inf. Technol. Comput. Eng., 2024, pp. 334\u2013338."},{"key":"null","unstructured":"Nordstr\u00f6m H. and Laukka P., \u201cThe Time Course of Emotion Recognition in Speech and Music,\u201d J. Acoust. Soc. Amer., vol. 145, no. 5, p. 3058, 2019."},{"key":"null","unstructured":"Tuncer T., Dogan S., and Acharya U. R., \u201cAutomated Accurate Speech Emotion Recognition System Using Twine Shuffle Pattern and Iterative Neighborhood Component Analysis Techniques,\u201d Knowl.-Based Syst., vol. 211, p. 106547, 2021."},{"key":"null","unstructured":"Mustaqeem and Kwon S., \u201cOptimal Feature Selection Based Speech Emotion Recognition Using Two-Stream Deep Convolutional Neural Network,\u201d Int. J. Intell. Syst., vol. 36, no. 9, pp. 5116\u20135135, 2021."},{"key":"null","unstructured":"Ozer I., \u201cPseudo-Colored Rate Map Representation for Speech Emotion Recognition,\u201d Biomed. Signal Process. Control, vol. 66, p. 102502, 2021."},{"key":"null","unstructured":"Wen X.-C. et al., \u201cThe Application of Capsule Neural Network Based CNN for Speech Emotion Recognition,\u201d in Proc. 25th Int. Conf. Pattern Recognit., 2021, pp. 9356\u20139362."},{"key":"null","unstructured":"Rajamani S. T. et al., \u201cA Novel Attention-Based Gated Recurrent Unit and Its Efficacy in Speech Emotion Recognition,\u201d in Proc. IEEE Int. Conf. Acoust., Speech, Signal Process., 2021, pp. 6294\u20136298."},{"key":"null","unstructured":"Wang J. et al., \u201cSpeech Emotion Recognition With Dual-Sequence LSTM Architecture,\u201d in Proc. IEEE Int. Conf. Acoust., Speech, Signal Process., 2020, pp. 6474\u20136478."},{"key":"null","unstructured":"Zhao Z. et al., \u201cExploring Spatio-Temporal Representations by Integrating Attention-Based Bidirectional LSTM-RNNs and FCNs for Speech Emotion Recognition,\u201d in Proc. Interspeech, 2018, pp. 272\u2013276."},{"key":"null","unstructured":"Li C., Zhou A., and Yao A., \u201cOmni-Dimensional Dynamic Convolution,\u201d in Proc. 10th Int. Conf. Learn. Represent., 2022."},{"key":"null","unstructured":"Ong K. L. et al., \u201cMel-MViTv2: Enhanced Speech Emotion Recognition With Mel Spectrogram and Improved Multiscale Vision Transformers,\u201d IEEE Access, vol. 11, pp. 108571\u2013108579, 2023."},{"key":"null","unstructured":"Ahmed Md. R. et al., \u201cAn Ensemble 1D-CNN-LSTM-GRU Model With Data Augmentation for Speech Emotion Recognition,\u201d Expert Syst. Appl., vol. 218, p. 119633, 2023."},{"key":"null","unstructured":"Ye J. et al., \u201cTemporal Modeling Matters: A Novel Temporal Emotional Modeling Approach for Speech Emotion Recognition,\u201d in Proc. IEEE Int. Conf. Acoust., Speech, Signal Process., 2023, pp. 1\u20135."},{"key":"null","unstructured":"Dutt A. and Gader P., \u201cWavelet Multiresolution Analysis Based Speech Emotion Recognition System Using 1D CNN LSTM Networks,\u201d IEEE\/ACM Trans. Audio, Speech, Lang. Process., vol. 31, pp. 2043\u20132054, 2023."},{"key":"null","unstructured":"Liu Z., Kang X., and Ren F., \u201cDual-TBNet: Improving the Robustness of Speech Features via Dual-Transformer-BiLSTM for Speech Emotion Recognition,\u201d IEEE\/ACM Trans. Audio, Speech, Lang. Process., vol. 31, pp. 2193\u20132203, 2023."},{"key":"null","unstructured":"Li G. et al., \u201cMPAF-CNN: Multiperspective Aware and Fine-Grained Fusion Strategy for Speech Emotion Recognition,\u201d Appl. Acoust., vol. 214, p. 109658, 2023."},{"key":"null","unstructured":"Khan M. et al., \u201cMSER: Multimodal Speech Emotion Recognition Using Cross-Attention With Deep Fusion,\u201d Expert Syst. Appl., vol. 245, p. 122946, 2024."},{"key":"null","unstructured":"Li H. et al., \u201cA Lightweight Multi-Scale Model for Speech Emotion Recognition,\u201d IEEE Access, vol. 12, pp. 130228\u2013130240, 2024."},{"key":"null","unstructured":"Fan Y., Huang H., and Han H., \u201cHierarchical Convolutional Neural Networks With Post-Attention for Speech Emotion Recognition,\u201d Neurocomputing, vol. 615, p. 128879, 2025."},{"key":"null","unstructured":"Striletchi V., Striletchi C., and Stan A., \u201cTBDM-Net: Bidirectional Dense Networks With Gender Information for Speech Emotion Recognition,\u201d in Proc. IEEE 34th Int. Workshop Mach. Learn. Signal Process., 2024, pp. 1\u20136."},{"key":"null","unstructured":"Li H. et al., \u201cMelTrans: Mel-Spectrogram Relationship-Learning for Speech Emotion Recognition via Transformers,\u201d Sensors, vol. 24, no. 17, p. 5506, 2024."},{"key":"null","unstructured":"Shixin P. et al., \u201cAn Autoencoder-Based Feature Level Fusion for Speech Emotion Recognition,\u201d Digit. Commun. Netw., vol. 10, no. 5, pp. 1341\u20131351, 2024."},{"key":"null","unstructured":"Tang X. et al., \u201cSpeech Emotion Recognition via CNN-Transformer and Multidimensional Attention Mechanism,\u201d Speech Commun., vol. 171, p. 103242, 2025."},{"key":"null","unstructured":"Lian H. et al., \u201cAMGCN: An Adaptive Multi-Graph Convolutional Network for Speech Emotion Recognition,\u201d Speech Commun., vol. 168, p. 103184, 2025."},{"key":"null","unstructured":"P. S. S., Menon V., and Gopalan S., \u201cHybrid CNN-BiLSTM Architecture With Multiple Attention Mechanisms to Enhance Speech Emotion Recognition,\u201d Biomed. Signal Process. Control, vol. 100, p. 106967, 2025."},{"key":"null","unstructured":"Dellaert F., Polzin T., and Waibel A., \u201cRecognizing Emotion in Speech,\u201d in Proc. 4th Int. Conf. Spoken Lang. Process., vol. 3, 1996, pp. 1970\u20131973."},{"key":"null","unstructured":"Neiberg D., Elenius K., and Laskowski K., \u201cEmotion Recognition in Spontaneous Speech Using GMMs,\u201d in Proc. Interspeech, 2006, paper 1581-Tue1A3O.5."},{"key":"null","unstructured":"Nogueiras A. et al., \u201cSpeech Emotion Recognition Using Hidden Markov Models,\u201d in Proc. 7th Eur. Conf. Speech Commun. Technol., 2001, pp. 2679\u20132682."},{"key":"null","unstructured":"Milton A., Roy S. S., and Selvi S. T., \u201cSVM Scheme for Speech Emotion Recognition Using MFCC Feature,\u201d Int. J. Comput. Appl., vol. 69, no. 9, pp. 34\u201339, 2013."},{"key":"null","unstructured":"Morrison D., Wang R., and De Silva L. C., \u201cEnsemble Methods for Spoken Emotion Recognition in Call Centres,\u201d Speech Commun., vol. 49, no. 2, pp. 98\u2013112, 2007."},{"key":"null","unstructured":"Albornoz E. M., Milone D. H., and Rufiner H. L., \u201cSpoken Emotion Recognition Using Hierarchical Classifiers,\u201d Comput. Speech Lang., vol. 25, no. 3, pp. 556\u2013570, 2011."},{"key":"null","unstructured":"Lee C.-C. et al., \u201cEmotion Recognition Using a Hierarchical Binary Decision Tree Approach,\u201d Speech Commun., vol. 53, no. 9, pp. 1162\u20131171, 2011."},{"key":"null","unstructured":"Abdelhamid A. A. et al., \u201cRobust Speech Emotion Recognition Using CNN + LSTM Based on Stochastic Fractal Search Optimization Algorithm,\u201d IEEE Access, vol. 10, pp. 49265\u201349284, 2022."},{"key":"null","unstructured":"Xie Y. et al., \u201cSpeech Emotion Classification Using Attention-Based LSTM,\u201d IEEE\/ACM Trans. Audio, Speech, Lang. Process., vol. 27, no. 11, pp. 1675\u20131685, 2019."},{"key":"null","unstructured":"Bai S., Kolter J. Z., and Koltun V., \u201cAn Empirical Evaluation of Generic Convolutional and Recurrent Networks for Sequence Modeling,\u201d arXiv preprint arXiv:1803.01271, 2018."},{"key":"null","unstructured":"Zhang L. et al., \u201cSpatiotemporal Causal Convolutional Network for Forecasting Hourly PM2.5 Concentrations in Beijing, China,\u201d Comput. Geosci., vol. 155, p. 104869, 2021."},{"key":"null","unstructured":"Kakuba S., Poulose A., and Han D. S., \u201cAttention-Based Multi-Learning Approach for Speech Emotion Recognition With Dilated Convolution,\u201d IEEE Access, vol. 10, pp. 122302\u2013122313, 2022."},{"key":"null","unstructured":"Yang B., Bender G., Le Q. V., and Ngiam J., \u201cCondConv: Conditionally Parameterized Convolutions for Efficient Inference,\u201d in Adv. Neural Inf. Process. Syst., vol. 32, 2019."},{"key":"null","unstructured":"Chen Y. et al., \u201cDynamic Convolution: Attention Over Convolution Kernels,\u201d in Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit., 2020, pp. 11027\u201311036."},{"key":"null","unstructured":"Peng Z. et al., \u201cEfficient Speech Emotion Recognition Using Multi-Scale CNN and Attention,\u201d in Proc. IEEE Int. Conf. Acoust., Speech, Signal Process., 2021, pp. 3020\u20133024."},{"key":"null","unstructured":"Wan D. et al., \u201cMixed Local Channel Attention for Object Detection,\u201d Eng. Appl. Artif. Intell., vol. 123, p. 106442, 2023."},{"key":"null","unstructured":"Li M. et al., \u201cMS-SENet: Enhancing Speech Emotion Recognition Through Multi-Scale Feature Fusion With Squeeze-and-Excitation Blocks,\u201d in Proc. IEEE Int. Conf. Acoust., Speech, Signal Process., 2024, pp. 12271\u201312275."},{"key":"null","unstructured":"Wang Q. et al., \u201cECA-Net: Efficient Channel Attention for Deep Convolutional Neural Networks,\u201d in Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit., 2020, pp. 11531\u201311539."},{"key":"null","unstructured":"Wen X.-C. et al., \u201cCTL-MTNet: A Novel CapsNet and Transfer Learning-Based Mixed Task Net for the Single-Corpus and Cross-Corpus Speech Emotion Recognition,\u201d arXiv preprint arXiv:2207.10644, 2022."},{"key":"null","unstructured":"Ma Z. et al., \u201cEmoBox: Multilingual Multi-Corpus Speech Emotion Recognition Toolkit and Benchmark,\u201d arXiv preprint arXiv:2406.07162, 2024."},{"key":"null","unstructured":"S\u00f6nmez Y. \u00dc. and Varol A., \u201cA Speech Emotion Recognition Model Based on Multi-Level Local Binary and Local Ternary Patterns,\u201d IEEE Access, vol. 8, pp. 190784\u2013190796, 2020."},{"key":"null","unstructured":"Shen S. et al., \u201cTemporal Shift Module With Pretrained Representations for Speech Emotion Recognition,\u201d Intell. Comput., vol. 3, p. 0073, 2024."},{"key":"null","unstructured":"Zhu W. and Li X., \u201cSpeech Emotion Recognition With Global-Aware Fusion on Multi-Scale Feature Representation,\u201d in Proc. IEEE Int. Conf. Acoust., Speech, Signal Process., 2022, pp. 6437\u20136441."},{"key":"null","unstructured":"Aftab A. et al., \u201cLIGHT-SERNET: A Lightweight Fully Convolutional Neural Network for Speech Emotion Recognition,\u201d in Proc. IEEE Int. Conf. Acoust., Speech, Signal Process., 2022, pp. 6912\u20136916."}],"container-title":["Data Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.sciengine.com\/sci-open\/api\/v1\/open\/file\/pdf\/F8F3FB6B939547E888D72286B3004A7A","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/www.sciengine.com\/doi\/10.3724\/2096-7004.di.2026.0249","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/www.sciengine.com\/sci-open\/api\/v1\/open\/file\/pdf\/F8F3FB6B939547E888D72286B3004A7A","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T07:59:04Z","timestamp":1781769544000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.sciengine.com\/doi\/10.3724\/2096-7004.di.2026.0249"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,5,21]]},"references-count":51,"journal-issue":{"issue":"2","published-online":{"date-parts":[[2026,5,21]]},"published-print":{"date-parts":[[2026,6,1]]}},"URL":"https:\/\/doi.org\/10.3724\/2096-7004.di.2026.0249","relation":{},"ISSN":["2096-7004"],"issn-type":[{"value":"2096-7004","type":"print"}],"subject":[],"published":{"date-parts":[[2026,5,21]]}}}