{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,5]],"date-time":"2026-05-05T04:41:48Z","timestamp":1777956108164,"version":"3.51.4"},"reference-count":31,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2024,12,10]],"date-time":"2024-12-10T00:00:00Z","timestamp":1733788800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,12,10]],"date-time":"2024-12-10T00:00:00Z","timestamp":1733788800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"the National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["No.U2133218"],"award-info":[{"award-number":["No.U2133218"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"name":"the National Key Research and Development Program of China","award":["No.2018YFB0204304"],"award-info":[{"award-number":["No.2018YFB0204304"]}]},{"name":"the Fundamental Research Funds for the Central Universities of China","award":["No.FRF-MP-19-007"],"award-info":[{"award-number":["No.FRF-MP-19-007"]}]},{"name":"the Fundamental Research Funds for the Central Universities of China","award":["No. FRF-TP-20-065A1Z"],"award-info":[{"award-number":["No. FRF-TP-20-065A1Z"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Appl Intell"],"published-print":{"date-parts":[[2025,1]]},"DOI":"10.1007\/s10489-024-05873-5","type":"journal-article","created":{"date-parts":[[2024,12,10]],"date-time":"2024-12-10T07:24:09Z","timestamp":1733815449000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["An fMRI-based auditory decoding framework combined with convolutional neural network for predicting the semantics of real-life sounds from brain activity"],"prefix":"10.1007","volume":"55","author":[{"given":"Mingqian","family":"Zhao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0975-2316","authenticated-orcid":false,"given":"Baolin","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,12,10]]},"reference":[{"issue":"1","key":"5873_CR1","doi-asserted-by":"publisher","first-page":"630","DOI":"10.1016\/j.neuron.2018.03.044","volume":"98","author":"AJE Kell","year":"2018","unstructured":"Kell AJE, Yamins DLK, Shook EN, Norman-Haignere SV, McDermott JH (2018) A task-optimized neural network replicates human auditory behavior, predicts brain responses, and reveals a cortical processing hierarchy. Neuron 98(1):630-644(e16)","journal-title":"Neuron"},{"issue":"7","key":"5873_CR2","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pbio.2005127","volume":"16","author":"SV Norman-Haignere","year":"2018","unstructured":"Norman-Haignere SV, McDermott JH (2018) Neural responses to natural and model-matched stimuli reveal distinct computations in primary and nonprimary auditory cortex. PLoS Biol 16(7):e2005127","journal-title":"PLoS Biol"},{"issue":"7","key":"5873_CR3","doi-asserted-by":"publisher","first-page":"01179","DOI":"10.3389\/fpsyg.2017.01179","volume":"8","author":"MA Casey","year":"2017","unstructured":"Casey MA (2017) Music of the 7ths: Predicting and decoding multivoxel fmri responses with acoustic, schematic, and categorical music features. Front Psychol 8(7):01179","journal-title":"Front Psychol"},{"issue":"1","key":"5873_CR4","doi-asserted-by":"publisher","DOI":"10.1002\/brb3.1936","volume":"11","author":"T Nakai","year":"2021","unstructured":"Nakai T, Koide-Majima N, Nishimoto S (2021) Correspondence of categorical and feature-based representations of music in the human brain. Brain Behav 11(1):e01936","journal-title":"Brain Behav"},{"issue":"18","key":"5873_CR5","doi-asserted-by":"publisher","first-page":"4799","DOI":"10.1073\/pnas.1617622114","volume":"114","author":"R Santoro","year":"2017","unstructured":"Santoro R, Moerel M, De Martino F, Valente G, Ugurbil K, Yacoub E, Formisano E (2017) Reconstructing the spectrotemporal modulations of real-life sounds from fMRI response patterns. Proc Natl Acad Sci USA 114(18):4799\u20134804","journal-title":"Proc Natl Acad Sci USA"},{"key":"5873_CR6","doi-asserted-by":"crossref","unstructured":"Szegedy C, Liu W, Jia Y, Sermanet P, Reed S, Anguelov D, Erhan D, Vanhoucke V, Rabinovich A (2015) Going deeper with convolutions. In: Proceedings of the IEEE conference on computer vision and pattern recognition (CVPR), pp 1\u20139","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"5873_CR7","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE conference on computer vision and pattern recognition (CVPR), pp 770\u2013778","DOI":"10.1109\/CVPR.2016.90"},{"key":"5873_CR8","doi-asserted-by":"crossref","unstructured":"Huang G, Liu Z, Van Der Maaten L, Weinberger KQ (2017) Densely connected convolutional networks. In: Proceedings of the IEEE conference on computer vision and pattern recognition (CVPR), pp 2261\u20132269","DOI":"10.1109\/CVPR.2017.243"},{"key":"5873_CR9","doi-asserted-by":"publisher","first-page":"682","DOI":"10.1109\/LSP.2022.3150258","volume":"29","author":"B Bahmei","year":"2022","unstructured":"Bahmei B, Birmingham E, Arzanpour S (2022) CNN-RNN and Data Augmentation Using Deep Convolutional Generative Adversarial Network for Environmental Sound Classification. IEEE Signal Process Lett 29:682\u2013686","journal-title":"IEEE Signal Process Lett"},{"key":"5873_CR10","doi-asserted-by":"crossref","unstructured":"Hershey S, Chaudhuri S, Ellis DPW, Gemmeke JF, Jansen A, Moore RC, Plakal M, Platt D, Saurous RA, Seybold B, Slaney M, Weiss RJ, Wilson K (2017) CNN architectures for large-scale audio classification. In: Proceedings of the international conference on acoustics, speech and signal processing (ICASSP), pp 131\u2013135","DOI":"10.1109\/ICASSP.2017.7952132"},{"key":"5873_CR11","doi-asserted-by":"publisher","first-page":"2880","DOI":"10.1109\/TASLP.2020.3030497","volume":"28","author":"Q Kong","year":"2020","unstructured":"Kong Q, Cao Y, Iqbal T, Wang Y, Wang W, Plumbley MD (2020) Panns: Large-scale pretrained audio neural networks for audio pattern recognition. IEEE\/ACM Trans Audio Speech Lang Process 28:2880\u20132894","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"key":"5873_CR12","doi-asserted-by":"crossref","unstructured":"Gemmeke JF, Ellis DPW, Freedman D, Jansen A, Lawrence W, Moore RC, Plakal M, Ritter M (2017) Audio set: an ontology and human-labeled dataset for audio events. In: Proceedings of the international conference on acoustics, speech and signal processing (ICASSP), pp 776\u2013780","DOI":"10.1109\/ICASSP.2017.7952261"},{"key":"5873_CR13","doi-asserted-by":"publisher","first-page":"329","DOI":"10.1016\/j.neuroimage.2015.12.036","volume":"145","author":"U Guclu","year":"2017","unstructured":"Guclu U, van Gerven MAJ (2017) Increasingly complex representations of natural movies across the dorsal stream are shared between subjects. Neuroimage 145:329\u2013336","journal-title":"Neuroimage"},{"issue":"7600","key":"5873_CR14","doi-asserted-by":"publisher","first-page":"453","DOI":"10.1038\/nature17637","volume":"532","author":"AG Huth","year":"2016","unstructured":"Huth AG, de Heer WA, Griffiths TL, Theunissen FE, Gallant JL (2016) Natural speech reveals the semantic maps that tile human cerebral cortex. Nature 532(7600):453\u2013458","journal-title":"Nature"},{"key":"5873_CR15","doi-asserted-by":"publisher","first-page":"963","DOI":"10.1038\/s41467-018-03068-4","volume":"9","author":"F Pereira","year":"2018","unstructured":"Pereira F, Lou B, Pritchett B, Ritter S, Gershman SJ, Kanwisher N, Botvinick M, Fedorenko E (2018) Toward a universal decoder of linguistic meaning from brain activation. Nat Commun 9:963","journal-title":"Nat Commun"},{"key":"5873_CR16","doi-asserted-by":"publisher","first-page":"232","DOI":"10.1016\/j.neuroimage.2017.08.017","volume":"180","author":"S Nishida","year":"2018","unstructured":"Nishida S, Nishimoto S (2018) Decoding naturalistic experiences from human brain activity via distributed representations of words. Neuroimage 180:232\u2013242","journal-title":"Neuroimage"},{"key":"5873_CR17","doi-asserted-by":"publisher","first-page":"223","DOI":"10.1016\/j.neuroimage.2017.06.042","volume":"180","author":"K Vodrahalli","year":"2018","unstructured":"Vodrahalli K, Chen P-H, Liang Y, Baldassano C, Chen J, Yong E, Honey C, Hasson U, Ramadge P, Norman KA, Arora S (2018) Mapping between fmri responses to movies and their natural language annotations. Neuroimage 180:223\u2013231","journal-title":"Neuroimage"},{"key":"5873_CR18","doi-asserted-by":"crossref","unstructured":"Matsuo E, Kobayashi I, Nishimoto S, Nishida S, Asoh H (2018) Describing semantic representations of brain activity evoked by visual stimuli. In: Proceedings of the IEEE international conference on systems, man, and cybernetics (SMC), pp 576\u2013583","DOI":"10.1109\/SMC.2018.00107"},{"issue":"12","key":"5873_CR19","doi-asserted-by":"publisher","first-page":"4136","DOI":"10.1093\/cercor\/bhx268","volume":"28","author":"H Wen","year":"2018","unstructured":"Wen H, Shi J, Zhang Y, Lu K-H, Cao J, Liu Z (2018) Neural Encoding and Decoding with Deep Learning for Dynamic Natural Vision. Cereb Cortex 28(12):4136\u20134160","journal-title":"Cereb Cortex"},{"key":"5873_CR20","doi-asserted-by":"publisher","DOI":"10.3389\/fninf.2021.577451","volume":"15","author":"S Yotsutsuji","year":"2021","unstructured":"Yotsutsuji S, Lei M, Akama H (2021) Evaluation of Task fMRI Decoding With Deep Learning on a Small Sample Dataset. Front Neuroinform 15:577451","journal-title":"Front Neuroinform"},{"key":"5873_CR21","doi-asserted-by":"crossref","unstructured":"Piczak KJ (2015) ESC: Dataset for environmental sound classification. In: Proceedings of the acm international conference on multimedia (ACM), pp 1015\u20131018","DOI":"10.1145\/2733373.2806390"},{"issue":"1","key":"5873_CR22","doi-asserted-by":"publisher","first-page":"12077","DOI":"10.1038\/s41598-020-68853-y","volume":"10","author":"J Berezutskaya","year":"2020","unstructured":"Berezutskaya J, Freudenburg Z, Ambrogioni VL, Guclu U, van Gerven MAJ, Ramsey NF (2020) Cortical network responses map onto data-driven features that capture visual semantics of movie fragments. Sci Rep 10(1):12077","journal-title":"Sci Rep"},{"key":"5873_CR23","unstructured":"Zhang H, Ciss M, Dauphin YN, Lopez-Paz D (2018) mixup: Beyond empirical risk minimization. In: Proceedings of the international conference on learning representations (ICLR), pp 1\u201313"},{"issue":"6","key":"5873_CR24","doi-asserted-by":"publisher","first-page":"607","DOI":"10.1109\/34.506411","volume":"18","author":"T Hastie","year":"1996","unstructured":"Hastie T, Tibshirani R (1996) Discriminant adaptive nearest neighbor classification. IEEE Trans Pattern Anal Mach Intell 18(6):607\u2013616","journal-title":"IEEE Trans Pattern Anal Mach Intell"},{"key":"5873_CR25","volume-title":"Orthogonal Procrustes Problem","author":"LM Surhone","year":"2010","unstructured":"Surhone LM, Tennoe MT, Henssonow SF (2010) Orthogonal Procrustes Problem. Betascript Publishing, Publisher"},{"issue":"1","key":"5873_CR26","doi-asserted-by":"publisher","first-page":"80","DOI":"10.1080\/00401706.2000.10485983","volume":"42","author":"AE Hoerl","year":"2000","unstructured":"Hoerl AE, Kennard RW (2000) Ridge regression: Biased estimation for nonorthogonal problems. Technometrics 42(1):80\u201386","journal-title":"Technometrics"},{"key":"5873_CR27","first-page":"26","volume":"4","author":"LG Bruno","year":"2023","unstructured":"Bruno LG, Michele E, Giancarlo V et al (2023) Intermediate acoustic-to-semantic representations link behavioral and neural responses to natural sounds. Nat Neurosci 4:26","journal-title":"Nat Neurosci"},{"key":"5873_CR28","unstructured":"Vincent KMC, Lana O, Kazuhisa S, Kosetsu T, Masataka G, Shinichi F (2023) Decoding drums, instrumentals, vocals, and mixed sources in music using human brain activity With fMRI. In: Proceedings of the international symposium conference on music information retrieval\u00a0(ISMIR), pp 197\u2013206"},{"issue":"4","key":"5873_CR29","doi-asserted-by":"publisher","first-page":"1247","DOI":"10.1007\/s40815-023-01664-1","volume":"26","author":"MS Aslam","year":"2024","unstructured":"Aslam MS, Radhika T, Chandrasekar A et al (2024) Improved Event-Triggered-Based Output Tracking for a Class of Delayed Networked T-S Fuzzy Systems. Int J Fuzzy Syst 26(4):1247\u20131260","journal-title":"Int J Fuzzy Syst"},{"key":"5873_CR30","first-page":"007","volume":"08","author":"Y Cao","year":"2023","unstructured":"Cao Y, Chandrasekar A, Radhika T, Vijayakumar V (2023) Input-to-state stability of stochastic Markovian jump genetic regulatory networks. Math Comput Simul 08:007","journal-title":"Math Comput Simul"},{"issue":"1","key":"5873_CR31","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1038\/ncomms15037","volume":"8","author":"T Horikawa","year":"2017","unstructured":"Horikawa T, Kamitani Y (2017) Generic decoding of seen and imagined objects using hierarchical visual features. Nat Commun 8(1):1\u201315","journal-title":"Nat Commun"}],"container-title":["Applied Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-024-05873-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10489-024-05873-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10489-024-05873-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,1,20]],"date-time":"2025-01-20T15:04:35Z","timestamp":1737385475000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10489-024-05873-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,10]]},"references-count":31,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2025,1]]}},"alternative-id":["5873"],"URL":"https:\/\/doi.org\/10.1007\/s10489-024-05873-5","relation":{},"ISSN":["0924-669X","1573-7497"],"issn-type":[{"value":"0924-669X","type":"print"},{"value":"1573-7497","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,12,10]]},"assertion":[{"value":"1 November 2024","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"10 December 2024","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"This research has no potential competing interests, which encompass financial, non-financial, or other associations with individuals or organizations that could improperly impact our work.Ethical and informed consent for data usedThe data used in this study is legally obtained. The experimental procedure was approved by the local ethics committee, and prior to the experiment, all participants signed informed consent, ensuring compliance with ethical standards.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Competing interests"}}],"article-number":"118"}}