{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,7]],"date-time":"2026-03-07T18:57:24Z","timestamp":1772909844878,"version":"3.50.1"},"reference-count":79,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"3","license":[{"start":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T00:00:00Z","timestamp":1772323200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T00:00:00Z","timestamp":1772323200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,3,1]],"date-time":"2026-03-01T00:00:00Z","timestamp":1772323200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100020895","name":"MIT-IBM Watson AI Lab","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100020895","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Pattern Anal. Mach. Intell."],"published-print":{"date-parts":[[2026,3]]},"DOI":"10.1109\/tpami.2025.3644853","type":"journal-article","created":{"date-parts":[[2025,12,17]],"date-time":"2025-12-17T18:46:02Z","timestamp":1765997162000},"page":"3571-3585","source":"Crossref","is-referenced-by-count":2,"title":["CMKD: CNN\/Transformer-Based Cross-Model Knowledge Distillation for Audio Classification"],"prefix":"10.1109","volume":"48","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4537-0078","authenticated-orcid":false,"given":"Yuan","family":"Gong","sequence":"first","affiliation":[{"name":"Computer Science and Artificial Intelligence Laboratory, Massachusetts Institute of Technology, Cambridge, MA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3182-1085","authenticated-orcid":false,"given":"Sameer","family":"Khurana","sequence":"additional","affiliation":[{"name":"Computer Science and Artificial Intelligence Laboratory, Massachusetts Institute of Technology, Cambridge, MA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0063-3612","authenticated-orcid":false,"given":"Andrew","family":"Rouditchenko","sequence":"additional","affiliation":[{"name":"Computer Science and Artificial Intelligence Laboratory, Massachusetts Institute of Technology, Cambridge, MA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3097-360X","authenticated-orcid":false,"given":"James","family":"Glass","sequence":"additional","affiliation":[{"name":"Computer Science and Artificial Intelligence Laboratory, Massachusetts Institute of Technology, Cambridge, MA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/78.143457"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1993.319077"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1017\/ATSIP.2014.12"},{"key":"ref4","first-page":"255","article-title":"Convolutional networks for images, speech, and time series","volume-title":"The Handbook of Brain Theory and Neural Networks","volume":"3361","author":"LeCun","year":"1995"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952132"},{"key":"ref6","article-title":"Rethinking CNN models for audio classification","author":"Palanisamy","year":"2020"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2017.2657381"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/MLSP.2015.7324337"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952651"},{"key":"ref10","article-title":"Comparison of time-frequency representations for environmental sound classification using convolutional neural networks","author":"Huzaifah","year":"2017"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2021-698"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2022-227"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746312"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1810.04805"},{"key":"ref16","article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Dosovitskiy","year":"2021"},{"key":"ref17","first-page":"10347","article-title":"Training data-efficient image transformers & distillation through attention","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Touvron","year":"2021"},{"key":"ref18","first-page":"12116","article-title":"Do vision transformers see like convolutional neural networks?","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Raghu","year":"2021"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01627"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3120633"},{"key":"ref21","first-page":"6105","article-title":"Efficientnet: Rethinking model scaling for convolutional neural networks","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Tan","year":"2019"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.243"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.2478\/eletel-2014-0042"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ICPR48806.2021.9413035"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.33682\/8axe-9243"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52688.2022.01167"},{"key":"ref27","article-title":"Patches are all you need?","author":"Trockman","year":"2022"},{"key":"ref28","article-title":"Distilling the knowledge in a neural network","volume-title":"Proc. NIPS Deep Learn. Representation Learn. Workshop","author":"Hinton","year":"2015"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01065"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2680"},{"key":"ref31","article-title":"mixup: Beyond empirical risk minimization","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Zhang","year":"2018"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3133208"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952261"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1145\/2733373.2806390"},{"key":"ref35","article-title":"A closer look at weak label learning for audio events","author":"Shah","year":"2018"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.33682\/y8xs-0463"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2020.3006378"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01070"},{"key":"ref39","first-page":"6076","article-title":"Svcca: Singular vector canonical correlation analysis for deep learning dynamics and interpretability","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Raghu","year":"2017"},{"key":"ref40","first-page":"23296","article-title":"Intriguing properties of vision transformers","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Naseer","year":"2021"},{"key":"ref41","article-title":"Transferring inductive biases through knowledge distillation","author":"Abnar","year":"2020"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746490"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2731"},{"key":"ref44","first-page":"100","article-title":"Convolution augmented transformer for semi-supervised sound event detection","volume-title":"Proc. DCASE Workshop","author":"Miyazaki","year":"2020"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.3014737"},{"key":"ref46","first-page":"876","article-title":"Averaging weights leads to wider optima and better generalization","volume-title":"Proc. Conf. Uncertainty Artif. Intell.","author":"Izmailov","year":"2018"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747669"},{"key":"ref48","article-title":"Sound classification with yamnet","year":"2024"},{"key":"ref49","article-title":"Audio transformers: Transformer architectures for large scale audio understanding. adieu convolutions","author":"Verma","year":"2021"},{"key":"ref50","article-title":"Improving sound event classification by increasing shift invariance in convolutional neural networks","author":"Fonseca","year":"2021"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.3030497"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682847"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref54","article-title":"Mobilenets: Efficient convolutional neural networks for mobile vision applications","author":"Howard","year":"2017"},{"key":"ref55","article-title":"Roberta: A robustly optimized bert pretraining approach","author":"Liu","year":"2019"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2020-3015"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/iros55552.2023.10342025"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00520"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01186"},{"key":"ref60","first-page":"30392","article-title":"Early convolutions help transformers see better","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Xiao","year":"2021"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00062"},{"key":"ref62","article-title":"Contnet: Why not use convolution and transformer at the same time?","author":"Yan","year":"2021"},{"key":"ref63","first-page":"3965","article-title":"Coatnet: Marrying convolution and attention for all data sizes","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","author":"Dai","year":"2021"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2017.2778423"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2020.3021711"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1007\/s11042-021-11610-8"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2019.2931656"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.33682\/gqpj-ac63"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1109\/ICSP56322.2022.9965274"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i12.17280"},{"key":"ref71","article-title":"Knowledge distillation from transformers for low-complexity acoustic scene classification","volume-title":"Proc. Detection Classification Acoustic Scenes Events Workshop","author":"Schmid","year":"2022"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747908"},{"key":"ref73","first-page":"173","article-title":"Iterative knowledge distillation in R-CNNs for weakly-labeled semi-supervised sound event detection","volume-title":"Proc. DCASE","author":"Koutini","year":"2018"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2023.3244507"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2835"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.23919\/Eusipco47968.2020.9287734"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2022.03.025"},{"key":"ref78","doi-asserted-by":"publisher","DOI":"10.1016\/j.dsp.2022.103446"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1337"}],"container-title":["IEEE Transactions on Pattern Analysis and Machine Intelligence"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/34\/11372200\/11301652.pdf?arnumber=11301652","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,9]],"date-time":"2026-02-09T21:05:22Z","timestamp":1770671122000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11301652\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3]]},"references-count":79,"journal-issue":{"issue":"3"},"URL":"https:\/\/doi.org\/10.1109\/tpami.2025.3644853","relation":{},"ISSN":["0162-8828","2160-9292","1939-3539"],"issn-type":[{"value":"0162-8828","type":"print"},{"value":"2160-9292","type":"electronic"},{"value":"1939-3539","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,3]]}}}