{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,15]],"date-time":"2025-11-15T10:36:29Z","timestamp":1763202989694,"version":"3.33.0"},"reference-count":21,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,12,3]],"date-time":"2024-12-03T00:00:00Z","timestamp":1733184000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,12,3]],"date-time":"2024-12-03T00:00:00Z","timestamp":1733184000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,12,3]]},"DOI":"10.1109\/apsipaasc63619.2025.10849053","type":"proceedings-article","created":{"date-parts":[[2025,1,27]],"date-time":"2025-01-27T18:37:05Z","timestamp":1738003025000},"page":"1-6","source":"Crossref","is-referenced-by-count":1,"title":["Enhancing Acoustic Scene Classification with Layer-wise Fine-Tuning on the SSAST Model"],"prefix":"10.1109","author":[{"given":"Shuting","family":"Hao","sequence":"first","affiliation":[{"name":"The University of Tokyo,Graduate School of Engineering,Tokyo"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Daisuke","family":"Saito","sequence":"additional","affiliation":[{"name":"The University of Tokyo,Graduate School of Engineering,Tokyo"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Nobuaki","family":"Minematsu","sequence":"additional","affiliation":[{"name":"The University of Tokyo,Graduate School of Engineering,Tokyo"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1016\/j.apacoust.2020.107502"},{"key":"ref2","doi-asserted-by":"crossref","DOI":"10.1007\/978-3-319-63450-0","volume-title":"Computational analysis of sound scenes and events, English","author":"Virtanen","year":"2018"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2023.121902"},{"key":"ref4","article-title":"Passt: Efficient training of audio transformers with patchout","author":"Kong","year":"2022","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i10.21315"},{"key":"ref6","first-page":"56","article-title":"Acoustic scene classification in dcase 2020 challenge: Generalization across devices and low complexity solutions","volume-title":"Proceedings of the Detection and Classification of Acoustic Scenes and Events 2020 Workshop","author":"Heittola"},{"article-title":"Efficient training of audio transformers with patchout","year":"2021","author":"Koutini","key":"ref7"},{"key":"ref8","article-title":"CP-JKU submission to dcase22: Distilling knowledge for low-complexity convolutional neural networks from a patchout audio transformer","author":"Schmid","year":"2022","journal-title":"DCASE2022 Challenge, Tech. Rep."},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-698"},{"article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","year":"2020","author":"Dosovitskiy","key":"ref10"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952261"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1145\/2733373.2806390"},{"issue":"4","key":"ref14","first-page":"1850","article-title":"Low-complexity acoustic scene classification for multi-device audio: Analysis of dcase 2021 challenge systems","volume":"11","author":"Mart\u00edn-Morat\u00f3","year":"2021","journal-title":"Applied Sciences"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1145\/1150402.1150464"},{"key":"ref16","article-title":"Binaryconnect: Training deep neural networks with binary weights during propagations","author":"Courbariaux","year":"2015","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-021-01453-z"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1145\/2647868.2655045"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053519"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1145\/3322240"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-63450-0"}],"event":{"name":"2024 Asia Pacific Signal and Information Processing Association Annual Summit and Conference (APSIPA ASC)","start":{"date-parts":[[2024,12,3]]},"location":"Macau, Macao","end":{"date-parts":[[2024,12,6]]}},"container-title":["2024 Asia Pacific Signal and Information Processing Association Annual Summit and Conference (APSIPA ASC)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10848542\/10848533\/10849053.pdf?arnumber=10849053","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,1,28]],"date-time":"2025-01-28T06:24:14Z","timestamp":1738045454000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10849053\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,3]]},"references-count":21,"URL":"https:\/\/doi.org\/10.1109\/apsipaasc63619.2025.10849053","relation":{},"subject":[],"published":{"date-parts":[[2024,12,3]]}}}