{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,22]],"date-time":"2026-07-22T16:03:24Z","timestamp":1784736204016,"version":"3.55.0"},"reference-count":59,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"3","license":[{"start":{"date-parts":[[2023,9,1]],"date-time":"2023-09-01T00:00:00Z","timestamp":1693526400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2023,9,1]],"date-time":"2023-09-01T00:00:00Z","timestamp":1693526400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,9,1]],"date-time":"2023-09-01T00:00:00Z","timestamp":1693526400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62071383"],"award-info":[{"award-number":["62071383"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Key Research and Development Plan of Shaanxi Province","award":["2021NY-036"],"award-info":[{"award-number":["2021NY-036"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Trans. Cogn. Dev. Syst."],"published-print":{"date-parts":[[2023,9]]},"DOI":"10.1109\/tcds.2022.3222350","type":"journal-article","created":{"date-parts":[[2022,11,15]],"date-time":"2022-11-15T20:43:03Z","timestamp":1668544983000},"page":"1501-1513","source":"Crossref","is-referenced-by-count":15,"title":["A Squeeze-and-Excitation and Transformer-Based Cross-Task Model for Environmental Sound Recognition"],"prefix":"10.1109","volume":"15","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9803-8212","authenticated-orcid":false,"given":"Jisheng","family":"Bai","sequence":"first","affiliation":[{"name":"Joint Laboratory of Environmental Sound Sensing, School of Marine Science and Technology, Northwestern Polytechnical University, Xi&#x2019;an, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jianfeng","family":"Chen","sequence":"additional","affiliation":[{"name":"Joint Laboratory of Environmental Sound Sensing, School of Marine Science and Technology, Northwestern Polytechnical University, Xi&#x2019;an, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6476-2501","authenticated-orcid":false,"given":"Mou","family":"Wang","sequence":"additional","affiliation":[{"name":"Joint Laboratory of Environmental Sound Sensing, School of Marine Science and Technology, Northwestern Polytechnical University, Xi&#x2019;an, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0452-8457","authenticated-orcid":false,"given":"Muhammad Saad","family":"Ayub","sequence":"additional","affiliation":[{"name":"Joint Laboratory of Environmental Sound Sensing, School of Marine Science and Technology, Northwestern Polytechnical University, Xi&#x2019;an, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0689-145X","authenticated-orcid":false,"given":"Qingli","family":"Yan","sequence":"additional","affiliation":[{"name":"School of Computer Science and Technology, Xi&#x2019;an University of Posts and Telecommunications, Xi&#x2019;an, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2007.07.018"},{"key":"ref2","first-page":"1851","article-title":"An abnormal sound detection and classification system for surveillance applications","volume-title":"Proc. 18th Eur. Signal Process. Conf.","author":"Chan"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2021.108289"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-65172-9_15"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/TCDS.2016.2552493"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/TCDS.2017.2721552"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2015.2428998"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2014.2326181"},{"key":"ref9","article-title":"SONYC-UST-V2: An urban sound tagging dataset with spatiotemporal context","author":"Cartwright","year":"2020","journal-title":"arXiv:2009.05188"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/TCDS.2022.3160168"},{"key":"ref11","article-title":"Description and discussion on DCASE 2021 challenge task2: Unsupervised anomalous sound detection for machine condition monitoring under domain shifted conditions","author":"Kawaguchi","year":"2021"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2017.2778423"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-63450-0"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2021.3090678"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2019.03.017"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1016\/j.apacoust.2020.107829"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.308"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2019.2915167"},{"key":"ref19","first-page":"3104","article-title":"Sequence to sequence learning with neural networks","volume-title":"Proc. Int. Conf. Adv. Neural Inf. Process. Syst.","author":"Sutskever"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/LGRS.2019.2900733"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747306"},{"key":"ref22","first-page":"1","article-title":"An MFCC-GMM approach for event detection and classification","volume-title":"Proc. IEEE Workshop Appl. Signal Process. Audio Acoust. (WASPAA)","author":"Vuegen"},{"issue":"5","key":"ref23","first-page":"3511","article-title":"Non-speech environmental sound classification using SVMs with a new set of features","volume":"8","author":"Uzkent","year":"2012","journal-title":"Int. J. Innov. Comput. Inf. Control"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1016\/j.apacoust.2020.107581"},{"key":"ref25","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2018","journal-title":"arXiv:1810.04805"},{"key":"ref26","first-page":"1","article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Dosovitskiy"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.3030497"},{"key":"ref28","article-title":"Cross-task learning for audio tagging, sound event detection and spatial localization: DCASE 2019 baseline systems","author":"Kong","year":"2019"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/TCDS.2021.3123979"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2020.08.069"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.3014737"},{"key":"ref32","first-page":"1","article-title":"Attention-based convolutional neural networks for acoustic scene classification","volume-title":"Proc. DCASE Workshop","author":"Ren"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00745"},{"key":"ref34","first-page":"5768","article-title":"Techniques and applications of wearable augmented reality audio","volume-title":"Proc. Audio Eng. Soc. Conv. 114","author":"Harma"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ROBOT.2007.363825"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/IWAENC.2018.8521242"},{"key":"ref37","article-title":"Integrating the data augmentation scheme with various classifiers for acoustic scene modeling","author":"Chen","year":"2019"},{"key":"ref38","article-title":"Designing acoustic scene classification models with CNN variants","author":"Suh","year":"2020"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1145\/3224204"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.33682\/8axe-9243"},{"key":"ref41","article-title":"Incorporating auxiliary data for urban sound tagging","author":"Iqbal","year":"2020"},{"key":"ref42","first-page":"81","article-title":"Description and discussion on DCASE2020 challenge Task2: Unsupervised anomalous sound detection for machine condition monitoring","volume-title":"Proc. Detection Classif. Acoustic Scenes Events Workshop (DCASE)","author":"Koizumi"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414662"},{"key":"ref44","article-title":"Anomalous sound detection using cnn-based features by self supervised learnING","author":"Morita","year":"2021"},{"key":"ref45","article-title":"Deep neural network baseline for DCASE challenge 2016","author":"Kong","year":"2016"},{"key":"ref46","article-title":"DCASE 2018 challenge surrey cross-task convolutional neural network baseline","author":"Kong","year":"2018"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1145\/3465055"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2020-3015"},{"key":"ref49","article-title":"Improved regularization of convolutional neural networks with cutout","author":"DeVries","year":"2017","journal-title":"arXiv:1708.04552"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2680"},{"key":"ref51","first-page":"1","article-title":"mixup: Beyond empirical risk minimization","volume-title":"Proc. Int. Conf. Learn. Represent.","author":"Zhang"},{"key":"ref52","first-page":"1","article-title":"FMix: Enhancing mixed sample data augmentation","volume-title":"Proc. ICLR","author":"Harris"},{"key":"ref53","article-title":"Layer normalization","author":"Ba","year":"2016","journal-title":"arXiv:1607.06450"},{"key":"ref54","first-page":"56","article-title":"Acoustic scene classification in DCASE 2020 challenge: Generalization across devices and low complexity solutions","volume-title":"Proc. Detection Classif. Acoustic Scenes Events Workshop (DCASE)","author":"Heittola"},{"key":"ref55","article-title":"Low-complexity acoustic scene classification for multi-device audio: Analysis of DCASE 2021 challenge systems","author":"Mart\u00edn-Morat\u00f3","year":"2021","journal-title":"arXiv:2105.13734"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.33682\/j5zw-2t88"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/WASPAA52581.2021.9632802"},{"key":"ref58","article-title":"ToyADMOS2: Another dataset of miniature-machine operating sounds for anomalous sound detection under domain shift conditions","author":"Harada","year":"2021","journal-title":"arXiv:2106.02369"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.semeval-1.6"}],"container-title":["IEEE Transactions on Cognitive and Developmental Systems"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/7274989\/10242607\/09951400.pdf?arnumber=9951400","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,2,1]],"date-time":"2024-02-01T03:25:55Z","timestamp":1706757955000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9951400\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,9]]},"references-count":59,"journal-issue":{"issue":"3"},"URL":"https:\/\/doi.org\/10.1109\/tcds.2022.3222350","relation":{},"ISSN":["2379-8920","2379-8939"],"issn-type":[{"value":"2379-8920","type":"print"},{"value":"2379-8939","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,9]]}}}