{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,9]],"date-time":"2026-07-09T15:23:40Z","timestamp":1783610620594,"version":"3.55.0"},"publisher-location":"Singapore","reference-count":33,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819947416","type":"print"},{"value":"9789819947423","type":"electronic"}],"license":[{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023]]},"DOI":"10.1007\/978-981-99-4742-3_23","type":"book-chapter","created":{"date-parts":[[2023,7,30]],"date-time":"2023-07-30T00:02:38Z","timestamp":1690675358000},"page":"283-295","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":8,"title":["Data Augmentation for Environmental Sound Classification Using Diffusion Probabilistic Model with Top-K Selection Discriminator"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8134-2314","authenticated-orcid":false,"given":"Yunhao","family":"Chen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zihui","family":"Yan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yunjie","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhen","family":"Ren","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jianlu","family":"Shen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yifan","family":"Huang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,7,30]]},"reference":[{"key":"23_CR1","first-page":"6840","volume":"33","author":"J Ho","year":"2020","unstructured":"Ho, J., et al.: Denoising diffusion probabilistic models. Adv. Neural. Inf. Process. Syst. 33, 6840\u20136851 (2020)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"23_CR2","doi-asserted-by":"publisher","first-page":"279","DOI":"10.1109\/LSP.2017.2657381","volume":"24","author":"J Salamon","year":"2016","unstructured":"Salamon, J., Bello, J.P.: Deep convolutional neural networks and data augmentation for environmental sound classification. IEEE Sig. Process. Lett. 24, 279\u2013283 (2016)","journal-title":"IEEE Sig. Process. Lett."},{"key":"23_CR3","doi-asserted-by":"crossref","unstructured":"Gong, Y., et al.: AST: Audio Spectrogram Transformer. ArXiv\u00a0abs\/2104.01778 (2021)","DOI":"10.21437\/Interspeech.2021-698"},{"key":"23_CR4","doi-asserted-by":"crossref","unstructured":"Bahmei, B., et al.: CNN-RNN and data augmentation using deep convolutional generative adversarial network for environmental sound classification. IEEE Sign. Process. Lett. 29, 682\u2013686 (2022)","DOI":"10.1109\/LSP.2022.3150258"},{"key":"23_CR5","doi-asserted-by":"crossref","unstructured":"Hershey, S., et al.: CNN architectures for large-scale audio classification. In: IEEE International Conference on Acoustics, Speech and Signal Processing, pp. 131\u2013135 (2016)","DOI":"10.1109\/ICASSP.2017.7952132"},{"key":"23_CR6","doi-asserted-by":"crossref","unstructured":"Zhu, X., et al.: Emotion classification with data augmentation using generative adversarial networks. In: Pacific-Asia Conference on Knowledge Discovery and Data Mining, pp. 349\u2013360 (2018)","DOI":"10.1007\/978-3-319-93040-4_28"},{"key":"23_CR7","unstructured":"Arjovsky, M., et al.: Wasserstein GAN. ArXiv abs\/1701.07875 (2017)"},{"key":"23_CR8","unstructured":"Zhao, H., et al.: Bias and generalization in deep generative models: an empirical study. Neural Inf. Process. Syst. 13 (2018)"},{"key":"23_CR9","unstructured":"Ho, J., et al.: Denoising Diffusion Probabilistic Models. ArXiv abs\/2006.11239 (2020)"},{"key":"23_CR10","unstructured":"Dhariwal, P., Nichol, A.: Diffusion Models Beat GANs on Image Synthesis. ArXiv abs\/2105.05233 (2021)"},{"key":"23_CR11","unstructured":"M\u00fcller-Franzes, G., et al.: Diffusion Probabilistic Models beat GANs on Medical Images. ArXiv abs\/2212.07501 (2022)"},{"key":"23_CR12","unstructured":"Maz\u2019e, F., Ahmed, F.: Diffusion Models Beat GANs on Topology Optimization (2022)"},{"key":"23_CR13","unstructured":"Song, J., et al.: Denoising Diffusion Implicit Models. ArXiv abs\/2010.02502 (2020)"},{"key":"23_CR14","unstructured":"Cheng, L., et al.: DPM-Solver++: Fast Solver for Guided Sampling of Diffusion Probabilistic Models. ArXiv abs\/2211.01095 (2022)"},{"key":"23_CR15","doi-asserted-by":"crossref","unstructured":"Cordts, M., et al.: The cityscapes dataset for semantic urban scene understanding. In: 2016 IEEE Conference on Computer Vision and Pattern Recognition, pp. 3213\u20133223 (2016)","DOI":"10.1109\/CVPR.2016.350"},{"key":"23_CR16","unstructured":"Dickstein, S., Narain, J., et al.: Deep Unsupervised Learning using Nonequilibrium Thermodynamics. ArXiv abs\/1503.03585 (2015)"},{"key":"23_CR17","doi-asserted-by":"crossref","unstructured":"Saharia, C., et al.: Photorealistic Text-to-Image Diffusion Models with Deep Language Understanding. ArXiv abs\/2205.11487 (2022)","DOI":"10.1145\/3528233.3530757"},{"key":"23_CR18","doi-asserted-by":"crossref","unstructured":"Font, F., et al.: Freesound technical demo. In: Proceedings of the 21st ACM International Conference on Multimedia, pp. 411\u2013412 (2013)","DOI":"10.1145\/2502081.2502245"},{"key":"23_CR19","doi-asserted-by":"crossref","unstructured":"He, K., et al.: Deep residual learning for image recognition. In: 2016 IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2015)","DOI":"10.1109\/CVPR.2016.90"},{"key":"23_CR20","unstructured":"lucidrains.2023.denoising-diffusion-pytorch (2023). https:\/\/github.com\/lucidrains\/denoising-diffusion-pytorch"},{"key":"23_CR21","doi-asserted-by":"crossref","unstructured":"Ronneberger, O., et al.: U-Net: Convolutional Networks for Biomedical Image Segmentation. ArXiv abs\/1505.04597 (2015)","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"23_CR22","doi-asserted-by":"crossref","unstructured":"Iwana, B.K., Uchida, S.: An empirical survey of data augmentation for time series classification with neural networks. Plos One 16 (2020)","DOI":"10.1371\/journal.pone.0254841"},{"key":"23_CR23","unstructured":"rw2019timm, Ross Wightman, PyTorch Image Models (2019). https:\/\/github.com\/rwightman\/pytorch-image-models"},{"key":"23_CR24","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. In: International Conference on Learning Representations (2017)"},{"key":"23_CR25","unstructured":"Kingma, D.P., Welling, M.: Auto-Encoding Variational Bayes. ArXiv. \/abs\/1312.6114 (2013). Accessed 22 March 2023"},{"key":"23_CR26","unstructured":"Ho, J.: Classifier-Free Diffusion Guidance. ArXiv abs\/2207.12598 (2022)"},{"key":"23_CR27","doi-asserted-by":"crossref","unstructured":"Chollet, F.: Xception: deep learning with depthwise separable convolutions. In: 2017 IEEE Conference on Computer Vision and Pattern Recognition, pp. 1251-1258 (2016)","DOI":"10.1109\/CVPR.2017.195"},{"key":"23_CR28","doi-asserted-by":"crossref","unstructured":"d\u2019Ascoli, S., et al.: ConViT: improving vision transformers with soft convolutional inductive biases. J. Statist. Mech. Theory Experiment 2022 (2021)","DOI":"10.1088\/1742-5468\/ac9830"},{"key":"23_CR29","unstructured":"Mehta, S., Rastegari, M.: MobileViT: Light-weight, General purpose, and Mobile-friendly Vision Transformer. ArXiv abs\/2110.02178 (2021)"},{"key":"23_CR30","doi-asserted-by":"crossref","unstructured":"Liu, Z., et al.: A ConvNet for the 2020s. In: 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11966\u201311976 1800\u20131807 (2022)","DOI":"10.1109\/CVPR52688.2022.01167"},{"key":"23_CR31","doi-asserted-by":"crossref","unstructured":"Touvron, H., et al.: DeiT III: Revenge of the ViT. ArXiv abs\/2204.07118 (2022)","DOI":"10.1007\/978-3-031-20053-3_30"},{"key":"23_CR32","unstructured":".von-platen-etal-2022-diffusers, Patrick von Platen et al. 2022, Diffusers: State-of-the-art diffusion models. https:\/\/github.com\/huggingface\/diffusers"},{"key":"23_CR33","unstructured":"Chen, Y., et al.: Effective audio classification network based on paired inverse pyramid structure and dense MLP Block. ArXiv abs\/2211.02940 (2022)"}],"container-title":["Lecture Notes in Computer Science","Advanced Intelligent Computing Technology and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-99-4742-3_23","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,8,1]],"date-time":"2023-08-01T23:26:07Z","timestamp":1690932367000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-99-4742-3_23"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023]]},"ISBN":["9789819947416","9789819947423"],"references-count":33,"URL":"https:\/\/doi.org\/10.1007\/978-981-99-4742-3_23","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023]]},"assertion":[{"value":"30 July 2023","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICIC","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Intelligent Computing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Zhengzhou","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2023","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"10 August 2023","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"13 August 2023","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"19","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icic2023a","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.ic-icc.cn\/2023\/index.htm","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}