{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,29]],"date-time":"2025-10-29T03:46:17Z","timestamp":1761709577290,"version":"3.37.3"},"reference-count":41,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2019,2,7]],"date-time":"2019-02-07T00:00:00Z","timestamp":1549497600000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Neural Comput &amp; Applic"],"published-print":{"date-parts":[[2020,2]]},"DOI":"10.1007\/s00521-019-04053-8","type":"journal-article","created":{"date-parts":[[2019,2,7]],"date-time":"2019-02-07T14:16:33Z","timestamp":1549548993000},"page":"1051-1065","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":11,"title":["Applying visual domain style transfer and texture synthesis techniques to audio: insights and challenges"],"prefix":"10.1007","volume":"32","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7188-3600","authenticated-orcid":false,"given":"Muhammad","family":"Huzaifah bin Md Shahrin","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9200-1048","authenticated-orcid":false,"given":"Lonce","family":"Wyse","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2019,2,7]]},"reference":[{"key":"4053_CR1","unstructured":"Athineos M, Ellis D (2003) Sound texture modelling with linear prediction in both time and frequency domains. In: 2003 IEEE international conference on acoustics, speech, and signal processing (ICASSP), vol 5. IEEE, pp V\u2013648"},{"key":"4053_CR2","doi-asserted-by":"crossref","unstructured":"Beauregard GT, Harish M, Wyse L (2015) Single pass spectrogram inversion. In: 2015 IEEE international conference on digital signal processing (DSP). IEEE, pp 427\u2013431","DOI":"10.1109\/ICDSP.2015.7251907"},{"issue":"10","key":"4053_CR3","doi-asserted-by":"publisher","first-page":"693","DOI":"10.1038\/nrn3565","volume":"14","author":"JK Bizley","year":"2013","unstructured":"Bizley JK, Cohen YE (2013) The what, where and how of auditory-object perception. Nat Rev Neurosci 14(10):693","journal-title":"Nat Rev Neurosci"},{"key":"4053_CR4","unstructured":"Choi K, Fazekas G, Sandler M (2016) Explaining deep convolutional neural networks on music classification. arXiv preprint \narXiv:160702444"},{"key":"4053_CR5","unstructured":"Cui B, Qi C, Wang A (2017) Multi-style transfer: generalizing fast style transfer to several genres"},{"key":"4053_CR6","first-page":"1","volume-title":"Handbook of Texture Analysis","author":"E. R. Davies","year":"2008","unstructured":"Davies ER (2008) Handbook of texture analysis. Imperial College Press, London, UK, chap introduction to texture analysis, pp 1\u201331"},{"key":"4053_CR7","doi-asserted-by":"crossref","unstructured":"Deng J, Dong W, Socher R, Li LJ, Li K, Fei-Fei L (2009) Imagenet: a large-scale hierarchical image database. In: Computer vision and pattern recognition, 2009. IEEE, pp 248\u2013255","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"4053_CR8","doi-asserted-by":"crossref","unstructured":"Deng L, Abdel-Hamid O, Yu D (2013) A deep convolutional neural network using heterogeneous pooling for trading acoustic invariance with phonetic confusion. In: 2013 IEEE international conference on acoustics. Speech and signal processing (ICASSP). IEEE, pp 6669\u20136673","DOI":"10.1109\/ICASSP.2013.6638952"},{"key":"4053_CR9","doi-asserted-by":"crossref","unstructured":"Dieleman S, Schrauwen B (2014) End-to-end learning for music audio. In: 2014 IEEE international conference on acoustics. Speech and signal processing (ICASSP). IEEE, pp 6964\u20136968","DOI":"10.1109\/ICASSP.2014.6854950"},{"key":"4053_CR10","doi-asserted-by":"publisher","first-page":"38","DOI":"10.1109\/MCG.2002.1016697","volume":"4","author":"S Dubnov","year":"2002","unstructured":"Dubnov S, Bar-Joseph Z, El-Yaniv R, Lischinski D, Werman M (2002) Synthesizing sound textures through wavelet tree learning. IEEE Comput Graph Appl 4:38\u201348","journal-title":"IEEE Comput Graph Appl"},{"key":"4053_CR11","unstructured":"Dumoulin V, Shlens J, Kudlur M (2017) A learned representation for artistic style. In: Proceedings of ICLR"},{"key":"4053_CR12","unstructured":"Ellis D (2013) Spectrograms: constant-q (log-frequency) and conventional (linear). \nhttp:\/\/www.ee.columbia.edu\/ln\/rosa\/matlab\/sgram\/"},{"key":"4053_CR13","doi-asserted-by":"crossref","unstructured":"Gatys LA, Ecker AS, Bethge M (2015a) A neural algorithm of artistic style. arXiv preprint \narXiv:150806576","DOI":"10.1167\/16.12.326"},{"key":"4053_CR14","doi-asserted-by":"crossref","unstructured":"Gatys LA, Ecker AS, Bethge M (2015b) Texture synthesis using convolutional neural networks. In: Advances in neural information processing systems, pp 262\u2013270","DOI":"10.1109\/CVPR.2016.265"},{"key":"4053_CR15","unstructured":"Gatys LA, Bethge M, Hertzmann A, Shechtman E (2016) Preserving color in neural artistic style transfer. arXiv preprint \narXiv:160605897"},{"key":"4053_CR16","doi-asserted-by":"crossref","unstructured":"Gatys LA, Ecker AS, Bethge M, Hertzmann A, Shechtman E (2017) Controlling perceptual factors in neural style transfer. In: 2017 IEEE conference on computer vision and pattern recognition (CVPR)","DOI":"10.1109\/CVPR.2017.397"},{"issue":"2","key":"4053_CR17","doi-asserted-by":"publisher","first-page":"236","DOI":"10.1109\/TASSP.1984.1164317","volume":"32","author":"D Griffin","year":"1984","unstructured":"Griffin D, Lim J (1984) Signal estimation from modified short-time Fourier transform. IEEE Trans Acoust 32(2):236\u2013243","journal-title":"IEEE Trans Acoust"},{"key":"4053_CR18","unstructured":"Grinstein E, Duong N, Ozerov A, Perez P (2017) Audio style transfer. arXiv preprint \narXiv:171011385"},{"key":"4053_CR19","unstructured":"Hoskinson R, Pai D (2001) Manipulation and resynthesis with natural grains. In: Proceedings of the 2001 international computer music conference, ICMC"},{"key":"4053_CR20","unstructured":"Huzaifah bin Md Shahrin M (2017) Comparison of time-frequency representations for environmental sound classification using convolutional neural networks. arXiv preprint \narXiv:170607156"},{"key":"4053_CR21","unstructured":"Jing Y, Yang Y, Feng Z, Ye J, Song M (2017) Neural style transfer: a review. arXiv preprint \narXiv:170504058"},{"key":"4053_CR22","doi-asserted-by":"crossref","unstructured":"Johnson J, Alahi A, Fei-Fei L (2016) Perceptual losses for real-time style transfer and super-resolution. In: European conference on computer vision. Springer, pp 694\u2013711","DOI":"10.1007\/978-3-319-46475-6_43"},{"issue":"2","key":"4053_CR23","doi-asserted-by":"publisher","first-page":"84","DOI":"10.1109\/TIT.1962.1057698","volume":"8","author":"B Julesz","year":"1962","unstructured":"Julesz B (1962) Visual pattern discrimination. IRE Trans Inf Theory 8(2):84\u201392","journal-title":"IRE Trans Inf Theory"},{"issue":"3","key":"4053_CR24","doi-asserted-by":"publisher","first-page":"35","DOI":"10.1109\/MSP.2014.2359987","volume":"32","author":"ZH Ling","year":"2015","unstructured":"Ling ZH, Kang SY, Zen H, Senior A, Schuster M, Qian XJ, Meng HM, Deng L (2015) Deep learning for acoustic modeling in parametric speech generation: a systematic review of existing techniques and future trends. IEEE Signal Process Mag 32(3):35\u201352","journal-title":"IEEE Signal Process Mag"},{"key":"4053_CR25","doi-asserted-by":"crossref","unstructured":"Mahendran A, Vedaldi A (2015) Understanding deep image representations by inverting them. In: 2015 IEEE conference on computer vision and pattern recognition (CVPR), pp 5188\u20135196","DOI":"10.1109\/CVPR.2015.7299155"},{"issue":"5","key":"4053_CR26","doi-asserted-by":"publisher","first-page":"926","DOI":"10.1016\/j.neuron.2011.06.032","volume":"71","author":"JH McDermott","year":"2011","unstructured":"McDermott JH, Simoncelli EP (2011) Sound texture perception via statistics of the auditory periphery: evidence from sound synthesis. Neuron 71(5):926\u2013940","journal-title":"Neuron"},{"key":"4053_CR27","unstructured":"Novak R, Nikulin Y (2016) Improving the neural algorithm of artistic style. arXiv preprint \narXiv:160504603"},{"key":"4053_CR28","unstructured":"Perez A, Proctor C, Jain A (2017) Style transfer for prosodic speech. Tech. rep., Tech. Rep., Stanford University"},{"key":"4053_CR29","doi-asserted-by":"crossref","unstructured":"Piczak KJ (2015) Esc: dataset for environmental sound classification. In: Proceedings of the 23rd ACM international conference on multimedia. ACM, pp 1015\u20131018","DOI":"10.1145\/2733373.2806390"},{"issue":"3","key":"4053_CR30","doi-asserted-by":"publisher","first-page":"678","DOI":"10.1121\/1.1914584","volume":"55","author":"TC Rand","year":"1974","unstructured":"Rand TC (1974) Dichotic release from masking for speech. J Acoust Soc Am 55(3):678\u2013680","journal-title":"J Acoust Soc Am"},{"issue":"3","key":"4053_CR31","doi-asserted-by":"publisher","first-page":"279","DOI":"10.1109\/LSP.2017.2657381","volume":"24","author":"J Salamon","year":"2017","unstructured":"Salamon J, Bello JP (2017) Deep convolutional neural networks and data augmentation for environmental sound classification. IEEE Signal Process Lett 24(3):279\u2013283","journal-title":"IEEE Signal Process Lett"},{"key":"4053_CR32","unstructured":"Schwarz D, Schnell N (2010) Descriptor-based sound texture sampling. In: Sound and music computing (SMC), pp 510\u2013515"},{"key":"4053_CR33","unstructured":"Simonyan K, Zisserman A (2014) Very deep convolutional networks for large-scale image recognition. arXiv preprint \narXiv:14091556"},{"key":"4053_CR34","unstructured":"Ulyanov D, Lebedev V (2016) Audio texture synthesis and style transfer. \nhttps:\/\/dmitryulyanov.github.io\/audio-texture-synthesis-and-style-transfer\/"},{"key":"4053_CR35","unstructured":"Ulyanov D, Lebedev V, Vedaldi A, Lempitsky VS (2016a) Texture networks: Feed-forward synthesis of textures and stylized images. In: ICML, pp 1349\u20131357"},{"key":"4053_CR36","unstructured":"Ulyanov D, Vedaldi A, Lempitsky VS (2016b) Instance normalization: the missing ingredient for fast stylization. arXiv preprint \narXiv:160708022"},{"key":"4053_CR37","doi-asserted-by":"crossref","unstructured":"Ulyanov D, Vedaldi A, Lempitsky VS (2017) Improved texture networks: maximizing quality and diversity in feed-forward stylization and texture synthesis. In: 2017 IEEE conference on computer vision and pattern recognition (CVPR), vol\u00a01, p\u00a03","DOI":"10.1109\/CVPR.2017.437"},{"key":"4053_CR38","unstructured":"Ustyuzhaninov I, Brendel W, Gatys LA, Bethge M (2016) Texture synthesis using shallow convolutional networks with random filters. arXiv preprint \narXiv:160600021"},{"key":"4053_CR39","unstructured":"Verma P, Smith JO (2018) Neural style transfer for audio spectograms. arXiv preprint \narXiv:180101589"},{"key":"4053_CR40","unstructured":"Wyse L (2017) Audio spectrogram representations for processing with convolutional neural networks. In: Proceedings of the first international workshop on deep learning and music joint with IJCNN, pp 37\u201341"},{"key":"4053_CR41","unstructured":"Zeiler MD, Fergus R (2014) Visualizing and understanding convolutional networks. In: European conference on computer vision. Springer, pp 818\u2013833"}],"container-title":["Neural Computing and Applications"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s00521-019-04053-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-019-04053-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s00521-019-04053-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2020,2,6]],"date-time":"2020-02-06T19:22:27Z","timestamp":1581016947000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s00521-019-04053-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,2,7]]},"references-count":41,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2020,2]]}},"alternative-id":["4053"],"URL":"https:\/\/doi.org\/10.1007\/s00521-019-04053-8","relation":{},"ISSN":["0941-0643","1433-3058"],"issn-type":[{"type":"print","value":"0941-0643"},{"type":"electronic","value":"1433-3058"}],"subject":[],"published":{"date-parts":[[2019,2,7]]},"assertion":[{"value":"3 April 2018","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"23 January 2019","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 February 2019","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}