{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,18]],"date-time":"2026-03-18T05:52:58Z","timestamp":1773813178013,"version":"3.50.1"},"reference-count":29,"publisher":"Springer Science and Business Media LLC","issue":"21-22","license":[{"start":{"date-parts":[[2020,4,6]],"date-time":"2020-04-06T00:00:00Z","timestamp":1586131200000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2020,4,6]],"date-time":"2020-04-06T00:00:00Z","timestamp":1586131200000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"published-print":{"date-parts":[[2020,6]]},"DOI":"10.1007\/s11042-020-08798-6","type":"journal-article","created":{"date-parts":[[2020,4,6]],"date-time":"2020-04-06T13:02:41Z","timestamp":1586178161000},"page":"15043-15057","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":9,"title":["Audio style transfer using shallow convolutional networks and random filters"],"prefix":"10.1007","volume":"79","author":[{"given":"Jiyou","family":"Chen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Gaobo","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Huihuang","family":"Zhao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Manimaran","family":"Ramasamy","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2020,4,6]]},"reference":[{"key":"8798_CR1","unstructured":"Aytar Y, Vondrick C, Torralba A (2016) Soundnet: Learning sound representations from unlabeled video[C]. Advances in Neural Information Processing Systems:892\u2013900"},{"key":"8798_CR2","unstructured":"Shaun Barry and Youngmoo Kim, Style transfer for musical audio using multiple time-frequency representations, Unpublished article available at: https:\/\/tinyurl.com\/y7nu7r9s, 2018."},{"key":"8798_CR3","unstructured":"Brunner G, Konrad A, Wang Y, et al. MIDI-VAE: Modeling dynamics and instrumentation of music with applications to style transfer[J]. arXiv preprint arXiv:1809.07600, 2018."},{"key":"8798_CR4","doi-asserted-by":"publisher","unstructured":"Ephrat A, Mosseri I, Lang O et al (2018) Looking to listen at the cocktail party: A speaker-independent audio-visual model for speech separation. ACM T Graphic. https:\/\/doi.org\/10.1145\/3197517.3201357","DOI":"10.1145\/3197517.3201357"},{"key":"8798_CR5","unstructured":"Gatys L A, Ecker A S, Bethge M. A neural algorithm of artistic style[J]. arXiv preprint arXiv:1508.06576, 2015."},{"key":"8798_CR6","doi-asserted-by":"crossref","unstructured":"Gatys LA, Ecker AS, Bethge M (2016) Image style transfer using convolutional neural networks[C]. Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition:2414\u20132423","DOI":"10.1109\/CVPR.2016.265"},{"key":"8798_CR7","unstructured":"Giurgiutiu V, Yu L (2003) Comparison of short-time Fourier transform and wavelet transform of transient and tone burst wave propagation signals for structural health monitoring[C]. Proceedings of 4th International Workshop on Structural Health Monitoring:1267\u20131274"},{"issue":"2","key":"8798_CR8","doi-asserted-by":"publisher","first-page":"236","DOI":"10.1109\/TASSP.1984.1164317","volume":"32","author":"D Griffin","year":"1984","unstructured":"Griffin D, Lim J (1984) Signal estimation from modified short-time Fourier transform[J]. IEEE Trans Acoust Speech Signal Process 32(2):236\u2013243","journal-title":"IEEE Trans Acoust Speech Signal Process"},{"key":"8798_CR9","unstructured":"Grinstein E, Duong NQK, Ozerov A et al (2018) Audio style transfer[C]\/\/2018 IEEE international conference on acoustics, speech and signal processing (ICASSP). IEEE:586\u2013590"},{"key":"8798_CR10","unstructured":"He K, Wang Y, Hopcroft J (2016) A powerful generative model using random weights for the deep image representation[C]. Advances in Neural Information Processing Systems:631\u2013639"},{"key":"8798_CR11","unstructured":"Krizhevsky A, Sutskever I, Hinton GE (2012) ImageNet classification with deep convolutional neural networks[C]. Advances in Neural Information Processing Systems:1097\u20131105"},{"key":"8798_CR12","doi-asserted-by":"publisher","first-page":"368","DOI":"10.1007\/s11036-017-0932-8","volume":"23","author":"H Lu","year":"2018","unstructured":"Lu H, Li Y, Chen M, Kim H, Serikawa S (2018) Brain intelligence: go beyond artificial intelligence. Mobile Networks and Applications 23:368\u2013375","journal-title":"Mobile Networks and Applications"},{"issue":"17","key":"8798_CR13","doi-asserted-by":"publisher","first-page":"21847","DOI":"10.1007\/s11042-017-4585-1","volume":"77","author":"H Lu","year":"2018","unstructured":"Lu H, Li Y, Uemura T et al (2018) FDCNet: filtering deep convolutional network for marine organism classification[J]. Multimed Tools Appl 77(17):21847\u201321860","journal-title":"Multimed Tools Appl"},{"key":"8798_CR14","doi-asserted-by":"publisher","first-page":"142","DOI":"10.1016\/j.future.2018.01.001","volume":"82","author":"H Lu","year":"2018","unstructured":"Lu H, Li Y, Uemura T, Kim H, Serikawa S (2018) Low illumination underwater light field images reconstruction using deep convolutional neural networks. Futur Gener Comput Syst 82:142\u2013148","journal-title":"Futur Gener Comput Syst"},{"issue":"4","key":"8798_CR15","doi-asserted-by":"publisher","first-page":"2315","DOI":"10.1109\/JIOT.2017.2737479","volume":"5","author":"H Lu","year":"2018","unstructured":"Lu H, Li Y, Mu S, Wang D, Kim H, Serikawa S (2018) Motor anomaly detection for unmanned aerial vehicles using reinforcement learning. IEEE Internet Things J 5(4):2315\u20132322","journal-title":"IEEE Internet Things J"},{"issue":"3","key":"8798_CR16","doi-asserted-by":"publisher","first-page":"90","DOI":"10.1109\/MWC.2019.1800325","volume":"26","author":"H Lu","year":"2019","unstructured":"Lu H, Wang D, Li Y et al (2019) CONet: a cognitive ocean network[J]. IEEE Wireless Communications 26(3):90\u201396","journal-title":"IEEE Wireless Communications"},{"key":"8798_CR17","unstructured":"Mital P K. Time domain neural audio style transfer[J]. arXiv preprint arXiv:1711.11160, 2017."},{"issue":"2","key":"8798_CR18","doi-asserted-by":"publisher","first-page":"286","DOI":"10.2307\/1969529","volume":"54","author":"J Nash","year":"1951","unstructured":"Nash J (1951) Non-cooperative games[J]. Annals of Mathematics (Second Series) 54(2):286\u2013295","journal-title":"Annals of Mathematics (Second Series)"},{"issue":"6","key":"8798_CR19","doi-asserted-by":"publisher","first-page":"200","DOI":"10.1145\/2508363.2508419","volume":"32","author":"Y Shih","year":"2013","unstructured":"Shih Y, Paris S, Durand F et al (2013) Data-driven hallucination of different times of day from a single outdoor photo[J]. ACM Transactions on Graphics (TOG) 32(6):200","journal-title":"ACM Transactions on Graphics (TOG)"},{"key":"8798_CR20","unstructured":"Simonyan K, Zisserman A. Very deep convolutional networks for large-scale image recognition[J]. arXiv preprint arXiv:1409.1556, 2014."},{"key":"8798_CR21","unstructured":"Simonyan K, Zisserman A. Very deep convolutional networks for large-scale image recognition[J]. arXiv preprint arXiv:1409.1556, 2014."},{"key":"8798_CR22","unstructured":"Ulyanov D, Lebedev V. Audio texture synthesis and style transfer[J]. URL https:\/\/dmitryulyanov.github.io\/audio-texture-synthesis-and-style-transfer, 2016."},{"key":"8798_CR23","unstructured":"Ustyuzhaninov I, Brendel W, Gatys L A, et al. Texture synthesis using shallow convolutional networks with random filters[J]. arXiv preprint arXiv:1606.00021, 2016."},{"key":"8798_CR24","unstructured":"Verma P, Smith J O. Neural style transfer for audio spectrograms[J]. arXiv preprint arXiv:1801.01589, 2018."},{"key":"8798_CR25","unstructured":"Wyse L. Audio spectrogram representations for processing with convolutional neural networks[J]. arXiv preprint arXiv:1706.09559, 2017."},{"key":"8798_CR26","doi-asserted-by":"publisher","first-page":"191","DOI":"10.1016\/j.neucom.2015.11.133","volume":"213","author":"X Xu","year":"2016","unstructured":"Xu X, He L, Shimada A et al (2016) Learning unified binary codes for cross-modal retrieval via latent semantic hashing[J]. Neurocomputing 213:191\u2013203","journal-title":"Neurocomputing"},{"issue":"5","key":"8798_CR27","doi-asserted-by":"publisher","first-page":"2494","DOI":"10.1109\/TIP.2017.2676345","volume":"26","author":"X Xu","year":"2017","unstructured":"Xu X, Shen F, Yang Y et al (2017) Learning discriminative binary codes for large-scale cross-modal retrieval[J]. IEEE Transactions on Image Processing 26(5):2494\u20132507","journal-title":"IEEE Transactions on Image Processing"},{"key":"8798_CR28","doi-asserted-by":"publisher","first-page":"354","DOI":"10.1016\/j.sigpro.2019.05.022","volume":"164","author":"X Xu","year":"2019","unstructured":"Xu X, Zhou X, Shen F et al (2019) Fusion by synthesizing: a multi-view deep neural network for zero-shot recognition[J]. Signal Processing 164:354\u2013367","journal-title":"Signal Processing"},{"issue":"4","key":"8798_CR29","doi-asserted-by":"publisher","first-page":"550","DOI":"10.1145\/279232.279236","volume":"23","author":"C Zhu","year":"1997","unstructured":"Zhu C, Byrd RH, Lu P et al (1997) Algorithm 778: L-BFGS-B: Fortran subroutines for large-scale bound-constrained optimization[J]. ACM Transactions on Mathematical Software (TOMS) 23(4):550\u2013560","journal-title":"ACM Transactions on Mathematical Software (TOMS)"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-020-08798-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11042-020-08798-6\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-020-08798-6.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2021,4,5]],"date-time":"2021-04-05T23:35:08Z","timestamp":1617665708000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11042-020-08798-6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,4,6]]},"references-count":29,"journal-issue":{"issue":"21-22","published-print":{"date-parts":[[2020,6]]}},"alternative-id":["8798"],"URL":"https:\/\/doi.org\/10.1007\/s11042-020-08798-6","relation":{},"ISSN":["1380-7501","1573-7721"],"issn-type":[{"value":"1380-7501","type":"print"},{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2020,4,6]]},"assertion":[{"value":"19 May 2019","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 January 2020","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 February 2020","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 April 2020","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}