{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,8]],"date-time":"2026-03-08T17:18:05Z","timestamp":1772990285534,"version":"3.50.1"},"reference-count":43,"publisher":"Springer Science and Business Media LLC","issue":"17","license":[{"start":{"date-parts":[[2022,3,19]],"date-time":"2022-03-19T00:00:00Z","timestamp":1647648000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,3,19]],"date-time":"2022-03-19T00:00:00Z","timestamp":1647648000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimed Tools Appl"],"published-print":{"date-parts":[[2022,7]]},"DOI":"10.1007\/s11042-022-12116-7","type":"journal-article","created":{"date-parts":[[2022,3,19]],"date-time":"2022-03-19T06:02:51Z","timestamp":1647669771000},"page":"24419-24430","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":8,"title":["Self-attention generative adversarial networks applied to conditional music generation"],"prefix":"10.1007","volume":"81","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8902-5743","authenticated-orcid":false,"given":"Pedro Lucas","family":"Tomaz Neves","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jos\u00e9","family":"Fornari","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jo\u00e3o","family":"Batista Florindo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2022,3,19]]},"reference":[{"key":"12116_CR1","unstructured":"Barratt S, Sharma R (2018) A note on the inception score. arXiv:1801.01973"},{"key":"12116_CR2","unstructured":"Binkowski M, Donahue J, Dieleman S, Clark A, Elsen E, Casagrande N, Cobo LC, Simonyan K (2019) High fidelity speech synthesis with adversarial networks. arXiv:1909.11646"},{"key":"12116_CR3","doi-asserted-by":"publisher","first-page":"41","DOI":"10.1016\/j.cviu.2018.10.009","volume":"179","author":"A Borji","year":"2019","unstructured":"Borji A (2019) Pros and cons of GAN evaluation measures. Comput Vis Image Underst 179:41\u201365","journal-title":"Comput Vis Image Underst"},{"key":"12116_CR4","unstructured":"van den Broek K (2021) Mp3net: coherent, minute-long music generation from raw audio with a simple convolutional GAN. arXiv:2101.04785"},{"key":"12116_CR5","unstructured":"Cordonnier J, Loukas A, Jaggi M (2019) On the relationship between self-attention and convolutional layers. arXiv:1911.03584"},{"key":"12116_CR6","doi-asserted-by":"crossref","unstructured":"Deng J, Dong W, Socher R, Li LJ, Li K, Fei-Fei L (2009) ImageNet: A Large-Scale Hierarchical Image Database. In: CVPR09","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"12116_CR7","unstructured":"Dhariwal P, Jun H, Payne C, Kim JW, Radford A, Sutskever I (2020) Jukebox: A generative model for music. arXiv:2005.00341"},{"key":"12116_CR8","unstructured":"Dieleman S, van den Oord A, Simonyan K (2018) The challenge of realistic music generation: modelling raw audio at scale. arXiv:1806.10474"},{"key":"12116_CR9","unstructured":"Donahue C, McAuley JJ, Puckette MS (2018) Synthesizing audio with generative adversarial networks. arXiv:1802.04208"},{"key":"12116_CR10","unstructured":"Donahue J, Dieleman S, Binkowski M, Elsen E, Simonyan K (2021) End-to-end adversarial text-to-speech. In: International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=rsf1z-JSj87"},{"key":"12116_CR11","unstructured":"Dong H, Yang Y (2018) Convolutional generative adversarial networks with binary neurons for polyphonic music generation. arXiv:1804.09399"},{"key":"12116_CR12","unstructured":"Engel J, Agrawal KK, Chen S, Gulrajani I, Donahue C, Roberts A (2019) GANSynth: Adversarial neural audio synthesis. In: International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=H1xQVn09FX"},{"key":"12116_CR13","unstructured":"Ferreira LN, Whitehead J (2021) Learning to generate music with sentiment. arXiv:210306125"},{"key":"12116_CR14","unstructured":"Goodfellow IJ, Pouget-Abadie J, Mirza M, Xu B, Warde-Farley D, Ozair S, Courville A, Bengio Y (2014) Generative adversarial networks. arXiv:1406.2661"},{"key":"12116_CR15","doi-asserted-by":"publisher","unstructured":"Guan F, Yu C, Yang S (2019) A gan model with self-attention mechanism to generate multi-instruments symbolic music. In: 2019 International Joint Conference on Neural Networks (IJCNN), pp 1\u20136. https:\/\/doi.org\/10.1109\/IJCNN.2019.8852291","DOI":"10.1109\/IJCNN.2019.8852291"},{"key":"12116_CR16","doi-asserted-by":"crossref","unstructured":"He K, Zhang X, Ren S, Sun J (2015) Deep residual learning for image recognition. arXiv:1512.03385","DOI":"10.1109\/CVPR.2016.90"},{"key":"12116_CR17","unstructured":"Heusel M, Ramsauer H, Unterthiner T, Nessler B, Klambauer G, Hochreiter S (2017a) Gans trained by a two time-scale update rule converge to a nash equilibrium. arXiv:1706.08500"},{"key":"12116_CR18","unstructured":"Heusel M, Ramsauer H, Unterthiner T, Nessler B, Klambauer G, Hochreiter S (2017b) Gans trained by a two time-scale update rule converge to a nash equilibrium. arXiv:1706.08500"},{"key":"12116_CR19","unstructured":"Kingma DP, Ba J (2015) Adam: A method for stochastic optimization. In: Bengio Y, LeCun Y (eds) 3rd International Conference on Learning Representations, ICLR 2015, San Diego, CA, USA, May 7-9, 2015, Conference Track Proceedings. arXiv:1412.6980"},{"key":"12116_CR20","unstructured":"Kumar K, Kumar R, de Boissiere T, Gestin L, Teoh WZ, Sotelo J, de Brebisson A, Bengio Y, Courville A (2019)"},{"key":"12116_CR21","unstructured":"Lim JH, Ye JC (2017) Geometric gan. arXiv:170502894"},{"key":"12116_CR22","unstructured":"Lostanlen V, Cella CE (2016) Deep convolutional networks on the pitch spiral for music instrument recognition. In: ISMIR"},{"key":"12116_CR23","doi-asserted-by":"crossref","unstructured":"Mao HH, Shin T, Cottrell G (2018) Deepj: Style-Specific music generation. In: 2018 IEEE 12Th international conference on semantic computing (ICSC). IEEE, pp 377\u2013382","DOI":"10.1109\/ICSC.2018.00077"},{"key":"12116_CR24","unstructured":"Mirza M, Osindero S (2014) Conditional generative adversarial nets. arXiv:1411.1784"},{"key":"12116_CR25","unstructured":"Miyato T, Koyama M (2018) Cgans with projection discriminator. arXiv:1802.05637"},{"key":"12116_CR26","unstructured":"Miyato T, Kataoka T, Koyama M, Yoshida Y (2018) Spectral normalization for generative adversarial networks. arXiv:1802.05957"},{"key":"12116_CR27","doi-asserted-by":"crossref","unstructured":"Muhamed A, Li L, Shi X, Yaddanapudi S, Chi W, Jackson D, Suresh R, Lipton ZC, Smola AJ (2021) Symbolic music generation with transformer-gans. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol 35, pp 408\u2013417","DOI":"10.1609\/aaai.v35i1.16117"},{"key":"12116_CR28","unstructured":"van den Oord A, Dieleman S, Zen H, Simonyan K, Vinyals O, Graves A, Kalchbrenner N, Senior AW, Kavukcuoglu K (2016) Wavenet: a generative model for raw audio. arXiv:1609.03499"},{"key":"12116_CR29","unstructured":"van den Oord A, Li Y, Babuschkin I, Simonyan K, Vinyals O, Kavukcuoglu K, van den Driessche G, Lockhart E, Cobo LC, Stimberg F, Casagrande N, Grewe D, Noury S, Dieleman S, Elsen E, Kalchbrenner N, Zen H, Graves A, King H, Walters T, Belov D, Hassabis D (2017a) Parallel wavenet: Fast high-fidelity speech synthesis. arXiv:1711.10433"},{"key":"12116_CR30","unstructured":"van den Oord A, Vinyals O, Kavukcuoglu K (2017b) Neural discrete representation learning. arXiv:1711.00937"},{"key":"12116_CR31","first-page":"8026","volume":"32","author":"A Paszke","year":"2019","unstructured":"Paszke A, Gross S, Massa F, Lerer A, Bradbury J, Chanan G, Killeen T, Lin Z, Gimelshein N, Antiga L et al (2019) Pytorch: an imperative style, high-performance deep learning library. Adv Neural Inform Process Syst 32:8026\u20138037","journal-title":"Adv Neural Inform Process Syst"},{"key":"12116_CR32","unstructured":"Razavi A, van den Oord A, Vinyals O (2019) Generating diverse high-fidelity images with VQ-VAE-2. arXiv:1906.00446"},{"key":"12116_CR33","unstructured":"Salimans T, Goodfellow IJ, Zaremba W, Cheung V, Radford A, Chen X (2016) Improved techniques for training gans. arXiv:1606.03498"},{"key":"12116_CR34","unstructured":"dos Santos Tanaka FHK, Aranha C (2019) Data augmentation using gans. arXiv:1904.09135"},{"key":"12116_CR35","doi-asserted-by":"publisher","unstructured":"Shen J, Pang R, Weiss RJ, Schuster M, Jaitly N, Yang Z, Chen Z, Zhang Y, Wang Y, Skerrv-Ryan R, Saurous RA, Agiomvrgiannakis Y, Wu Y (2018) Natural tts synthesis by conditioning wavenet on mel spectrogram predictions. In: 2018 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 4779\u20134783. https:\/\/doi.org\/10.1109\/ICASSP.2018.8461368","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"12116_CR36","doi-asserted-by":"crossref","unstructured":"Szegedy C, Vanhoucke V, Ioffe S, Shlens J, Wojna Z (2015) Rethinking the inception architecture for computer vision. arXiv:1512.00567","DOI":"10.1109\/CVPR.2016.308"},{"key":"12116_CR37","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser L, Polosukhin I (2017) Attention is all you need. arXiv:1706.03762"},{"key":"12116_CR38","doi-asserted-by":"crossref","unstructured":"Wang Y, Skerry-Ryan R, Stanton D, Wu Y, Weiss RJ, Jaitly N, Yang Z, Xiao Y, Chen Z, Bengio S, Le QV, Agiomyrgiannakis Y, Clark R, Saurous R (2017) Tacotron: Towards end-to-end speech synthesis. In: INTERSPEECH","DOI":"10.21437\/Interspeech.2017-1452"},{"key":"12116_CR39","doi-asserted-by":"crossref","unstructured":"Weiss RJ, Skerry-Ryan R, Battenberg E, Mariooryad S, Kingma DP (2021) Wave-tacotron: Spectrogram-Free end-to-end text-to-speech synthesis. In: ICASSP 2021-2021 IEEE International conference on acoustics, speech and signal processing (ICASSP). IEEE, pp 5679\u20135683","DOI":"10.1109\/ICASSP39728.2021.9413851"},{"key":"12116_CR40","unstructured":"Yang L, Chou S, Yang Y (2017) Midinet: a convolutional generative adversarial network for symbolic-domain music generation using 1d and 2d conditions. arXiv:1703.10847"},{"key":"12116_CR41","doi-asserted-by":"publisher","unstructured":"Yu Y, Srivastava A, Canales S (2021) Conditional lstm-gan for melody generation from lyrics. ACM Trans Multimedia Comput Commun Appl, 17(1). https:\/\/doi.org\/10.1145\/3424116","DOI":"10.1145\/3424116"},{"key":"12116_CR42","unstructured":"Zhang H, Goodfellow I, Metaxas D, Odena A (2019) Self-attention generative adversarial networks. arXiv:1805.08318"},{"key":"12116_CR43","unstructured":"Zhao S, Liu Z, Lin J, Zhu JY, Han S (2020) Differentiable augmentation for data-efficient gan training. arXiv:2006.10738"}],"container-title":["Multimedia Tools and Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-022-12116-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11042-022-12116-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11042-022-12116-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,1,29]],"date-time":"2023-01-29T14:17:35Z","timestamp":1675001855000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11042-022-12116-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,3,19]]},"references-count":43,"journal-issue":{"issue":"17","published-print":{"date-parts":[[2022,7]]}},"alternative-id":["12116"],"URL":"https:\/\/doi.org\/10.1007\/s11042-022-12116-7","relation":{},"ISSN":["1380-7501","1573-7721"],"issn-type":[{"value":"1380-7501","type":"print"},{"value":"1573-7721","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,3,19]]},"assertion":[{"value":"19 April 2021","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"27 July 2021","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"3 January 2022","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"19 March 2022","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"<!--Emphasis Type='Bold' removed-->Conflict of Interests"}}]}}