{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,21]],"date-time":"2026-07-21T15:11:10Z","timestamp":1784646670854,"version":"3.55.0"},"reference-count":27,"publisher":"IEEE","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2018,4]]},"DOI":"10.1109\/icassp.2018.8462020","type":"proceedings-article","created":{"date-parts":[[2018,9,21]],"date-time":"2018-09-21T22:24:48Z","timestamp":1537568688000},"page":"4789-4793","source":"Crossref","is-referenced-by-count":41,"title":["Forward Attention in Sequence- To-Sequence Acoustic Modeling for Speech Synthesis"],"prefix":"10.1109","author":[{"given":"Jing-Xuan","family":"Zhang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhen-Hua","family":"Ling","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Li-Rong","family":"Dai","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref10","first-page":"3104","article-title":"Sequence to sequence learning with neural networks","author":"sutskever","year":"2014","journal-title":"Advances in neural information processing systems"},{"key":"ref11","author":"kyunghyun","year":"0","journal-title":"Learning phrase representations using RNN encoder-decoder for statistical machine translation"},{"key":"ref12","article-title":"Neural machine translation by jointly learning to align and translate","author":"bahdanau","year":"2014","journal-title":"ArXiv Preprint"},{"key":"ref13","article-title":"Effective approaches to attention-based neural machine translation","author":"luong","year":"2015","journal-title":"ArXiv Preprint"},{"key":"ref14","first-page":"2048","article-title":"Show, attend and tell: Neural image caption generation with visual attention","author":"kelvin","year":"2015","journal-title":"International Conference on Machine Learning"},{"key":"ref15","first-page":"577","article-title":"Attention-based models for speech recognition","author":"chorowski","year":"2015","journal-title":"Advances in neural information processing systems"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472621"},{"key":"ref17","first-page":"5067","article-title":"An online sequence-to-sequence model using partial conditioning","author":"navdeep","year":"2016","journal-title":"Advances in neural information processing systems"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-134"},{"key":"ref19","article-title":"Char2Wav: End-to-end speech synthesis","author":"jose","year":"2017","journal-title":"ICLR2017 workshop submission"},{"key":"ref4","first-page":"7962","article-title":"Statistical parametric speech synthesis using deep neural networks","author":"heiga","year":"2013","journal-title":"Proceedings of the IEEE International Conference on Acoustics Speech and Signal Processing (ICASSP)"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1016\/S0167-6393(98)00085-5"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2009.04.004"},{"key":"ref6","article-title":"Acoustic modeling in statistical parametric speech synthesis - from HMM to LSTM-RNN","author":"heiga","year":"2015","journal-title":"Proc MLSLP"},{"key":"ref5","article-title":"TTS synthesis with bidirectional LSTM based recurrent neural networks","author":"fan","year":"2014","journal-title":"Fifteenth Annual Conference of the International Speech Communication Association"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2014.2359987"},{"key":"ref7","article-title":"From HMMs to DNNs: where do the improvements come from?","volume":"41","author":"watts","year":"2016","journal-title":"Proc ICASSP"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511816338"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1007\/s12046-011-0048-y"},{"key":"ref20","article-title":"Tacotron: A fully end-to-end text-to-speech synthesis model","author":"wang","year":"2017","journal-title":"ArXiv Preprint"},{"key":"ref22","author":"kim","year":"2017","journal-title":"Structured attention networks"},{"key":"ref21","article-title":"Generating sequences with recurrent neural networks","author":"graves","year":"2013","journal-title":"ArXiv Preprint"},{"key":"ref24","first-page":"1","article-title":"Products of experts","volume":"1","author":"hinton","year":"1999","journal-title":"1999 Ninth International Conference on Artificial Neural Networks ICANN 99 (Conf Publ No 470)"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1145\/1143844.1143891"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/TASSP.1984.1164317"},{"key":"ref25","doi-asserted-by":"crossref","first-page":"3879","DOI":"10.4249\/scholarpedia.3879","article-title":"Product of experts","volume":"2","author":"max","year":"2007","journal-title":"Scholarpedia"}],"event":{"name":"ICASSP 2018 - 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","location":"Calgary, AB","start":{"date-parts":[[2018,4,15]]},"end":{"date-parts":[[2018,4,20]]}},"container-title":["2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/8450881\/8461260\/08462020.pdf?arnumber=8462020","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2020,8,24]],"date-time":"2020-08-24T04:27:46Z","timestamp":1598243266000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8462020\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,4]]},"references-count":27,"URL":"https:\/\/doi.org\/10.1109\/icassp.2018.8462020","relation":{},"subject":[],"published":{"date-parts":[[2018,4]]}}}