{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,13]],"date-time":"2026-05-13T08:56:42Z","timestamp":1778662602116,"version":"3.51.4"},"reference-count":22,"publisher":"IEEE","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2018,4]]},"DOI":"10.1109\/icassp.2018.8462166","type":"proceedings-article","created":{"date-parts":[[2018,9,21]],"date-time":"2018-09-21T18:24:48Z","timestamp":1537554288000},"page":"5489-5493","source":"Crossref","is-referenced-by-count":31,"title":["Time-Delayed Bottleneck Highway Networks Using a DFT Feature for Keyword Spotting"],"prefix":"10.1109","author":[{"given":"Jinxi","family":"Guo","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kenichi","family":"Kumatani","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ming","family":"Sun","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Minhua","family":"Wu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Anirudh","family":"Raju","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Nikko","family":"Strom","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Arindam","family":"Mandal","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-1485"},{"key":"ref11","article-title":"Deep and wide: Multiple layers in automatic speech recognition","volume":"20","author":"morgan","year":"2012","journal-title":"IEEE Trans on Audio Speech & Lan-guage Processing"},{"key":"ref12","article-title":"Architectures for deep neural network based acoustic models defined over windowed speech waveforms","author":"bhargava","year":"2015","journal-title":"Proc INTERSPEECH"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-1495"},{"key":"ref14","article-title":"Learning the speech front-end with raw waveform CLDNNs","author":"sainath","year":"2015","journal-title":"Proc INTERSPEECH"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-1459"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1016\/0893-6080(90)90044-L"},{"key":"ref17","author":"srivastava","year":"2015","journal-title":"Highway networks"},{"key":"ref18","article-title":"Small-footprint deep neural networks with highway connections for speech recognition","author":"lu","year":"2016","journal-title":"Proc INTERSPEECH"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1002\/9780470714089"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/ICMLA.2015.121"},{"key":"ref3","article-title":"Spoken Language Understanding for Amazon Echo","author":"prasad","year":"2015","journal-title":"Keynote in Speech and Audio in the Northeast (SANE)"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2016.7846306"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-1393"},{"key":"ref8","article-title":"An empirical study of cross-lingual transfer learning techniques for small-footprint keyword spotting","author":"sun","year":"2017","journal-title":"Sixth Int Conference on Machine Learning and Applications (ICMLA"},{"key":"ref7","doi-asserted-by":"crossref","first-page":"3607","DOI":"10.21437\/Interspeech.2017-480","article-title":"Compressed time delay neural network for small-footprint keyword spotting","author":"sun","year":"2017","journal-title":"Proc INTERSPEECH"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2012.2205285"},{"key":"ref1","article-title":"Direct modeling of raw audio with dnns for wake word detection","author":"kumatani","year":"2017","journal-title":"Proc IEEE Workshop Automatic Speech Recognition and Understanding (ASRU)"},{"key":"ref9","article-title":"monophone-based background modeling for two-stage on-device wake word detection","author":"wu","year":"2018","journal-title":"IEEE International Conference on Acoustics Speech and Signal Processing (ICASSP)"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-92"},{"key":"ref22","first-page":"1488","article-title":"Scalable distributed DNN training using commodity GPU cloud computing","author":"strom","year":"2015","journal-title":"Proc INTERSPEECH"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-173"}],"event":{"name":"ICASSP 2018 - 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","location":"Calgary, AB","start":{"date-parts":[[2018,4,15]]},"end":{"date-parts":[[2018,4,20]]}},"container-title":["2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/8450881\/8461260\/08462166.pdf?arnumber=8462166","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2020,8,23]],"date-time":"2020-08-23T20:51:41Z","timestamp":1598215901000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8462166\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,4]]},"references-count":22,"URL":"https:\/\/doi.org\/10.1109\/icassp.2018.8462166","relation":{},"subject":[],"published":{"date-parts":[[2018,4]]}}}