{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T04:54:52Z","timestamp":1777611292681,"version":"3.51.4"},"reference-count":141,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"9","license":[{"start":{"date-parts":[[2013,9,1]],"date-time":"2013-09-01T00:00:00Z","timestamp":1377993600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2013,9,1]],"date-time":"2013-09-01T00:00:00Z","timestamp":1377993600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2013,9,1]],"date-time":"2013-09-01T00:00:00Z","timestamp":1377993600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["IIS-I0916918"],"award-info":[{"award-number":["IIS-I0916918"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Proc. IEEE"],"published-print":{"date-parts":[[2013,9]]},"DOI":"10.1109\/jproc.2013.2252316","type":"journal-article","created":{"date-parts":[[2013,7,24]],"date-time":"2013-07-24T14:54:16Z","timestamp":1374677656000},"page":"1968-1985","source":"Crossref","is-referenced-by-count":13,"title":["Perceptual Properties of Current Speech Recognition Technology"],"prefix":"10.1109","volume":"101","author":[{"given":"Hynek","family":"Hermansky","sequence":"first","affiliation":[{"name":"Center for Language and Speech Processing, The Johns Hopkins University, Baltimore, MD, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jordan R.","family":"Cohen","sequence":"additional","affiliation":[{"name":"Spelamode, Inc., Kure Beach, NC, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Richard M.","family":"Stern","sequence":"additional","affiliation":[{"name":"Department of Electrical and Computer Engineering and the Language Technologies Institute, Carnegie Mellon University, Pittsburgh, PA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref39","first-page":"914","article-title":"Improve the implementation of pitch features for mandarin digit string recognition task","author":"ding","year":"0","journal-title":"Proc Interspeech 2012"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2012.10.004"},{"key":"ref33","author":"helmholtz","year":"1954","journal-title":"On the Sensations of Tone"},{"key":"ref32","author":"ladefoged","year":"1967","journal-title":"Three Areas of Experimental Phonetics Stress and Respiratory Activity the Nature of Vowel Quality Units in the Perception and Production of Speech"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1121\/1.380687"},{"key":"ref30","author":"stevens","year":"1998","journal-title":"Acoustic Phonetics"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2011.2134086"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2006.11.002"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1016\/0378-5955(79)90012-1"},{"key":"ref34","author":"fant","year":"1970","journal-title":"Acoustic Theory of Speech Production"},{"key":"ref28","doi-asserted-by":"crossref","first-page":"613","DOI":"10.1121\/1.1907979","article-title":"A difference limen for vowel formant frequency","volume":"27","author":"flanagan","year":"1955","journal-title":"J Acoust Soc Amer"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1121\/1.1915893"},{"key":"ref29","author":"goldstein","year":"1980","journal-title":"An articulatory model for the vocal tracts of growing children"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1121\/1.1906946"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1985.1168384"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1986.1168649"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ASPAA.1991.634094"},{"key":"ref23","doi-asserted-by":"crossref","first-page":"303","DOI":"10.1126\/science.270.5234.303","article-title":"Speech recognition with primarily temporal cues","volume":"270","author":"shannon","year":"1995","journal-title":"Science"},{"key":"ref26","author":"von b\ufffdk\ufffdsy","year":"1960","journal-title":"Experiments in Hearing"},{"key":"ref101","first-page":"2329","article-title":"Adaptive stream fusion in multistream recognition of speech","author":"mesgarani","year":"0","journal-title":"Proc Interspeech 2011"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1006\/dspr.1999.0363"},{"key":"ref100","author":"valente","year":"2011","journal-title":"Handbook of Natural Language Processing and Machine Translation DARPA Global Autonomous Language Exploitation"},{"key":"ref50","first-page":"361","article-title":"Multi-resolution RASTA filtering for TANDEM-based ASR","author":"hermanksy","year":"0","journal-title":"Proc Interspeech 2005"},{"key":"ref51","first-page":"2237","article-title":"Phonotactic language identification using high quality phoneme recognition","author":"matejka","year":"1990","journal-title":"Proc Interspeech 2005"},{"key":"ref59","author":"fukunaga","year":"1990","journal-title":"Introduction to Statistical Pattern Classification"},{"key":"ref58","first-page":"1630","article-title":"Two protocols comparing human and machine phonetic recognition performance in conversational speech","author":"shen","year":"0","journal-title":"Proc Interspeech 2008"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2011.2129510"},{"key":"ref56","author":"bourlard","year":"1993","journal-title":"Connectionist Speech Recognition A Hybrid Approach"},{"key":"ref55","article-title":"MAP estimation of whole-word acoustic models with dictionary priors","author":"kintzley","year":"0","journal-title":"Proc Interspeech 2012"},{"key":"ref54","first-page":"1905","article-title":"Event selection from phone posteriorgrams using matched filters","author":"kintzley","year":"0","journal-title":"Proc Interspeech 2011"},{"key":"ref53","article-title":"Hierarchical approach for spotting keywords","author":"lehtonen","year":"2005","journal-title":"Proc 2nd Workshop Multimodal Interaction Related Mach Learn Algorithms"},{"key":"ref52","first-page":"2414","article-title":"Combining evidence from a generative and a discriminative model in phoneme recognition","author":"pinto","year":"2008","journal-title":"Proc IEEE Int Conf Acoust Speech Signal Process"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2006.876880"},{"key":"ref4","first-page":"2","author":"galt","year":"0","journal-title":"Galt Study of speech and hearing at Bell Telephone Laboratories Correspondence files (1917?1933) and other internal reports 1917?1933"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1121\/1.399423"},{"key":"ref6","first-page":"349","article-title":"Discriminant linear processing of time-frequency plane","author":"valente","year":"0","journal-title":"Proc Interspeech 2006"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1989.266468"},{"key":"ref8","first-page":"374","author":"mermelstein","year":"1976","journal-title":"Pattern Recognition and Artificial Intelligence"},{"key":"ref49","doi-asserted-by":"crossref","first-page":"437","DOI":"10.21437\/Eurospeech.2003-164","article-title":"Beyond a single critical-band in TRAP based ASR","author":"jain","year":"2003","journal-title":"Proc Eur Conf Speech Commun Technol"},{"key":"ref7","author":"bridle","year":"1974","journal-title":"An Experimental Automatic Word-Recognition System"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1976.1170074"},{"key":"ref46","first-page":"375","article-title":"Similarity measure for automatic speech and speaker recognition","volume":"32","author":"schroeder","year":"1967","journal-title":"J Acoust Soc Amer"},{"key":"ref45","first-page":"81","article-title":"Speech discrimination by dynamic programming","volume":"4","author":"vintsyuk","year":"1968","journal-title":"Kibernetika"},{"key":"ref48","first-page":"199","author":"fanty","year":"1992","journal-title":"Advances in Neural Information Processing Systems 4"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1109\/34.62605"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2008.4518613"},{"key":"ref41","author":"potter","year":"1966","journal-title":"Visible Speech"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/TASSP.1978.1163055"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1121\/1.1911801"},{"key":"ref127","doi-asserted-by":"publisher","DOI":"10.1121\/1.1610463"},{"key":"ref126","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1983.1171927"},{"key":"ref125","first-page":"183","article-title":"Speech intelligibility, spatial unmasking, and realism in reverberant spatial auditory displays","author":"shinn-cunningham","year":"2002","journal-title":"Proc Int Conf Auditory Display"},{"key":"ref124","doi-asserted-by":"publisher","DOI":"10.1121\/1.428503"},{"key":"ref73","doi-asserted-by":"crossref","first-page":"55","DOI":"10.1016\/S0095-4470(19)30466-8","article-title":"A joint synchrony\/mean-rate model of auditory speech processing","volume":"15","author":"seneff","year":"1988","journal-title":"J Phonetics"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1121\/1.397756"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/TASSP.1981.1163530"},{"key":"ref129","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2004.03.005"},{"key":"ref70","first-page":"257","volume":"3","author":"mlouka","year":"1975","journal-title":"Speech Communication"},{"key":"ref128","doi-asserted-by":"publisher","DOI":"10.1109\/TSMCB.2004.830345"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.1109\/89.326616"},{"key":"ref77","first-page":"409","article-title":"Data-driven design of RASTA-like filters","author":"van vuuren","year":"0","journal-title":"Proc Eurospeech 1997"},{"key":"ref130","doi-asserted-by":"publisher","DOI":"10.1109\/TSA.2005.860354"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1016\/S0885-2308(86)80018-3"},{"key":"ref75","author":"stern","year":"2012","journal-title":"Techniques for Noise Robustness in Automatic Speech Recognition"},{"key":"ref133","doi-asserted-by":"publisher","DOI":"10.2307\/1418275"},{"key":"ref134","doi-asserted-by":"publisher","DOI":"10.1109\/ASPAA.1997.625622"},{"key":"ref78","doi-asserted-by":"publisher","DOI":"10.1121\/1.414981"},{"key":"ref131","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2006.09.003"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.1121\/1.387576"},{"key":"ref132","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2008.05.012"},{"key":"ref136","doi-asserted-by":"publisher","DOI":"10.1121\/1.394325"},{"key":"ref135","first-page":"2058","article-title":"Nonlinear enhancement of onset for robust speech recognition","author":"kim","year":"0","journal-title":"Proc INTERSPEECH 2010"},{"key":"ref138","doi-asserted-by":"crossref","first-page":"288","DOI":"10.1152\/jn.1966.29.2.288","article-title":"Some neural mechanisms in the inferior colliculus of the cat which may be relevant to localization of a sound source","volume":"29","author":"rose","year":"1966","journal-title":"J Neurophysiol"},{"key":"ref137","doi-asserted-by":"publisher","DOI":"10.1121\/1.394326"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2000.862024"},{"key":"ref139","doi-asserted-by":"crossref","first-page":"465","DOI":"10.1152\/jn.1990.64.2.465","article-title":"Interaural time sensitivity in medial superior olive of cat","volume":"64","author":"yin","year":"1990","journal-title":"J Neurophysiol"},{"key":"ref62","author":"kozhevnikov","year":"1967","journal-title":"Speech Articulation and perception"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1016\/S0167-6393(99)00050-3"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1037\/0033-2909.96.2.341"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1016\/S0167-6393(98)00027-2"},{"key":"ref140","doi-asserted-by":"crossref","first-page":"347","DOI":"10.1016\/B978-012505626-7\/50012-1","author":"stern","year":"1995","journal-title":"Hearing"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1037\/h0044155"},{"key":"ref141","author":"stern","year":"2006","journal-title":"Computational Auditory Scene Analysis"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1038\/19652"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/89.593318"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1109\/PROC.1975.9800"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1109\/TASSP.1975.1162641"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/PROC.1976.10159"},{"key":"ref1","author":"flanagan","year":"1983","journal-title":"Speech Analysis Synthesis and Perception"},{"key":"ref95","first-page":"462","article-title":"Towards ASR on partially corrupted speech","author":"hermansky","year":"1996","journal-title":"Proc Int Conf Spoken Lang Process"},{"key":"ref109","first-page":"1975","article-title":"Physiologically-motivated synchrony-based processing for robust automatic speech recognition","author":"kim","year":"0","journal-title":"Proc Interspeech 2006"},{"key":"ref94","first-page":"1579","article-title":"Towards subband-based speech recognition","author":"bourlard","year":"1996","journal-title":"Proc Eur Signal Process Conf"},{"key":"ref108","doi-asserted-by":"publisher","DOI":"10.1109\/89.736331"},{"key":"ref93","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2005.1511828"},{"key":"ref107","doi-asserted-by":"publisher","DOI":"10.1121\/1.1945807"},{"key":"ref92","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2004.03.007"},{"key":"ref106","doi-asserted-by":"publisher","DOI":"10.1121\/1.427950"},{"key":"ref91","doi-asserted-by":"publisher","DOI":"10.1016\/S0167-6393(00)00034-0"},{"key":"ref105","author":"slaney","year":"1998","journal-title":"Auditory Toolbox (V 2)"},{"key":"ref90","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1997.596069"},{"key":"ref104","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1982.1171644"},{"key":"ref103","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2012.2236871"},{"key":"ref102","first-page":"477","article-title":"A performance monitoring approach to fusing enhanced spectrogram channels in robust speech recognition","author":"badiezadegan","year":"0","journal-title":"Proc Interspeech 2011"},{"key":"ref111","article-title":"Power-normalized cepstral coefficients (PNCC) for robust speech recognition","author":"kim","year":"2013","journal-title":"IEEE Trans Audio Speech Lang Process"},{"key":"ref112","author":"pickles","year":"1986","journal-title":"Frequency Selectivity in Hearing"},{"key":"ref110","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2010.5495570"},{"key":"ref98","first-page":"1757","article-title":"Detection of out-of-vocabulary words in posterior based ASR","author":"ketabdar","year":"0","journal-title":"Proc Interspeech 2007"},{"key":"ref99","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2008.4518547"},{"key":"ref96","first-page":"426","article-title":"A new ASR approach based on independent processing and re-combination of partial frequency bands","author":"bourlard","year":"1996","journal-title":"Proc Int Conf Spoken Lang Process"},{"key":"ref97","article-title":"Multi-stream approach in acoustic modeling","author":"tibrewala","year":"1997","journal-title":"Proc DARPA Large Vocabulary Continuous Speech Recognit Hub 5 Workshop"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1976.1170013"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1121\/1.384992"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1016\/0167-6393(85)90045-7"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1983.1172025"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1994.389250"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICSLP.1996.607142"},{"key":"ref16","author":"kamm","year":"1997","journal-title":"Learning the Mel-scale and Optimal VTN Mapping"},{"key":"ref82","author":"singh","year":"2002","journal-title":"Noise Reduction in Speech Applications"},{"key":"ref118","doi-asserted-by":"publisher","DOI":"10.1121\/1.383532"},{"key":"ref17","first-page":"1379","article-title":"Spectral basis functions from discriminant analysis","author":"hermansky","year":"1998","journal-title":"Proc Int Conf Spoken Lang Process"},{"key":"ref81","doi-asserted-by":"publisher","DOI":"10.1016\/0885-2308(91)90011-E"},{"key":"ref117","doi-asserted-by":"publisher","DOI":"10.1121\/1.1906346"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2009.2014096"},{"key":"ref84","doi-asserted-by":"publisher","DOI":"10.1002\/9781118392683"},{"key":"ref19","article-title":"An experiment in systematic speaker variability","author":"andreou","year":"1994","journal-title":"Proc DoD Workshop Front Speech Process II"},{"key":"ref83","author":"singh","year":"2002","journal-title":"Noise Reduction in Speech Applications"},{"key":"ref119","first-page":"1","article-title":"Auditory models and isolated word recognition","volume":"24","author":"blomberg","year":"1983","journal-title":"STL-QPRS"},{"key":"ref114","doi-asserted-by":"publisher","DOI":"10.1109\/89.294356"},{"key":"ref113","doi-asserted-by":"publisher","DOI":"10.1121\/1.1910947"},{"key":"ref80","doi-asserted-by":"publisher","DOI":"10.1007\/s12046-011-0044-2"},{"key":"ref116","first-page":"1000","article-title":"Analysis of physiologically-motivated signal processing for robust speech recognition","author":"chiu","year":"0","journal-title":"Proc Interspeech 2008"},{"key":"ref115","doi-asserted-by":"crossref","first-page":"1799","DOI":"10.1152\/jn.1988.60.6.1799","article-title":"Periodicity coding in the inferior colliculus of the cat. I. neuronal mechanisms","volume":"60","author":"langner","year":"1988","journal-title":"J Neurophysiol"},{"key":"ref89","author":"seltzer","year":"2012","journal-title":"Techniques for Noise Robustness in Automatic Speech Recognition"},{"key":"ref120","doi-asserted-by":"publisher","DOI":"10.1109\/TSA.2002.800556"},{"key":"ref121","first-page":"18","article-title":"Signal separation for robust speech recognition based on phase difference information obtained in the frequency domain","author":"kim","year":"0","journal-title":"Proc Interspeech 2009"},{"key":"ref122","doi-asserted-by":"publisher","DOI":"10.1080\/14786440709463595"},{"key":"ref123","doi-asserted-by":"publisher","DOI":"10.1121\/1.424670"},{"key":"ref85","doi-asserted-by":"publisher","DOI":"10.1109\/TASSP.1979.1163209"},{"key":"ref86","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1990.115971"},{"key":"ref87","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1996.543225"},{"key":"ref88","author":"hershey","year":"2012","journal-title":"Techniques for Noise Robustness in Automatic Speech Recognition"}],"container-title":["Proceedings of the IEEE"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/5\/6582526\/06566018.pdf?arnumber=6566018","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,5,9]],"date-time":"2024-05-09T17:33:48Z","timestamp":1715276028000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/6566018\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2013,9]]},"references-count":141,"journal-issue":{"issue":"9"},"URL":"https:\/\/doi.org\/10.1109\/jproc.2013.2252316","relation":{},"ISSN":["0018-9219","1558-2256"],"issn-type":[{"value":"0018-9219","type":"print"},{"value":"1558-2256","type":"electronic"}],"subject":[],"published":{"date-parts":[[2013,9]]}}}