{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,21]],"date-time":"2025-12-21T06:24:55Z","timestamp":1766298295419,"version":"3.41.0"},"reference-count":45,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2018,4,17]],"date-time":"2018-04-17T00:00:00Z","timestamp":1523923200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Speech Technol"],"published-print":{"date-parts":[[2018,6]]},"DOI":"10.1007\/s10772-018-9511-z","type":"journal-article","created":{"date-parts":[[2018,4,17]],"date-time":"2018-04-17T16:49:11Z","timestamp":1523983751000},"page":"343-354","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Adaptive framing based similarity measurement between time warped speech signals using Kalman filter"],"prefix":"10.1007","volume":"21","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7511-3873","authenticated-orcid":false,"given":"Wasiq","family":"Khan","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Keeley","family":"Crockett","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Muhammad","family":"Bilal","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2018,4,17]]},"reference":[{"key":"9511_CR1","doi-asserted-by":"crossref","unstructured":"Abad, A., Rodriguez-Fuentes, L. J., Penagarikano, M., Varona, A., Diez, M., & Bordel, G. (2013). On the calibration and fusion of heterogeneous spoken term detection systems. Conference of the International Speech Communication Association, Interspeech, France, 25\u201329 August 2013.","DOI":"10.21437\/Interspeech.2013-5"},{"issue":"12","key":"9511_CR2","first-page":"3411","volume":"2","author":"A Akila","year":"2013","unstructured":"Akila, A., & Chandra, E. (2013). Slope finder\u2014A distance measure for DTW based isolated word speech recognition. International Journal of Engineering and Computer Science, 2(12), 3411\u20133417.","journal-title":"International Journal of Engineering and Computer Science"},{"key":"9511_CR3","unstructured":"Anguera, X., Metze, F., Buzo, A., Szoke, I., & Rodriguez-Fuentes, L. J. (2013). The spoken web search task. In Proceedings of MediaEval (pp. 1\u20132), Aachen, Germany: CEUR Workshop Proceedings."},{"key":"9511_CR4","unstructured":"Anguera, X., Rodriguez-Fuentes, L. J., Szoke, I., Buzo, A., & Metze, F. (2014). Query by example search on speech. In Proceedings of MediaEval (pp. 1\u20132). Spain"},{"key":"9511_CR5","doi-asserted-by":"crossref","unstructured":"Chan, C.-A., & Lee, L. S. (2010). Unsupervised spoken-term detection with spoken queries using segment-based dynamic time warping. In Proceedings of Interspeech (pp.\u00a0693\u2013696). Prague","DOI":"10.21437\/Interspeech.2010-262"},{"key":"9511_CR6","doi-asserted-by":"publisher","unstructured":"Cheng-Tao, C., Chun-an, C., & Lin-Shan, L. (2014). Unsupervised spoken term detection with spoken queries by multi-level acoustic patterns with varying model granularity. IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP) (pp. 7814\u20137818), 4\u20139 May 2014. https:\/\/doi.org\/10.1109\/ICASSP.2014.6855121 .","DOI":"10.1109\/ICASSP.2014.6855121"},{"key":"9511_CR7","unstructured":"Chotirat, R., & Eamonn, K. (2005). Three myths about dynamic time warping data mining. In The Proceedings of SIAM International Conference on Data Mining (pp. 506\u2013510)."},{"key":"9511_CR8","unstructured":"Chun-An, C., & Lin-Shan, L. (2011). Unsupervised hidden markov modeling of spoken queries for spoken term detection without speech recognition. In Proceedings of Interspeech (pp. 2141\u20132144)."},{"issue":"7","key":"9511_CR9","doi-asserted-by":"publisher","first-page":"1330","DOI":"10.1109\/TASL.2013.2248714","volume":"21","author":"C Chun-An","year":"2013","unstructured":"Chun-An, C., & Lin-Shan, L. (2013). Model-based unsupervised spoken term detection with spoken queries. IEEE Transactions on Audio, Speech, and Language Processing, 21(7), 1330\u20131342.","journal-title":"IEEE Transactions on Audio, Speech, and Language Processing"},{"issue":"6","key":"9511_CR10","first-page":"1","volume":"1","author":"N Dave","year":"2013","unstructured":"Dave, N. (2013). Feature extraction methods LPC, PLP and MFCC in speech recognition. International Journal for Advance Research in Engineering and Technology, 1(6), 1\u20134.","journal-title":"International Journal for Advance Research in Engineering and Technology"},{"issue":"8","key":"9511_CR11","first-page":"1","volume":"2","author":"S Dhingra","year":"2013","unstructured":"Dhingra, S., Nijhawan, G., & Pandit, P. (2013). Isolated speech recognition using MFCC and DTW. International Journal of Advanced Research in Electrical Electronics and Instrumentation Engineering, 2(8), 1\u20138.","journal-title":"International Journal of Advanced Research in Electrical Electronics and Instrumentation Engineering"},{"key":"9511_CR12","doi-asserted-by":"crossref","unstructured":"Ezzaidi, H., & Jean, R. (2004). Pitch and MFCC dependent GMM models for speaker identification systems. Canadian Conference on Electrical and Computer Engineering (Vol.\u00a01, pp. 43\u201346).","DOI":"10.1109\/CCECE.2004.1344954"},{"key":"9511_CR13","unstructured":"Giannakopoulos, T. (2014). A method for silence removal and segmentation of speech signals, implemented in Matlab, 2014. Retrieved May 13, 2014 from http:\/\/cgi.di.uoa.gr\/~tyiannak\/Software.html ."},{"key":"9511_CR14","unstructured":"Greg, W., & Gary, B. (2006). An introduction to Kalman Filter. TR 95-041. Course 8. Chapel Hill: University of North Carolina at Chapel Hill."},{"key":"9511_CR15","unstructured":"Haipeng, W., Tan, L., & Cheung-Chi, L. (2011). Unsupervised spoken term detection with acoustic segment model. In IEEE Proceedings of the International Conference on Speech Database and Assessments (Oriental COCOSDA) (pp. 106\u2013111)."},{"key":"9511_CR16","unstructured":"Hung, H., & Chittaranjan, G. (2010). The Idiap wolf corpus: Exploring group behaviour in a competitive role-playing game. Florence, Italy: ACM Multimedia. Retrieved January 27, 2011 from http:\/\/homepage.tudelft.nl\/3e2t5\/mmsct22567-hung.pdf ."},{"key":"9511_CR17","doi-asserted-by":"crossref","unstructured":"Jansen, A., & Van Durme, B. (2012). Indexing raw acoustic features for scalable zero resource search. In Proceedings of Interspeech","DOI":"10.21437\/Interspeech.2012-566"},{"key":"9511_CR18","first-page":"1","volume":"21","author":"T Javier","year":"2015","unstructured":"Javier, T., Doroteo, T. T., Paula, L., Laura, D., Carmen, G., Antonio, C., Julian, D., Alejandro, C., Julia, O., & Antonio, M. (2015). Spoken term detection ALBAYZIN 2014 evaluation: Overview, systems, results, and discussion. EURASIP Journal on Audio, Speech, and Music Processing, 21, 1\u201327.","journal-title":"EURASIP Journal on Audio, Speech, and Music Processing"},{"key":"9511_CR45","unstructured":"Joho, H., & Kishida, K. (2014). Overview of the NTCIR-11 SpokenQuery&Doc task. In Proceedings of NTCIR-11 (pp. 1\u20137). Tokyo, Japan: National Institute of Informatics (NII)."},{"issue":"2","key":"9511_CR19","first-page":"1214","volume":"37","author":"RR Lawrence","year":"1989","unstructured":"Lawrence, R. R., Jay, G. W., & Frank, K. S. (1989). High performance connected digit recognition using hidden Markov models. IEEE Transactions on Acoustics, Speech, and Signal Processing, 37(2), 1214\u20131225.","journal-title":"IEEE Transactions on Acoustics, Speech, and Signal Processing"},{"key":"9511_CR20","doi-asserted-by":"crossref","unstructured":"Liscombe, M., & Asif, A. (2009). A new method for instantaneous signal period identification by repetitive pattern matching. In IEEE 13th International Multitopic Conference, INMIC (pp. 1\u20135).","DOI":"10.1109\/INMIC.2009.5383086"},{"key":"9511_CR21","unstructured":"Marijn, H., Mitchell, M., & David, V. L. (2011). Unsupervised acoustic sub-word unit detection for query-by-example spoken term detection. In IEEE Proceedings of the International Conference on Acoustics, Speech and Signal Processing (ICASSP) (pp. 4436\u20134439)."},{"key":"9511_CR22","doi-asserted-by":"crossref","unstructured":"McCool, C., Marcel, S., Hadid, A., Pietik\u00e4inen, M., Mat\u011bjka, P., \u010cernock\u00fd, J., Poh, N., Kittler, J., Larcher, A., L\u00e9vy, C., Matrouf, D., Bonastre, J., Tresadern, P., & Cootes, T. (2012). Bi-modal person recognition on a mobile phone: Using mobile phone data. IEEE ICME Workshop on Hot Topics in Mobile Multimedia.","DOI":"10.1109\/ICMEW.2012.116"},{"key":"9511_CR23","unstructured":"Michael, E. (2013). Top 100 speeches, American Rhetoric, 2001. Retrieved December 12, 2013 from http:\/\/www.americanrhetoric.com\/top100speechesall.html ."},{"key":"9511_CR24","volume-title":"Kalman filtering: Theory and practice","author":"SG Mohinder","year":"1993","unstructured":"Mohinder, S. G., & Angus, P. A. (1993). Kalman filtering: Theory and practice. Upper Saddle River, NJ: Prentice-Hall, Inc."},{"key":"9511_CR25","first-page":"15","volume-title":"Kalman filtering: Theory and practice using MATLAB","author":"SG Mohinder","year":"2001","unstructured":"Mohinder, S. G., & Angus, P. A. (2001). Kalman filtering: Theory and practice using MATLAB (2nd\u00a0ed., pp.\u00a015\u201317). New York: Wiley).","edition":"2"},{"key":"9511_CR26","unstructured":"Olivier, S. (1995). On the robustness of linear discriminant analysis as a pre-processing step for noisy speech recognition. In International Conference on Acoustics, Speech, and Signal Processing, 9\u201312 May 1995 (Vol.\u00a01, pp. 125\u2013128)."},{"issue":"1","key":"9511_CR27","first-page":"840","volume":"3","author":"MM Pour","year":"2009","unstructured":"Pour, M. M., & Farokhi, F. (2009). An advanced method for speech recognition. International Scholarly and Scientific Research & Innovation, 3(1), 840\u2013845.","journal-title":"International Scholarly and Scientific Research & Innovation"},{"issue":"1","key":"9511_CR28","doi-asserted-by":"publisher","first-page":"85","DOI":"10.4236\/jbise.2010.31013","volume":"3","author":"G Ravindran","year":"2010","unstructured":"Ravindran, G., Shenbagadevi, S., & Salai, S. V. (2010). Cepstral and linear prediction techniques for improving intelligibility and audibility of impaired speech. Journal of Biomedical Science and Engineering, 3(1), 85\u201394.","journal-title":"Journal of Biomedical Science and Engineering"},{"key":"9511_CR29","unstructured":"Saha, G., Sandipan, C., & Suman, S. (2005). A new silence removal and endpoint detection algorithm for speech and speaker recognition applications. In Proceedings of the NCC."},{"key":"9511_CR30","unstructured":"Sen, Z., & Graduate, S. (2006). An energy-based adaptive voice detection approach. 8th International Conference on Signal Processing (Vol. 1). Beijing: Chinese Academy of Science"},{"key":"9511_CR31","unstructured":"Shahzadi, F., & Azra, S. (2013). Speaker recognition system using mel-frequency cepstrum coefficients, linear prediction coding and vector quantization. International Conference on Computer, Control & Communication (IC4) (pp. 1\u20135)."},{"issue":"5","key":"9511_CR32","first-page":"87","volume":"3","author":"P Sharma","year":"2013","unstructured":"Sharma, P., & Rajpoot, A. K. (2013). Automatic Identification of silence, unvoiced and voiced chunks in speech. Journal of Computer Science & Information Technology (CS & IT), 3(5), 87\u201396.","journal-title":"Journal of Computer Science & Information Technology (CS & IT)"},{"issue":"2","key":"9511_CR33","first-page":"43","volume":"10","author":"OA Soluade","year":"2010","unstructured":"Soluade, O. A. (2010). Establishment of confidence threshold for interactive voice response systems using ROC Analysis. Communications of the IIMA, 10(2), 43\u201357.","journal-title":"Communications of the IIMA"},{"key":"9511_CR34","first-page":"1","volume":"23","author":"J Tejedor","year":"2013","unstructured":"Tejedor, J., Toledano, D. T., Anguera, X., Varona, A., Hurtado, L. F., Miguel, A., & Colas, J. (2013). Query-by-example spoken term detection ALBAYZIN 2012 evaluation: Overview, systems, results, and discussion. Journal on Audio, Speech, and Music Processing, EURASIP, 23, 1\u201317.","journal-title":"Journal on Audio, Speech, and Music Processing, EURASIP"},{"issue":"1","key":"9511_CR35","doi-asserted-by":"publisher","first-page":"346","DOI":"10.1109\/TASL.2006.872615","volume":"15","author":"K Thambiratmann","year":"2007","unstructured":"Thambiratmann, K., & Sridharan, S. (2007). Rapid yet accurate speech indexing using dynamic match lattice spotting. IEEE Transactions on Audio, Speech and Language Processing, 15(1), 346\u2013357.","journal-title":"IEEE Transactions on Audio, Speech and Language Processing"},{"key":"9511_CR36","unstructured":"Timothy, J. H., Wade, S., & Christopher, W. (2009). Query-by-example spoken term detection using phonetic posteriorgram templates. In IEEE Proceedings of the Automatic Speech Recognition & Understanding (ASRU) Workshop, 17 December 2009 (pp. 421\u2013426)."},{"key":"9511_CR37","first-page":"27","volume":"6","author":"RS Tushar","year":"2014","unstructured":"Tushar, R. S., Ranjan, S., & Sabyasachi, P. (2014). Silence removal and endpoint detection of speech signal for text independent speaker identification. International Journal of Image, Graphics and Signal Processing, 6, 27\u201335.","journal-title":"International Journal of Image, Graphics and Signal Processing"},{"issue":"1","key":"9511_CR38","doi-asserted-by":"publisher","first-page":"70","DOI":"10.1109\/MIS.2017.13","volume":"32","author":"K Wasiq","year":"2017","unstructured":"Wasiq, K., & Kaya, K. (2017). An intelligent system for spoken term detection that uses belief combination. IEEE Intelligent Systems, 32(1), 70\u201379.","journal-title":"IEEE Intelligent Systems"},{"issue":"1","key":"9511_CR39","first-page":"1381","volume":"18","author":"K Wasiq","year":"2015","unstructured":"Wasiq, K., & Rob, H. (2015). Time Warped continuous speech signal matching using Kalman filter. International Journal of Speech Technology, 18(1), 1381\u20132416.","journal-title":"International Journal of Speech Technology"},{"key":"9511_CR40","doi-asserted-by":"crossref","unstructured":"Yaodong, Z., & James, R. G. (2011a). A piecewise aggregate approximation lower-bound estimate for posteriorgram-based dynamic time warping. In Proceedings of Interspeech (pp. 1909\u20131912).","DOI":"10.21437\/Interspeech.2011-355"},{"key":"9511_CR41","doi-asserted-by":"crossref","unstructured":"Yaodong, Z., & James, R. G. (2011b). An inner-product lower-bound estimate for dynamic time warping. In IEEE Proceedings of the International Conference on Acoustics, Speech and Signal Processing (ICASSP) (pp. 5660\u20135663).","DOI":"10.1109\/ICASSP.2011.5947644"},{"key":"9511_CR42","unstructured":"Yaodong, Z., Kiarash, A., & James, G. (2012). Fast spoken query detection using lower-bound dynamic time warping on graphical processing units. IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP) (pp. 5173\u20135176)."},{"issue":"2","key":"9511_CR43","doi-asserted-by":"publisher","first-page":"161","DOI":"10.1016\/0165-1684(84)90073-2","volume":"17","author":"B Yegnanarayana","year":"1984","unstructured":"Yegnanarayana, B., & Sreekumar, T. (1984). Signal dependent matching for isolated word speech recognition system. Journal of Signal Processing, 17(2), 161\u2013173.","journal-title":"Journal of Signal Processing"},{"issue":"6","key":"9511_CR44","doi-asserted-by":"publisher","first-page":"4559","DOI":"10.1121\/1.2916590","volume":"123","author":"SA Zahorian","year":"2008","unstructured":"Zahorian, S. A., & Hu, H. (2008). A spectral\/temporal method for robust fundamental frequency tracking. Journal of Acoustic Society of America, 123(6), 4559\u20134571.","journal-title":"Journal of Acoustic Society of America"}],"container-title":["International Journal of Speech Technology"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10772-018-9511-z\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-018-9511-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-018-9511-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,3]],"date-time":"2025-07-03T18:52:46Z","timestamp":1751568766000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10772-018-9511-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,4,17]]},"references-count":45,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2018,6]]}},"alternative-id":["9511"],"URL":"https:\/\/doi.org\/10.1007\/s10772-018-9511-z","relation":{},"ISSN":["1381-2416","1572-8110"],"issn-type":[{"type":"print","value":"1381-2416"},{"type":"electronic","value":"1572-8110"}],"subject":[],"published":{"date-parts":[[2018,4,17]]},"assertion":[{"value":"21 November 2017","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 April 2018","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 April 2018","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}