{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T06:31:22Z","timestamp":1777444282617,"version":"3.51.4"},"reference-count":46,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2013,8,28]],"date-time":"2013-08-28T00:00:00Z","timestamp":1377648000000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Int J Speech Technol"],"published-print":{"date-parts":[[2014,3]]},"DOI":"10.1007\/s10772-013-9206-4","type":"journal-article","created":{"date-parts":[[2013,8,27]],"date-time":"2013-08-27T09:15:27Z","timestamp":1377594927000},"page":"65-74","source":"Crossref","is-referenced-by-count":2,"title":["Film segmentation and indexing using autoassociative neural networks"],"prefix":"10.1007","volume":"17","author":[{"given":"K. Sreenivasa","family":"Rao","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dipanjan","family":"Nandi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shashidhar G.","family":"Koolagudi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2013,8,28]]},"reference":[{"key":"9206_CR1","volume-title":"Movie content analysis indexing and skimming","author":"Y. Li","year":"2003","unstructured":"Li, Y., Narayanan, S., & Kuo, C. C. J. (2003). Movie content analysis indexing and skimming (Vol.\u00a06). Dordrecht: Kluwer Academic. Video Mining, Chap. 5."},{"key":"9206_CR2","doi-asserted-by":"crossref","first-page":"459","DOI":"10.1016\/S0893-6080(02)00019-9","volume":"15","author":"B. Yegnanarayana","year":"2002","unstructured":"Yegnanarayana, B., & Kishore, S. P. (2002). AANN an alternative to GMM for pattern recognition. Neural Networks, 15, 459\u2013469.","journal-title":"Neural Networks"},{"key":"9206_CR3","volume-title":"Neural networks: a comprehensive foundation","author":"S. Haykin","year":"1999","unstructured":"Haykin, S. (1999). Neural networks: a comprehensive foundation. New Delhi: Pearson Education Aisa, Inc."},{"key":"9206_CR4","volume-title":"Artificial neural networks","author":"B. Yegnanarayana","year":"1999","unstructured":"Yegnanarayana, B. (1999). Artificial neural networks. New Delhi: Prentice-Hall."},{"issue":"1","key":"9206_CR5","doi-asserted-by":"crossref","first-page":"474","DOI":"10.1016\/j.csl.2009.03.003","volume":"24","author":"K. S. Rao","year":"2010","unstructured":"Rao, K. S. (2010). Voice conversion by mapping the speaker-specific features using pitch synchronous approach. Computer Speech & Language, 24(1), 474\u2013494.","journal-title":"Computer Speech & Language"},{"key":"9206_CR6","volume-title":"INTERSPEECH-2010","author":"S. H. R. Mallidi","year":"2010","unstructured":"Mallidi, S. H. R., Prahallad, K., Gangashetty, S. V., & Yegnanarayana, B. (2010). Significance of pitch synchronous analysis for speaker recognition using AANN models. In INTERSPEECH-2010, Makuhari, Japan, Sept. 2010."},{"key":"9206_CR7","doi-asserted-by":"crossref","first-page":"305","DOI":"10.1109\/ICISIP.2004.1287672","volume-title":"The international conference on intelligent sensing and information processing 2004 (ICISIP 2004)","author":"A. Bajpai","year":"2004","unstructured":"Bajpai, A., & Yegnanarayana, B. (2004). Exploring features for audio clip classification using LP residual and AANN models. In The international conference on intelligent sensing and information processing 2004 (ICISIP 2004), Chennai, India, Jan. 2004 (pp. 305\u2013310)."},{"key":"9206_CR8","first-page":"409","volume-title":"Proc. IEEE int. conf. acoust., speech, signal processing","author":"B. Yegnanarayana","year":"2001","unstructured":"Yegnanarayana, B., Reddy, K. S., & Kishore, S. P. (2001). Source and system features for speaker recognition using AANN models. In Proc. IEEE int. conf. acoust., speech, signal processing, Salt Lake City, Utah, USA, May 2001 (pp. 409\u2013412)."},{"key":"9206_CR9","first-page":"317","volume-title":"International conference on intelligent sensing and information processing","author":"L. Mary","year":"2004","unstructured":"Mary, L., & Yegnanarayana, B. (2004). Autoassociative neural network models for language identification. In International conference on intelligent sensing and information processing (pp. 317\u2013320). New York: IEEE Press. doi: 10.1109\/ICISIP.2004.1287674 ."},{"key":"9206_CR10","volume-title":"Int. conf. on cognitive and neural systems (ICCNS)","author":"L. Mary","year":"2004","unstructured":"Mary, L., Rao, K. S., Gangashetty, S., & Yegnanarayana, B. (2004). Neural network models for capturing duration and intonation knowledge for language and speaker identification. In Int. conf. on cognitive and neural systems (ICCNS), Boston, MA, USA, May 2004."},{"key":"9206_CR11","doi-asserted-by":"crossref","first-page":"783","DOI":"10.1007\/s12046-011-0047-z","volume":"36","author":"K. S. Rao","year":"2011","unstructured":"Rao, K. S. (2011). Role of neural network models for developing speech systems. Sadhana (Springer), 36, 783\u2013836.","journal-title":"Sadhana (Springer)"},{"key":"9206_CR12","unstructured":"Rao, K. S. (2008). Acquisition and incorporation prosody knowledge for speech systems in Indian languages. Ph.D. thesis, Dept. of Computer Science and Engineering, Indian Institute of Technology, Madras, Chennai, India, May 2008."},{"key":"9206_CR13","volume-title":"2nd int. conf. intelligent sensing and information processing (ICISIP-2005)","author":"L. Mary","year":"2005","unstructured":"Mary, L., Rao, K. S., & Yegnanarayana, B. (2005). Neural network classifiers for language identification using syntactic and prosodic features. In 2nd int. conf. intelligent sensing and information processing (ICISIP-2005), Chennai, India, Jan. 2005."},{"key":"9206_CR14","doi-asserted-by":"crossref","first-page":"282","DOI":"10.1016\/j.csl.2006.06.003","volume":"21","author":"K. S. Rao","year":"2007","unstructured":"Rao, K. S., & Yegnanarayana, B. (2007). Modeling durations of syllables using neural networks. Computer Speech & Language, 21, 282\u2013295.","journal-title":"Computer Speech & Language"},{"key":"9206_CR15","volume-title":"Proc. IEEE int. conf. acoust., speech, signal processing","author":"L. Mary","year":"2004","unstructured":"Mary, L., Rao, K. S., Gangashetty, S., & Yegnanarayana, B. (2004). Modeling syllable duration in Indian languages using neural networks. In Proc. IEEE int. conf. acoust., speech, signal processing, Montreal, Quebec, Canada, May 2004."},{"key":"9206_CR16","doi-asserted-by":"crossref","first-page":"71","DOI":"10.1007\/978-3-540-75398-8_4","volume-title":"Speech, audio, image and biomedical signal processing using neural networks","author":"K. S. Rao","year":"2008","unstructured":"Rao, K. S. (2008). Modeling supra-segmental features of syllables using neural networks. In Speech, audio, image and biomedical signal processing using neural networks (pp. 71\u201395). Berlin: Springer."},{"key":"9206_CR17","series-title":"Lecture notes in computer science","doi-asserted-by":"crossref","first-page":"1179","DOI":"10.1007\/978-3-540-30499-9_183","volume-title":"Neural information processing","author":"K. S. Rao","year":"2004","unstructured":"Rao, K. S., & Yegnanarayana, B. (2004). Two-stage duration model for Indian languages using neural networks. In Lecture notes in computer science: Vol.\u00a03316. Neural information processing (pp. 1179\u20131185). Berlin: Springer."},{"issue":"10","key":"9206_CR18","doi-asserted-by":"crossref","first-page":"1105","DOI":"10.1016\/j.csl.2013.02.003","volume":"27","author":"V. R. Reddy","year":"2013","unstructured":"Reddy, V. R., & Rao, K. S. (2013). Two-stage intonation modeling using feedforward neural networks for syllable based text-to-speech synthesis. Computer Speech & Language, 27(10), 1105\u20131126.","journal-title":"Computer Speech & Language"},{"key":"9206_CR19","volume-title":"IEEE international conference on signal processing and communication (SPCOM)","author":"S. G. Koolagudi","year":"2010","unstructured":"Koolagudi, S. G., Reddy, R., & Rao, K. S. (2010). Emotion recognition from speech signal using epoch parameters. In IEEE international conference on signal processing and communication (SPCOM), IISc Bangalore, India, July 2010."},{"key":"9206_CR20","volume-title":"12th int. conf. on cognitive and neural systems (ICCNS)","author":"S. G. Koolagudi","year":"2008","unstructured":"Koolagudi, S. G., & Rao, K. S. (2008). Neural network models for capturing prosodic knowledge for emotion recognition. In 12th int. conf. on cognitive and neural systems (ICCNS), Boston, MA, USA, May 2008."},{"key":"9206_CR21","first-page":"520","volume-title":"5th international conference on knowledge based computer systems (KBCS-2004)","author":"K. S. Rao","year":"2004","unstructured":"Rao, K. S., & Yegnanarayana, B. (2004). Neural network models for text-to-speech synthesis. In 5th international conference on knowledge based computer systems (KBCS-2004), Hyderabad, India, Dec. 2004 (pp. 520\u2013530)."},{"key":"9206_CR22","doi-asserted-by":"crossref","first-page":"13181","DOI":"10.1016\/j.eswa.2011.04.129","volume":"38","author":"K. S. Rao","year":"2011","unstructured":"Rao, K. S., Saroj, V. K., Maity, S., & Koolagudi, S. G. (2011). Recognition of emotions from video using neural network models. Expert Systems with Applications, 38, 13181\u201313185.","journal-title":"Expert Systems with Applications"},{"issue":"3","key":"9206_CR23","doi-asserted-by":"crossref","first-page":"335","DOI":"10.1007\/s10772-012-9148-2","volume":"15","author":"K. S. Rao","year":"2012","unstructured":"Rao, K. S., Yadav, J., Sarkar, S., Koolagudi, S. G., & Vuppala, A. K. (2012). Neural network based feature transformation for emotion independent speaker identification. International Journal of Speech Technology, 15(3), 335\u2013349.","journal-title":"International Journal of Speech Technology"},{"key":"9206_CR24","series-title":"Lecture notes in computer science","doi-asserted-by":"crossref","first-page":"479","DOI":"10.1007\/978-3-540-77046-6_59","volume-title":"Pattern recognition and machine intelligence","author":"K. S. Rao","year":"2007","unstructured":"Rao, K. S., Laskar, R. H., & Koolagudi, S. G. (2007). Voice transformation by mapping the features at syllable level. In Lecture notes in computer science: Vol.\u00a04815. Pattern recognition and machine intelligence (pp. 479\u2013486)."},{"key":"9206_CR25","first-page":"1338","volume-title":"Proceedings of the IEEE","author":"J. Makhoul","year":"2000","unstructured":"Makhoul, J., Kubala, F., Leek, T., Liu, D., Nguyen, L., Schwartz, R., & Srivastava, A. (2000). Speech and language technologies for audio indexing and retrieval. In Proceedings of the IEEE (Vol.\u00a088, pp. 1338\u20131353)."},{"key":"9206_CR26","first-page":"452","volume-title":"Proc. IEEE int. conf. acoust., speech, signal processing","author":"S. Johnson","year":"2000","unstructured":"Johnson, S., & Woodland, P. C. (2000). A method for direct audio search with applications to indexing and retrieval. In Proc. IEEE int. conf. acoust., speech, signal processing (Vol.\u00a01, pp. 452\u2013455)."},{"key":"9206_CR27","volume-title":"7th international workshop on multimedia data mining","author":"D. Brezeale","year":"2006","unstructured":"Brezeale, D., & Cook, D. J. (2006). Using closed captions and visual features to classify movies by genre. In 7th international workshop on multimedia data mining."},{"key":"9206_CR28","first-page":"295","volume-title":"ACM international conference on multimedia","author":"S. Fischer","year":"1995","unstructured":"Fischer, S., Lienhart, R., & Effelsberg, W. (1995). Automatic recognition of film genres. In ACM international conference on multimedia (pp. 295\u2013304)."},{"key":"9206_CR29","first-page":"26","volume":"3","author":"H.-Y. Huang","year":"2008","unstructured":"Huang, H.-Y., Shih, W.-S., & Hsu, W.-H. (2008). A film classifier based on low-level visual features. Journal of Multimedia, 3, 26\u201333.","journal-title":"Journal of Multimedia"},{"key":"9206_CR30","doi-asserted-by":"crossref","first-page":"145","DOI":"10.1023\/A:1011139631724","volume":"42","author":"A. Oliva","year":"2001","unstructured":"Oliva, A., & Torralba, A. (2001). Modeling the shape of the scene: A holistic representation of the spatial envelope. International Journal of Computer Vision, 42, 145\u2013175.","journal-title":"International Journal of Computer Vision"},{"key":"9206_CR31","volume-title":"ACM international conference on multimedia","author":"C. Ramachandran","year":"2009","unstructured":"Ramachandran, C., Malik, R., Jin, X., Gao, J., Nahrstedt, K., & Han, J. (2009). Videomule: a consensus learning approach to multi-label classification from noisy user-generated videos. In ACM international conference on multimedia."},{"key":"9206_CR32","volume-title":"International conference on pattern recognition (ICPR)","author":"Z. Rasheed","year":"2002","unstructured":"Rasheed, Z., & Shah, M. (2002). Movie genre classification by exploiting audio-visual features of previews. In International conference on pattern recognition (ICPR)."},{"key":"9206_CR33","doi-asserted-by":"crossref","first-page":"52","DOI":"10.1109\/TCSVT.2004.839993","volume":"15","author":"Z. Rasheed","year":"2003","unstructured":"Rasheed, Z., Sheikh, Y., & Shah, M. (2003). On the use of computable features for film classification. IEEE Transactions on Circuits and Systems for Video Technology, 15, 52\u201364.","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"9206_CR34","doi-asserted-by":"crossref","first-page":"2693","DOI":"10.21437\/Eurospeech.2001-630","volume-title":"Proc. Eurospeech","author":"M. Roach","year":"2001","unstructured":"Roach, M., & Mason, J. (2001). Classification of video genre using audio. In Proc. Eurospeech (Vol.\u00a015, pp. 2693\u20132696)."},{"key":"9206_CR35","volume-title":"Proc. IEEE conf. on computer vision and pattern recognition (CVPR)","author":"K. E. A. Sande van de","year":"2008","unstructured":"van de Sande, K. E. A., Gevers, T., & Snoek, C. G. M. (2008). Evaluation of color descriptors for object and scene recognition. In Proc. IEEE conf. on computer vision and pattern recognition (CVPR)."},{"key":"9206_CR36","volume-title":"Proc. IEEE conf. on computer vision and pattern recognition (CVPR)","author":"Z. Wang","year":"2010","unstructured":"Wang, Z., Zhao, M., Song, Y., Kumar, S., & Li, B. (2010). Youtubecat: Learning to categorize wild web videos. In Proc. IEEE conf. on computer vision and pattern recognition (CVPR)."},{"key":"9206_CR37","first-page":"196","volume-title":"16th IPPR conference on computer vision, graphics and image processing (CVGIP 2003)","author":"Y.-K. Wang","year":"2003","unstructured":"Wang, Y.-K., & Chang, C.-Y. (2003). Movie scene classification using hidden Markov model. In 16th IPPR conference on computer vision, graphics and image processing (CVGIP 2003), Kinmen, ROC, Aug. 2003 (pp. 196\u2013202)."},{"key":"9206_CR38","doi-asserted-by":"crossref","first-page":"375","DOI":"10.1109\/ICPR.1996.546973","volume-title":"International conference on pattern recognition","author":"M. M. Yeung","year":"1996","unstructured":"Yeung, M. M., & Liu, B. L. (1996). Time-constrained clustering for segmentation of video into story unit. In International conference on pattern recognition (pp. 375\u2013380)."},{"key":"9206_CR39","unstructured":"Delezoide, B. (2006). Multimedia movie segmentation using low-level and semantic features."},{"key":"9206_CR40","volume-title":"ACM international conference on multimedia","author":"H. Zhou","year":"2010","unstructured":"Zhou, H., Hermans, T., Karandikar, A. V., & Rehg, J. M. (2010). Movie genre classification via scene categorization. In ACM international conference on multimedia, Firenze, Italy, Oct. 2010."},{"key":"9206_CR41","volume-title":"Proceedings of the IEEE international conference on multimedia and Expo (ICME 05)","author":"B. Delezoide","year":"2005","unstructured":"Delezoide, B. (2005). Hierarchical film segmentation using audio and visual similarity. In Proceedings of the IEEE international conference on multimedia and Expo (ICME 05)."},{"key":"9206_CR42","volume-title":"17th international conference on pattern recognition","author":"Y. Zhai","year":"2004","unstructured":"Zhai, Y., Rasheed, Z., & Shah, M. (2004). Finite state machines in movie scene classification. In 17th international conference on pattern recognition, Cambridge, UK."},{"key":"9206_CR43","volume-title":"Fundamentals of speech recognition","author":"L. R. Rabiner","year":"1993","unstructured":"Rabiner, L. R., & Juang, B. H. (1993). Fundamentals of speech recognition. Englewood Cliffs: Prentice Hall."},{"key":"9206_CR44","volume-title":"Discrete-time speech signal processing: principles and practice","author":"T. F. Quatieri","year":"2001","unstructured":"Quatieri, T. F. (2001). Discrete-time speech signal processing: principles and practice. Englewood Cliffs: Prentice Hall."},{"issue":"6","key":"9206_CR45","doi-asserted-by":"crossref","first-page":"582","DOI":"10.1007\/BF02943243","volume":"16","author":"F. Zheng","year":"2001","unstructured":"Zheng, F., Zhang, G., & Song, Z. (2001). Comparison of different implementations of MFCC. Journal of Computer Science and Technology, 16(6), 582\u2013589.","journal-title":"Journal of Computer Science and Technology"},{"key":"9206_CR46","volume-title":"Engineering statistics","author":"R. V. Hogg","year":"1987","unstructured":"Hogg, R. V., & Ledolter, J. (1987). Engineering statistics. New York: Macmillan."}],"container-title":["International Journal of Speech Technology"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-013-9206-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10772-013-9206-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-013-9206-4","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,7,4]],"date-time":"2023-07-04T03:49:42Z","timestamp":1688442582000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10772-013-9206-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2013,8,28]]},"references-count":46,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2014,3]]}},"alternative-id":["9206"],"URL":"https:\/\/doi.org\/10.1007\/s10772-013-9206-4","relation":{},"ISSN":["1381-2416","1572-8110"],"issn-type":[{"value":"1381-2416","type":"print"},{"value":"1572-8110","type":"electronic"}],"subject":[],"published":{"date-parts":[[2013,8,28]]}}}