{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T04:26:58Z","timestamp":1750220818616,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":28,"publisher":"ACM","license":[{"start":{"date-parts":[[2019,9,9]],"date-time":"2019-09-09T00:00:00Z","timestamp":1567987200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2019,9,9]]},"DOI":"10.1145\/3341162.3344861","type":"proceedings-article","created":{"date-parts":[[2019,9,11]],"date-time":"2019-09-11T16:16:21Z","timestamp":1568218581000},"page":"468-473","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Audio-visual TED corpus"],"prefix":"10.1145","author":[{"given":"Guan-Lin","family":"Chao","sequence":"first","affiliation":[{"name":"Carnegie Mellon University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chih Chi","family":"Hu","sequence":"additional","affiliation":[{"name":"Carnegie Mellon University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bing","family":"Liu","sequence":"additional","affiliation":[{"name":"Carnegie Mellon University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"John Paul","family":"Shen","sequence":"additional","affiliation":[{"name":"Carnegie Mellon University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ian","family":"Lane","sequence":"additional","affiliation":[{"name":"Carnegie Mellon University"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2019,9,9]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"The OpenCV Library. Dr. Dobb's Journal of Software Tools","author":"Bradski Gary","year":"2000","unstructured":"Gary Bradski . 2000. The OpenCV Library. Dr. Dobb's Journal of Software Tools ( 2000 ). Gary Bradski. 2000. The OpenCV Library. Dr. Dobb's Journal of Software Tools (2000)."},{"key":"e_1_3_2_1_2_1","volume-title":"An audio-visual corpus for speech perception and automatic speech recognition. Journal of the Acoustical Society of America","author":"Cooke Martin","year":"2006","unstructured":"Martin Cooke , Jon Barker , Stuart Cunningham , and Xu Shao . 2006. An audio-visual corpus for speech perception and automatic speech recognition. Journal of the Acoustical Society of America ( 2006 ). Martin Cooke, Jon Barker, Stuart Cunningham, and Xu Shao. 2006. An audio-visual corpus for speech perception and automatic speech recognition. Journal of the Acoustical Society of America (2006)."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10844-016-0438-z"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.5244\/C.28.65"},{"volume-title":"ffmpeg version 2.8.11--0","author":"The","key":"e_1_3_2_1_5_1","unstructured":"The FFmpeg developers. 2019. ffmpeg version 2.8.11--0 . http:\/\/ffmpeg.org\/. Accessed: 2019-06-14. The FFmpeg developers. 2019. ffmpeg version 2.8.11--0. http:\/\/ffmpeg.org\/. Accessed: 2019-06-14."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-014-0733-5"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46487-9_6"},{"key":"e_1_3_2_1_8_1","volume-title":"Long Short-Term Memory. Neural Computation","author":"Hochreiter Sepp","year":"1997","unstructured":"Sepp Hochreiter and J\u00fcrgen Schmidhuber . 1997. Long Short-Term Memory. Neural Computation ( 1997 ). Sepp Hochreiter and J\u00fcrgen Schmidhuber. 1997. Long Short-Term Memory. Neural Computation (1997)."},{"key":"e_1_3_2_1_9_1","volume-title":"Dlib-ml: A Machine Learning Toolkit. Journal of Machine Learning Research","author":"King Davis E.","year":"2009","unstructured":"Davis E. King . 2009 . Dlib-ml: A Machine Learning Toolkit. Journal of Machine Learning Research (2009). Davis E. King. 2009. Dlib-ml: A Machine Learning Toolkit. Journal of Machine Learning Research (2009)."},{"key":"e_1_3_2_1_10_1","volume-title":"Real-world Database for Facial Landmark Localization. In ICCV Workshop on Benchmarking Facial Image Analysis Technologies.","author":"Koestinger Martin","year":"2011","unstructured":"Martin Koestinger , Paul Wohlhart , Peter M Roth , and Horst Bischof . 2011 . Annotated Facial Landmarks in the Wild: A Large-scale , Real-world Database for Facial Landmark Localization. In ICCV Workshop on Benchmarking Facial Image Analysis Technologies. Martin Koestinger, Paul Wohlhart, Peter M Roth, and Horst Bischof. 2011. Annotated Facial Landmarks in the Wild: A Large-scale, Real-world Database for Facial Landmark Localization. In ICCV Workshop on Benchmarking Facial Image Analysis Technologies."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2004-424"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICMEW.2012.116"},{"key":"e_1_3_2_1_13_1","volume-title":"International Conference on Audio and Video-based Biometric Person Authentication.","author":"Messer Kieron","year":"1999","unstructured":"Kieron Messer , Jiri Matas , Josef Kittler , Juergen Luettin , and Gilbert Maitre . 1999 . XM2VTSDB: The extended M2VTS database . In International Conference on Audio and Video-based Biometric Person Authentication. Kieron Messer, Jiri Matas, Josef Kittler, Juergen Luettin, and Gilbert Maitre. 1999. XM2VTSDB: The extended M2VTS database. In International Conference on Audio and Video-based Biometric Person Authentication."},{"key":"e_1_3_2_1_14_1","unstructured":"Javier R Movellan. 1995. Visual speech recognition with stochastic networks. In Advances in Neural Information Processing Systems.   Javier R Movellan. 1995. Visual speech recognition with stochastic networks. In Advances in Neural Information Processing Systems ."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICIP.2014.7025068"},{"key":"e_1_3_2_1_16_1","volume-title":"Deep Face Recognition. In British Machine Vision Conference (BMVC).","author":"Parkhi Omkar M","year":"2015","unstructured":"Omkar M Parkhi , Andrea Vedaldi , Andrew Zisserman , 2015 . Deep Face Recognition. In British Machine Vision Conference (BMVC). Omkar M Parkhi, Andrea Vedaldi, Andrew Zisserman, et al. 2015. Deep Face Recognition. In British Machine Vision Conference (BMVC)."},{"key":"e_1_3_2_1_17_1","volume-title":"International Conference on Acoustics, Speech and Signal Processing (ICASSP).","author":"Patterson Eric K","year":"2002","unstructured":"Eric K Patterson , Sabri Gurbuz , Zekeriya Tufekci , and John N Gowdy . 2002 . CUAVE: A new audio-visual database for multimodal human-computer interface research . In International Conference on Acoustics, Speech and Signal Processing (ICASSP). Eric K Patterson, Sabri Gurbuz, Zekeriya Tufekci, and John N Gowdy. 2002. CUAVE: A new audio-visual database for multimodal human-computer interface research. In International Conference on Acoustics, Speech and Signal Processing (ICASSP)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/BTAS.2008.4699323"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"crossref","unstructured":"Joseph Redmon and Ali Farhadi. 2017. YOLO9000: Better Faster Stronger. In Computer Vision and Pattern Recognition (CVPR).  Joseph Redmon and Ali Farhadi. 2017. YOLO9000: Better Faster Stronger. In Computer Vision and Pattern Recognition (CVPR) .","DOI":"10.1109\/CVPR.2017.690"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.218"},{"key":"e_1_3_2_1_21_1","volume-title":"International Conference on Language Resources and Evaluation (LREC).","author":"Rousseau Anthony","year":"2012","unstructured":"Anthony Rousseau , Paul Del\u00e9glise , and Yannick Est\u00e8ve . 2012 . TED-LIUM: an Automatic Speech Recognition dedicated corpus . In International Conference on Language Resources and Evaluation (LREC). Anthony Rousseau, Paul Del\u00e9glise, and Yannick Est\u00e8ve. 2012. TED-LIUM: an Automatic Speech Recognition dedicated corpus. In International Conference on Language Resources and Evaluation (LREC)."},{"key":"e_1_3_2_1_22_1","volume-title":"Enhancing the TED-LIUM Corpus with Selected Data for Language Modeling and More TED Talks. In International Conference on Language Resources and Evaluation (LREC).","author":"Rousseau A","year":"2014","unstructured":"A Rousseau , P Del\u00e9glise , and Y Est\u00e8ve . 2014 . Enhancing the TED-LIUM Corpus with Selected Data for Language Modeling and More TED Talks. In International Conference on Language Resources and Evaluation (LREC). A Rousseau, P Del\u00e9glise, and Y Est\u00e8ve. 2014. Enhancing the TED-LIUM Corpus with Selected Data for Language Modeling and More TED Talks. In International Conference on Language Resources and Evaluation (LREC)."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0816-y"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.imavis.2016.01.002"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"crossref","unstructured":"Florian Schroff Dmitry Kalenichenko and James Philbin. 2015. FaceNet: A Unified Embedding for Face Recognition and Clustering. In Computer Vision and Pattern Recognition (CVPR).  Florian Schroff Dmitry Kalenichenko and James Philbin. 2015. FaceNet: A Unified Embedding for Face Recognition and Clustering. In Computer Vision and Pattern Recognition (CVPR) .","DOI":"10.1109\/CVPR.2015.7298682"},{"key":"e_1_3_2_1_26_1","volume-title":"An Overview of the Tesseract OCR Engine. In International Conference on Document Analysis and Recognition (ICDAR).","author":"Smith Ray","year":"2007","unstructured":"Ray Smith . 2007 . An Overview of the Tesseract OCR Engine. In International Conference on Document Analysis and Recognition (ICDAR). Ray Smith. 2007. An Overview of the Tesseract OCR Engine. In International Conference on Document Analysis and Recognition (ICDAR)."},{"key":"e_1_3_2_1_27_1","unstructured":"Fred Weinhaus. 2019. ImageMagick. http:\/\/www.fmwconcepts.com\/imagemagick\/. Accessed: 2019-06-14.  Fred Weinhaus. 2019. ImageMagick. http:\/\/www.fmwconcepts.com\/imagemagick\/. Accessed: 2019-06-14."},{"key":"e_1_3_2_1_28_1","volume-title":"Chen Change Loy, and Xiaoou Tang","author":"Yang Shuo","year":"2016","unstructured":"Shuo Yang , Ping Luo , Chen Change Loy, and Xiaoou Tang . 2016 . WIDER FACE : A Face Detection Benchmark. In Computer Vision and Pattern Recognition (CVPR) . Shuo Yang, Ping Luo, Chen Change Loy, and Xiaoou Tang. 2016. WIDER FACE: A Face Detection Benchmark. In Computer Vision and Pattern Recognition (CVPR)."}],"event":{"name":"UbiComp '19: The 2019 ACM International Joint Conference on Pervasive and Ubiquitous Computing","sponsor":["SIGMOBILE ACM Special Interest Group on Mobility of Systems, Users, Data and Computing","SIGCHI ACM Special Interest Group on Computer-Human Interaction"],"location":"London United Kingdom","acronym":"UbiComp '19"},"container-title":["Adjunct Proceedings of the 2019 ACM International Joint Conference on Pervasive and Ubiquitous Computing and Proceedings of the 2019 ACM International Symposium on Wearable Computers"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3341162.3344861","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3341162.3344861","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T23:13:09Z","timestamp":1750201989000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3341162.3344861"}},"subtitle":["enhancing the TED-LIUM corpus with facial information, contextual text and object recognition"],"short-title":[],"issued":{"date-parts":[[2019,9,9]]},"references-count":28,"alternative-id":["10.1145\/3341162.3344861","10.1145\/3341162"],"URL":"https:\/\/doi.org\/10.1145\/3341162.3344861","relation":{},"subject":[],"published":{"date-parts":[[2019,9,9]]},"assertion":[{"value":"2019-09-09","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}