{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,2]],"date-time":"2026-06-02T14:39:04Z","timestamp":1780411144909,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":23,"publisher":"ACM","license":[{"start":{"date-parts":[[2008,10,20]],"date-time":"2008-10-20T00:00:00Z","timestamp":1224460800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2008,10,20]]},"DOI":"10.1145\/1452392.1452446","type":"proceedings-article","created":{"date-parts":[[2008,10,22]],"date-time":"2008-10-22T12:25:44Z","timestamp":1224678344000},"page":"257-264","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":60,"title":["A realtime multimodal system for analyzing group meetings by combining face pose tracking and speaker diarization"],"prefix":"10.1145","author":[{"given":"Kazuhiro","family":"Otsuka","sequence":"first","affiliation":[{"name":"NTT Communication Science Labs, Atsugi, Japan"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shoko","family":"Araki","sequence":"additional","affiliation":[{"name":"NTT Communication Science Labs, Kyoto, Japan"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Kentaro","family":"Ishizuka","sequence":"additional","affiliation":[{"name":"NTT Communication Science Labs, Kyoto, Japan"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Masakiyo","family":"Fujimoto","sequence":"additional","affiliation":[{"name":"NTT Communication Science Labs, Kyoto, Japan"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Martin","family":"Heinrich","sequence":"additional","affiliation":[{"name":"NTT Communication Science Labs, Atsugi, Japan"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Junji","family":"Yamato","sequence":"additional","affiliation":[{"name":"NTT Communication Science Labs, Atsugi, Japan"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2008,10,20]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/HSCMA.2008.4538680"},{"key":"e_1_3_2_1_2_1","volume-title":"Bodily Communication -","author":"Argyle M.","year":"1988","unstructured":"M. Argyle . Bodily Communication - 2 nd ed. Routledge , London and New York, 1988 . M. Argyle. Bodily Communication - 2nd ed. Routledge, London and New York, 1988.","edition":"2"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1007\/11965152_7"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2007.366328"},{"key":"e_1_3_2_1_5_1","first-page":"1","volume-title":"Proc. MLMI2007","author":"Douxchamps D.","year":"2007","unstructured":"D. Douxchamps and N. Campbell . Robust real time face tracking for the analysis of human behaviour . In Proc. MLMI2007 , pages 1 -- 10 , 2007 . D. Douxchamps and N. Campbell. Robust real time face tracking for the analysis of human behaviour. In Proc. MLMI2007, pages 1--10, 2007."},{"key":"e_1_3_2_1_6_1","first-page":"4441","volume-title":"Proc. ICASSP2008","author":"Fujimoto M.","year":"2008","unstructured":"M. Fujimoto , K. Ishizuka , and T. Nakatani . A voice activity detection based on the adaptive integration of multiple speech features and a signal decision scheme . In Proc. ICASSP2008 , pages 4441 -- 4444 , 2008 . M. Fujimoto, K. Ishizuka, and T. Nakatani. A voice activity detection based on the adaptive integration of multiple speech features and a signal decision scheme. In Proc. ICASSP2008, pages 4441--4444, 2008."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/MFI.2006.265658"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1016\/0001-6918(67)90005-4"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASSP.1976.1162830"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1007\/11677482_4"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11265-008-0250-2"},{"key":"e_1_3_2_1_13_1","first-page":"713","volume-title":"Proc. ICASSP2008","author":"Mateo Lozano O.","year":"2008","unstructured":"O. Mateo Lozano and K. Otsuka . Simultaneous and fast 3D tracking of multiple faces in video by GPU-based stream processing . In Proc. ICASSP2008 , pages 713 -- 716 , 2008 . O. Mateo Lozano and K. Otsuka. Simultaneous and fast 3D tracking of multiple faces in video by GPU-based stream processing. In Proc. ICASSP2008, pages 713--716, 2008."},{"key":"e_1_3_2_1_14_1","first-page":"16","volume-title":"Proc. MVA2007","author":"Matsusaka Y.","year":"2007","unstructured":"Y. Matsusaka , H. Asoh , and F. Asano . Multi human trajectory estimation using stochastic sampling and its application to meeting recognition . In Proc. MVA2007 , pages 16 -- 18 , 2007 . Y. Matsusaka, H. Asoh, and F. Asano. Multi human trajectory estimation using stochastic sampling and its application to meeting recognition. In Proc. MVA2007, pages 16--18, 2007."},{"key":"e_1_3_2_1_15_1","volume-title":"NIST","author":"NIST Speech Group","year":"2007","unstructured":"NIST Speech Group . Spring 2007 (RT-07) rich transcription meeting recognition evaluation plan. Technical Report rt07-meeting-eval-plan-v2 , NIST , 2007. NIST Speech Group. Spring 2007 (RT-07) rich transcription meeting recognition evaluation plan. Technical Report rt07-meeting-eval-plan-v2, NIST, 2007."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/1088463.1088497"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-85853-9_2"},{"key":"e_1_3_2_1_18_1","first-page":"949","volume-title":"Proc. ICME'06","author":"Otsuka K.","year":"2006","unstructured":"K. Otsuka , J. Yamato , and H. Murase . Conversation scene analysis with dynamic Bayesian network based on visual head tracking . In Proc. ICME'06 , pages 949 -- 952 , 2006 . K. Otsuka, J. Yamato, and H. Murase. Conversation scene analysis with dynamic Bayesian network based on visual head tracking. In Proc. ICME'06, pages 949--952, 2006."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/HSCMA.2008.4538700"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1007\/11965152_8"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNN.2002.1021893"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1023\/B:VISI.0000013087.49260.fb"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/1180995.1181050"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.5555\/1018430.1021383"}],"event":{"name":"ICMI '08: INTERNATIONAL CONFERENCE ON MULTIMODAL INTERFACES","location":"Chania Crete Greece","acronym":"ICMI '08","sponsor":["ACM Association for Computing Machinery","SIGCHI ACM Special Interest Group on Computer-Human Interaction"]},"container-title":["Proceedings of the 10th international conference on Multimodal interfaces"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/1452392.1452446","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/1452392.1452446","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T13:56:33Z","timestamp":1750254993000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/1452392.1452446"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2008,10,20]]},"references-count":23,"alternative-id":["10.1145\/1452392.1452446","10.1145\/1452392"],"URL":"https:\/\/doi.org\/10.1145\/1452392.1452446","relation":{},"subject":[],"published":{"date-parts":[[2008,10,20]]},"assertion":[{"value":"2008-10-20","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}