{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,7,4]],"date-time":"2025-07-04T05:54:25Z","timestamp":1751608465743},"reference-count":43,"publisher":"IEEE","license":[{"start":{"date-parts":[[2019,12,1]],"date-time":"2019-12-01T00:00:00Z","timestamp":1575158400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2019,12,1]],"date-time":"2019-12-01T00:00:00Z","timestamp":1575158400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2019,12,1]],"date-time":"2019-12-01T00:00:00Z","timestamp":1575158400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2019,12]]},"DOI":"10.1109\/asru46091.2019.9003816","type":"proceedings-article","created":{"date-parts":[[2020,2,21]],"date-time":"2020-02-21T07:01:33Z","timestamp":1582268493000},"page":"1062-1069","source":"Crossref","is-referenced-by-count":4,"title":["Improving Speech-Based End-of-Turn Detection Via Cross-Modal Representation Learning with Punctuated Text Data"],"prefix":"10.1109","author":[{"given":"Ryo","family":"Masumura","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mana","family":"Ihori","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tomohiro","family":"Tanaka","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Atsushi","family":"Ando","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ryo","family":"Ishii","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Takanobu","family":"Oba","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ryuichiro","family":"Higashinaka","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-82"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683869"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1746"},{"journal-title":"Cycle-consistency training for end-to-end speech recognition","year":"2018","author":"hori","key":"ref32"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2017.8268950"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2018.8639589"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01246-5_42"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W18-6317"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1021"},{"key":"ref34","article-title":"Semi-supervised end-to-end speech recognition using text-to-speech and autoen-coders","author":"karita","year":"2019","journal-title":"Proc International Conference on Acoustics Speech and Signal Processing (ICASSP)"},{"key":"ref10","first-page":"17","article-title":"Modelling turn-taking in human conversations","author":"guntakandla","year":"2015","journal-title":"In AAAI Spring Symposium on Turn-taking and Coordination in Human-Machine Interaction"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-2341"},{"key":"ref11","first-page":"2061","article-title":"In the speaker done yet? faster and more accurate end-of-utterance detection using prosody in human-computer dialog","author":"ferrer","year":"2002","journal-title":"Proc International Conference on Spoken Language Processing (ICSLP)"},{"key":"ref12","first-page":"11","article-title":"Towards incremental end-of-utterance detection in dialogue systems","author":"atterer","year":"2008","journal-title":"Proc International Conference on Computational Linguistics (COLING)"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W17-5527"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-651"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W18-5024"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2006.878255"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2009.4960690"},{"key":"ref18","first-page":"683","article-title":"LSTM for punctuation restoration in speech transcripts","author":"tilk","year":"2015","journal-title":"Proc Annual Conference of the International Speech Communication Association (INTERSPEECH)"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-1517"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/IALP.2018.8629114"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2001.940814"},{"journal-title":"Wisebe Window-based sentence boundary evaluation","year":"2018","author":"gonzalez-gallardo","key":"ref27"},{"key":"ref3","first-page":"104","article-title":"Ten challenges in highly-interactive dialog systems","author":"ward","year":"2015","journal-title":"In AAAI Spring Symposium on Turn-taking and Coordination in Human-Machine Interaction"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1016\/S0167-6393(00)00028-5"},{"key":"ref29","first-page":"426","article-title":"Back-translation-style data augmentation for end-to-end ASR","author":"hayashi","year":"2018","journal-title":"Proc IEEE\/ACL Workshop Spoken Lang Technol (SLT)"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1177\/002383099804100404"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2010.10.003"},{"key":"ref7","first-page":"17","article-title":"From reaction to prediction: Experiments with computational models of turn taking","author":"schlangen","year":"2006","journal-title":"Proc Annual Conference of the International Speech Communication Association (INTERSPEECH)"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2014.02.002"},{"key":"ref9","first-page":"861","article-title":"Learning decison trees to determine turn-taking by spoken dialogue systems","author":"sato","year":"2002","journal-title":"Proc International Conference on Spoken Language Processing (IC-SLP)"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.2307\/412243"},{"key":"ref20","first-page":"654","article-title":"Punctuation prediction for unsegmented transcript based on word vector","author":"che","year":"2016","journal-title":"Proc Int Conference on Language Resources and Evaluation (LREC)"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682260"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1096"},{"key":"ref42","first-page":"947","article-title":"Spon-taneous speech corpus of Japanese","author":"maekawa","year":"2000","journal-title":"Proc Int Conference on Language Resources and Evaluation (LREC)"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1016\/S0167-6393(00)00028-5"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6637694"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/ICSLP.1996.607773"},{"journal-title":"Evaluating word embeddings for sentence boundary detection in speech transcripts","year":"2017","author":"treviso","key":"ref26"},{"key":"ref43","first-page":"220","article-title":"Intelligent selection of language model training data","author":"moore","year":"2010","journal-title":"Proc Annu Meeting Association of Computational Linguists (ACL)"},{"key":"ref25","first-page":"985","article-title":"Sen-tence boundary detection: A long solved problem?","author":"read","year":"2012","journal-title":"Proc International Conference on Computational Linguistics (COLING)"}],"event":{"name":"2019 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU)","start":{"date-parts":[[2019,12,14]]},"location":"SG, Singapore","end":{"date-parts":[[2019,12,18]]}},"container-title":["2019 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/8985378\/9003727\/09003816.pdf?arnumber=9003816","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,7,18]],"date-time":"2022-07-18T14:49:25Z","timestamp":1658155765000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9003816\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,12]]},"references-count":43,"URL":"https:\/\/doi.org\/10.1109\/asru46091.2019.9003816","relation":{},"subject":[],"published":{"date-parts":[[2019,12]]}}}