{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,14]],"date-time":"2026-03-14T18:03:47Z","timestamp":1773511427461,"version":"3.50.1"},"reference-count":28,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"funder":[{"DOI":"10.13039\/501100014188","name":"Institute of Information and Communications Technology Planning and Evaluation (IITP) Grant funded by the Korean Government","doi-asserted-by":"publisher","award":["MSIT; 2019-0-00004"],"award-info":[{"award-number":["MSIT; 2019-0-00004"]}],"id":[{"id":"10.13039\/501100014188","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100014188","name":"\u201cDevelopment of Semi-Supervised Learning Language Intelligence Technology and Korean Tutoring Service for Foreigners,\u201d","doi-asserted-by":"publisher","award":["2022-0-00989"],"award-info":[{"award-number":["2022-0-00989"]}],"id":[{"id":"10.13039\/501100014188","id-type":"DOI","asserted-by":"publisher"}]},{"name":"\u201cDevelopment of Artificial Intelligence Technology for Multi-speaker Dialog Modeling\u201d"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Access"],"published-print":{"date-parts":[[2023]]},"DOI":"10.1109\/access.2023.3254529","type":"journal-article","created":{"date-parts":[[2023,3,9]],"date-time":"2023-03-09T19:41:04Z","timestamp":1678390864000},"page":"27426-27432","source":"Crossref","is-referenced-by-count":3,"title":["Audio-Visual Overlapped Speech Detection for Spontaneous Distant Speech"],"prefix":"10.1109","volume":"11","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-2814-9386","authenticated-orcid":false,"given":"Minyoung","family":"Kyoung","sequence":"first","affiliation":[{"name":"Electronics and Telecommunications Research Institute (ETRI), Daejeon, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hyungbae","family":"Jeon","sequence":"additional","affiliation":[{"name":"Electronics and Telecommunications Research Institute (ETRI), Daejeon, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kiyoung","family":"Park","sequence":"additional","affiliation":[{"name":"Electronics and Telecommunications Research Institute (ETRI), Daejeon, Republic of Korea"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2013-27"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-188"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-26061-3_26"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053096"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2018.2889052"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU46091.2019.9004036"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414567"},{"key":"ref8","article-title":"Audio-visual speech recognition is worth 32\u00d732\u00d78 voxels","author":"Serdyuk","year":"2021","journal-title":"arXiv:2109.09536"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053974"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/CIDM.2013.6597233"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2019.07.003"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2019.2901195"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00123"},{"key":"ref14","article-title":"Look&Listen: Multi-modal correlation learning for active speaker detection and speech enhancement","author":"Xiong","year":"2022","journal-title":"arXiv:2203.02216"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2016.7846321"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2671"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2021.101306"},{"key":"ref18","first-page":"137","article-title":"The AMI meeting corpus","volume-title":"Proc. 5th Int. Conf. Methods Techn. Behav. Res.","author":"McCowan"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.367"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2020.3040906"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/WACV48630.2021.00290"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00675"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462548"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2018.2877892"},{"key":"ref26","article-title":"MediaPipe: A framework for building perception pipelines","author":"Lugaresi","year":"2019","journal-title":"arXiv:1906.08172"},{"key":"ref27","first-page":"1","article-title":"On the variance of the adaptive learning rate and beyond","volume-title":"Proc. ICLR","author":"Liu"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1121\/1.4799597"}],"container-title":["IEEE Access"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6287639\/10005208\/10064301.pdf?arnumber=10064301","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,2,13]],"date-time":"2024-02-13T12:47:45Z","timestamp":1707828465000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10064301\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023]]},"references-count":28,"URL":"https:\/\/doi.org\/10.1109\/access.2023.3254529","relation":{},"ISSN":["2169-3536"],"issn-type":[{"value":"2169-3536","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023]]}}}