{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,27]],"date-time":"2025-10-27T16:21:30Z","timestamp":1761582090384,"version":"3.37.3"},"reference-count":23,"publisher":"IEEE","license":[{"start":{"date-parts":[[2021,1,10]],"date-time":"2021-01-10T00:00:00Z","timestamp":1610236800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2021,1,10]],"date-time":"2021-01-10T00:00:00Z","timestamp":1610236800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2021,1,10]],"date-time":"2021-01-10T00:00:00Z","timestamp":1610236800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61673030,U1613209"],"award-info":[{"award-number":["61673030,U1613209"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100009019","name":"National Natural Science Foundation of Shenzhen","doi-asserted-by":"publisher","award":["JCYJ20190808182209321"],"award-info":[{"award-number":["JCYJ20190808182209321"]}],"id":[{"id":"10.13039\/501100009019","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021,1,10]]},"DOI":"10.1109\/icpr48806.2021.9412349","type":"proceedings-article","created":{"date-parts":[[2021,5,6]],"date-time":"2021-05-06T02:15:54Z","timestamp":1620267354000},"page":"5348-5353","source":"Crossref","is-referenced-by-count":4,"title":["Mutual Alignment between Audiovisual Features for End-to-End Audiovisual Speech Recognition"],"prefix":"10.1109","author":[{"given":"Hong","family":"Liu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yawei","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bing","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1037\/a0019952"},{"key":"ref11","first-page":"2013","article-title":"A coupled HMM for audio-visual speech recognition","author":"nefian","year":"0","journal-title":"IEEE International Conference on Acoustics Speech and Signal Processing(ICASSP)"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1145\/3242969.3243014"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ICME.2018.8486455"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683733"},{"key":"ref15","first-page":"6847","article-title":"Aligning visual regions and textual concepts for semantic-grounded image representations","author":"liu","year":"0","journal-title":"Neural Information Processing Systems Conference"},{"key":"ref16","first-page":"5998","article-title":"Attention is all you need","author":"vaswani","year":"0","journal-title":"Neural Information Processing Systems Conference"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461326"},{"key":"ref18","first-page":"1929","article-title":"Dropout: A simple way to prevent neural networks from overfitting","volume":"15","author":"srivastava","year":"2014","journal-title":"Journal of Machine Learning Research"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1038\/264746a0"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462105"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1017\/S095267570000230X"},{"key":"ref5","article-title":"Audio-visual asynchrony modeling and analysis for speech alignment and recognition","author":"terry","year":"2011","journal-title":"Northwestern University"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/TSA.2005.857572"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1994.389567"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6638947"},{"key":"ref1","doi-asserted-by":"crossref","first-page":"30","DOI":"10.1109\/TASL.2011.2134090","article-title":"Context-dependent pre-trained deep neural networks for large-vocabulary speech recognition","volume":"20","author":"dahl","year":"2012","journal-title":"IEEE Trans Audio Speech & Language Processing"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1016\/j.neuropsychologia.2006.01.001"},{"journal-title":"Layer normalization","year":"2016","author":"lei ba","key":"ref20"},{"key":"ref22","article-title":"Adam: A method for stochastic optimization","author":"kingma","year":"0","journal-title":"International Conference on Learning Representations(ICLR)"},{"key":"ref21","first-page":"87","article-title":"Lip reading in the wild","volume":"10112","author":"chung","year":"0","journal-title":"Asian Conf Computer Vision"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1016\/0167-6393(93)90095-3"}],"event":{"name":"2020 25th International Conference on Pattern Recognition (ICPR)","start":{"date-parts":[[2021,1,10]]},"location":"Milan, Italy","end":{"date-parts":[[2021,1,15]]}},"container-title":["2020 25th International Conference on Pattern Recognition (ICPR)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9411940\/9411911\/09412349.pdf?arnumber=9412349","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,5,10]],"date-time":"2022-05-10T15:40:51Z","timestamp":1652197251000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9412349\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,1,10]]},"references-count":23,"URL":"https:\/\/doi.org\/10.1109\/icpr48806.2021.9412349","relation":{},"subject":[],"published":{"date-parts":[[2021,1,10]]}}}