{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T08:21:34Z","timestamp":1783153294735,"version":"3.54.6"},"reference-count":189,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,4,20]],"date-time":"2026-04-20T00:00:00Z","timestamp":1776643200000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by-nc\/4.0\/"}],"funder":[{"DOI":"10.13039\/501100001602","name":"Taighde \u00c9ireann - Research Ireland","doi-asserted-by":"publisher","award":["22\/FFP-A\/11059"],"award-info":[{"award-number":["22\/FFP-A\/11059"]}],"id":[{"id":"10.13039\/501100001602","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Computer Speech &amp; Language"],"published-print":{"date-parts":[[2027,1]]},"DOI":"10.1016\/j.csl.2026.101990","type":"journal-article","created":{"date-parts":[[2026,4,20]],"date-time":"2026-04-20T07:03:39Z","timestamp":1776668619000},"page":"101990","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["A multimodal perspective on adaptive communication: Extending the hyper- and hypo-articulation theory"],"prefix":"10.1016","volume":"101","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0664-219X","authenticated-orcid":false,"given":"Delphine","family":"Charuau","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Naomi","family":"Harte","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"issue":"12","key":"10.1016\/j.csl.2026.101990_b1","doi-asserted-by":"crossref","first-page":"8717","DOI":"10.1109\/TPAMI.2018.2889052","article-title":"Deep audio-visual speech recognition","volume":"44","author":"Afouras","year":"2018","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"issue":"23","key":"10.1016\/j.csl.2026.101990_b2","doi-asserted-by":"crossref","first-page":"5163","DOI":"10.3390\/s19235163","article-title":"Multimodal speaker diarization using a pre-trained audio-visual synchronization model","volume":"19","author":"Ahmad","year":"2019","journal-title":"Sensors"},{"key":"10.1016\/j.csl.2026.101990_b3","doi-asserted-by":"crossref","first-page":"169","DOI":"10.1006\/jmla.2000.2752","article-title":"Efects of visibility between speaker and listener on gesture production: Some gestures are meant to be seen","volume":"44","author":"Alibali","year":"2001","journal-title":"J. Mem. Lang."},{"key":"10.1016\/j.csl.2026.101990_b4","series-title":"Speech Technology: Theory and Applications","first-page":"123","article-title":"Interacting with embodied conversational agents","author":"Andr\u00e9","year":"2010"},{"key":"10.1016\/j.csl.2026.101990_b5","doi-asserted-by":"crossref","unstructured":"Aneja, D., Hoegen, R., McDuff, D., Czerwinski, M., 2021. Understanding conversational and expressive style in a multimodal embodied conversational agent. In: Proceedings of the 2021 CHI Conference on Human Factors in Computing Systems. pp. 1\u201310.","DOI":"10.1145\/3411764.3445708"},{"issue":"2","key":"10.1016\/j.csl.2026.101990_b6","doi-asserted-by":"crossref","first-page":"356","DOI":"10.1109\/TASL.2011.2125954","article-title":"Speaker diarization: A review of recent research","volume":"20","author":"Anguera","year":"2012","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"issue":"6","key":"10.1016\/j.csl.2026.101990_b7","doi-asserted-by":"crossref","first-page":"848","DOI":"10.1017\/S0007125000073980","article-title":"Gaze and mutual gaze","volume":"165","author":"Argyle","year":"1994","journal-title":"Br. J. Psychiatry"},{"issue":"2","key":"10.1016\/j.csl.2026.101990_b8","doi-asserted-by":"crossref","first-page":"495","DOI":"10.1016\/j.jml.2007.02.004","article-title":"Gesturing on the telephone: Independent effects of dialogue and visibility","volume":"58","author":"Bavelas","year":"2008","journal-title":"J. Mem. Lang."},{"issue":"2","key":"10.1016\/j.csl.2026.101990_b9","doi-asserted-by":"crossref","first-page":"145","DOI":"10.1017\/S004740450001037X","article-title":"Language style as audience design","volume":"13","author":"Bell","year":"1984","journal-title":"Lang. Soc."},{"issue":"1","key":"10.1016\/j.csl.2026.101990_b10","doi-asserted-by":"crossref","first-page":"25","DOI":"10.1016\/0010-0277(81)90021-4","article-title":"Phonetic features and acoustic invariance in speech","volume":"10","author":"Blumstein","year":"1981","journal-title":"Cognition"},{"key":"10.1016\/j.csl.2026.101990_b11","series-title":"2015 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"4799","article-title":"Audiovisual speaker diarization of TV series","author":"Bost","year":"2015"},{"issue":"1","key":"10.1016\/j.csl.2026.101990_b12","doi-asserted-by":"crossref","first-page":"272","DOI":"10.1121\/1.1487837","article-title":"The clear speech effect for non-native listeners","volume":"112","author":"Bradlow","year":"2002","journal-title":"J. Acoust. Soc. Am."},{"issue":"1","key":"10.1016\/j.csl.2026.101990_b13","doi-asserted-by":"crossref","first-page":"80","DOI":"10.1044\/1092-4388(2003\/007)","article-title":"Speaking clearly for children with learning disabilities","volume":"46","author":"Bradlow","year":"2003","journal-title":"J. Speech Lang. Hear. Res."},{"issue":"6","key":"10.1016\/j.csl.2026.101990_b14","doi-asserted-by":"crossref","first-page":"1482","DOI":"10.1037\/0278-7393.22.6.1482","article-title":"Conceptual pacts and lexical choice in conversation","volume":"22","author":"Brennan","year":"1996","journal-title":"J. Exp. Psychol. [Learn. Mem. Cogn.]"},{"issue":"2","key":"10.1016\/j.csl.2026.101990_b15","doi-asserted-by":"crossref","first-page":"274","DOI":"10.1111\/j.1756-8765.2009.01019.x","article-title":"Partner-specific adaptation in dialog","volume":"1","author":"Brennan","year":"2009","journal-title":"Top. Cogn. Sci."},{"key":"10.1016\/j.csl.2026.101990_b16","doi-asserted-by":"crossref","first-page":"413","DOI":"10.3389\/fnhum.2013.00413","article-title":"Theory of mind: mechanisms, methods, and new directions","volume":"7","author":"Byom","year":"2013","journal-title":"Front. Hum. Neurosci."},{"key":"10.1016\/j.csl.2026.101990_b17","series-title":"Proceedings of 19th ACM International Conference on Multimodal Interaction","first-page":"350","article-title":"The NoXi database: Multimodal recordings of mediated novice-expert interactions","author":"Cafaro","year":"2017"},{"key":"10.1016\/j.csl.2026.101990_b18","doi-asserted-by":"crossref","first-page":"1422","DOI":"10.1109\/TASLP.2022.3162078","article-title":"Incorporating visual information in audio based self-supervised speaker recognition","volume":"30","author":"Cai","year":"2022","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"issue":"3","key":"10.1016\/j.csl.2026.101990_b19","doi-asserted-by":"crossref","first-page":"157","DOI":"10.3766\/jaaa.16.3.4","article-title":"Clear speech for adults with a hearing loss: Does intervention with communication partners make a difference?","volume":"16","author":"Caissie","year":"2005","journal-title":"J. Am. Acad. Audiol."},{"key":"10.1016\/j.csl.2026.101990_b20","doi-asserted-by":"crossref","DOI":"10.3389\/fpsyg.2019.00560","article-title":"The role of eye gaze during natural social interactions in typical and autistic people","volume":"10","author":"Ca\u00f1igueral","year":"2019","journal-title":"Front. Psychol."},{"issue":"1","key":"10.1016\/j.csl.2026.101990_b21","first-page":"3","article-title":"Announcing the AMI meeting corpus","volume":"11","author":"Carletta","year":"2006","journal-title":"ELRA Newsl."},{"key":"10.1016\/j.csl.2026.101990_b22","doi-asserted-by":"crossref","unstructured":"Cave, C., Guaitella, I., Bertrand, R., Santi, S., Harlay, F., Espesser, R., 1996. About the relationship between eyebrow movements and Fo variations. In: International Conference on Spoken Language Processing. ICSLP, Vol. 4, Philadelphia, PA, USA, pp. 2175\u20132178. http:\/\/dx.doi.org\/10.1109\/ICSLP.1996.607235.","DOI":"10.21437\/ICSLP.1996-551"},{"issue":"7","key":"10.1016\/j.csl.2026.101990_b23","doi-asserted-by":"crossref","DOI":"10.1371\/journal.pcbi.1000436","article-title":"The natural statistics of audiovisual speech","volume":"5","author":"Chandrasekaran","year":"2009","journal-title":"PLoS Comput. Biol."},{"issue":"4","key":"10.1016\/j.csl.2026.101990_b24","doi-asserted-by":"crossref","first-page":"454","DOI":"10.1016\/j.jvoice.2004.01.004","article-title":"Perceived phonatory effort and phonation threshold pressure across a prolonged voice loading task: a study of vocal fatigue","volume":"18","author":"Chang","year":"2004","journal-title":"J. Voice: Off. J. Voice Found."},{"key":"10.1016\/j.csl.2026.101990_b25","doi-asserted-by":"crossref","unstructured":"Chen, Y., Wang, J., Lin, L., Qi, Z., Ma, J., Shan, Y., 2023. Tagging before alignment: Integrating multi-modal tags for video-text retrieval. In: Proceedings of the AAAI Conference on Artificial Intelligence. Vol. 37, pp. 396\u2013404.","DOI":"10.1609\/aaai.v37i1.25113"},{"key":"10.1016\/j.csl.2026.101990_b26","series-title":"Integrating audio, visual, and semantic information for enhanced multimodal speaker diarization","author":"Cheng","year":"2024"},{"issue":"3","key":"10.1016\/j.csl.2026.101990_b27","doi-asserted-by":"crossref","first-page":"263","DOI":"10.1016\/j.bandl.2013.05.016","article-title":"Computational modeling of stuttering caused by impairments in a basal ganglia thalamo-cortical circuit involved in syllable selection and initiation","volume":"126","author":"Civier","year":"2013","journal-title":"Brain Lang."},{"key":"10.1016\/j.csl.2026.101990_b28","series-title":"Advances in Psychology","doi-asserted-by":"crossref","first-page":"287","DOI":"10.1016\/S0166-4115(09)60059-5","article-title":"Audience design in meaning and reference","volume":"Vol. 9","author":"Clark","year":"1982"},{"issue":"1","key":"10.1016\/j.csl.2026.101990_b29","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1016\/0010-0277(86)90010-7","article-title":"Referring as a collaborative process","volume":"22","author":"Clark","year":"1986","journal-title":"Cognition"},{"issue":"10","key":"10.1016\/j.csl.2026.101990_b30","doi-asserted-by":"crossref","first-page":"753","DOI":"10.1016\/j.parkreldis.2011.08.001","article-title":"An investigation of co-speech gesture production during action description in Parkinson\u2019s disease","volume":"17","author":"Cleary","year":"2011","journal-title":"Parkinsonism Rel. Disord."},{"issue":"5","key":"10.1016\/j.csl.2026.101990_b31","doi-asserted-by":"crossref","first-page":"2421","DOI":"10.1121\/1.2229005","article-title":"An audio-visual corpus for speech perception and automatic speech recognition","volume":"120","author":"Cooke","year":"2006","journal-title":"J. Acoust. Soc. Am."},{"issue":"4","key":"10.1016\/j.csl.2026.101990_b32","doi-asserted-by":"crossref","first-page":"2059","DOI":"10.1121\/1.3478775","article-title":"Spectral and temporal changes to speech produced in the presence of energetic and informational maskersa)","volume":"128","author":"Cooke","year":"2010","journal-title":"J. Acoust. Soc. Am."},{"issue":"3","key":"10.1016\/j.csl.2026.101990_b33","doi-asserted-by":"crossref","first-page":"442","DOI":"10.1016\/j.cognition.2011.11.013","article-title":"Recognizing prosody across modalities, face areas and speakers: Examining perceivers\u2019 sensitivity to variable realizations of visual prosody","volume":"122","author":"Cvejic","year":"2012","journal-title":"Cognition"},{"key":"10.1016\/j.csl.2026.101990_b34","series-title":"Interspeech","first-page":"1433","article-title":"Prosody for the eyes: quantifying visual prosody using guided principal component analysis","author":"Cvejic","year":"2010"},{"key":"10.1016\/j.csl.2026.101990_b35","doi-asserted-by":"crossref","DOI":"10.3389\/fpsyg.2021.616471","article-title":"The role of eye gaze in regulating turn taking in conversations: a systematized review of methods and findings","volume":"12","author":"Degutyte","year":"2021","journal-title":"Front. Psychol."},{"issue":"2\u20133","key":"10.1016\/j.csl.2026.101990_b36","doi-asserted-by":"crossref","first-page":"177","DOI":"10.1177\/0023830909103166","article-title":"Interaction of audition and vision for the perception of prosodic contrastive focus","volume":"52","author":"Dohen","year":"2009","journal-title":"Lang. Speech"},{"issue":"1","key":"10.1016\/j.csl.2026.101990_b37","doi-asserted-by":"crossref","first-page":"212","DOI":"10.1044\/2016_JSLHR-H-16-0101","article-title":"Visual context enhanced: The joint contribution of iconic gestures and visible speech to degraded speech comprehension","volume":"60","author":"Drijvers","year":"2017","journal-title":"J. Speech Lang. Hear. Res.: JSLHR"},{"key":"10.1016\/j.csl.2026.101990_b38","doi-asserted-by":"crossref","first-page":"7","DOI":"10.1016\/j.bandl.2018.01.003","article-title":"Native language status of the listener modulates the neural integration of speech and iconic gestures in clear and adverse listening conditions","volume":"177\u2013178","author":"Drijvers","year":"2018","journal-title":"Brain Lang."},{"issue":"2","key":"10.1016\/j.csl.2026.101990_b39","doi-asserted-by":"crossref","first-page":"209","DOI":"10.1177\/0023830919831311","article-title":"Non-native listeners benefit less from gestures and visible speech than native listeners during degraded speech comprehension","volume":"63","author":"Drijvers","year":"2020","journal-title":"Lang. Speech"},{"issue":"10","key":"10.1016\/j.csl.2026.101990_b40","doi-asserted-by":"crossref","DOI":"10.1111\/cogs.12789","article-title":"Degree of language experience modulates visual attention to visible speech and iconic gestures during clear and degraded speech comprehension","volume":"43","author":"Drijvers","year":"2019","journal-title":"Cogn. Sci."},{"key":"10.1016\/j.csl.2026.101990_b41","series-title":"Proceedings of the 25th ACM International Conference on Intelligent Virtual Agents","article-title":"Synthetically expressive: Evaluating gesture and voice for emotion and empathy in VR and 2D scenarios","author":"Du","year":"2025"},{"issue":"8","key":"10.1016\/j.csl.2026.101990_b42","doi-asserted-by":"crossref","first-page":"1004","DOI":"10.1080\/02687038.2013.869307","article-title":"Motor speech disorders associated with primary progressive aphasia","volume":"28","author":"Duffy","year":"2014","journal-title":"Aphasiology"},{"issue":"2","key":"10.1016\/j.csl.2026.101990_b43","doi-asserted-by":"crossref","first-page":"283","DOI":"10.1037\/h0033031","article-title":"Some signals and rules for taking speaking turns in conversations","volume":"23","author":"Duncan","year":"1972","journal-title":"J. Pers. Soc. Psychol."},{"key":"10.1016\/j.csl.2026.101990_b44","doi-asserted-by":"crossref","DOI":"10.1016\/j.neuroimage.2022.119734","article-title":"The CABB dataset: A multimodal corpus of communicative interactions for behavioural and neural analyses","volume":"264","author":"Eijk","year":"2022","journal-title":"NeuroImage"},{"issue":"3","key":"10.1016\/j.csl.2026.101990_b45","doi-asserted-by":"crossref","first-page":"886","DOI":"10.2307\/1129478","article-title":"Deliberate facial movement","volume":"51","author":"Ekman","year":"1980","journal-title":"Child Dev."},{"key":"10.1016\/j.csl.2026.101990_b46","series-title":"Proceedings of the 23rd Annual Meeting of the Special Interest Group on Discourse and Dialogue","first-page":"541","article-title":"How much does prosody help turn-taking? Investigations using voice activity projection models","author":"Ekstedt","year":"2022"},{"issue":"1","key":"10.1016\/j.csl.2026.101990_b47","doi-asserted-by":"crossref","DOI":"10.1121\/10.0024364","article-title":"Optimization-based modeling of lombard speech articulation: Supraglottal characteristics","volume":"4","author":"Elie","year":"2024","journal-title":"JASA Express Lett."},{"key":"10.1016\/j.csl.2026.101990_b48","series-title":"Invariance and Variability in Speech Processes","first-page":"360","article-title":"Exploiting lawful variability in the speech wace","author":"Elman","year":"1986"},{"key":"10.1016\/j.csl.2026.101990_b49","series-title":"Looking to listen at the cocktail party: A speaker-independent audio-visual model for speech separation","author":"Ephrat","year":"2018"},{"issue":"1","key":"10.1016\/j.csl.2026.101990_b50","doi-asserted-by":"crossref","first-page":"159","DOI":"10.1044\/2017_JSLHR-H-17-0082","article-title":"Talker differences in clear and conversational speech: Perceived sentence clarity for Young adults with normal hearing and older adults with hearing loss","volume":"61","author":"Ferguson","year":"2018","journal-title":"J. Speech Lang. Hear. Res. : JSLHR"},{"key":"10.1016\/j.csl.2026.101990_b51","doi-asserted-by":"crossref","DOI":"10.1016\/j.cognition.2024.106049","article-title":"Beat gestures and prosodic prominence interactively influence language comprehension","volume":"256","author":"Ferrari","year":"2025","journal-title":"Cognition"},{"key":"10.1016\/j.csl.2026.101990_b52","doi-asserted-by":"crossref","first-page":"37","DOI":"10.1016\/j.specom.2015.08.001","article-title":"The effect of seeing the interlocutor on auditory and visual speech production in noise","volume":"74","author":"Fitzpatrick","year":"2015","journal-title":"Speech Commun."},{"key":"10.1016\/j.csl.2026.101990_b53","doi-asserted-by":"crossref","first-page":"3","DOI":"10.1016\/S0095-4470(19)30607-2","article-title":"An event approach to the study of speech perception from a direct-realist perspective","volume":"14","author":"Fowler","year":"1986","journal-title":"J. Phon."},{"key":"10.1016\/j.csl.2026.101990_b54","series-title":"European Conference on Computer Vision","first-page":"214","article-title":"Multi-modal transformer for video retrieval","author":"Gabeur","year":"2020"},{"key":"10.1016\/j.csl.2026.101990_b55","series-title":"The multimodal information based speech processing (misp) 2025 challenge: Audio-visual diarization and recognition","author":"Gao","year":"2025"},{"issue":"2","key":"10.1016\/j.csl.2026.101990_b56","doi-asserted-by":"crossref","first-page":"580","DOI":"10.1016\/j.csl.2013.07.005","article-title":"Speaking in noise: How does the Lombard effect improve acoustic contrasts between speech and ambient noise?","volume":"28","author":"Garnier","year":"2014","journal-title":"Comput. Speech Lang."},{"issue":"3","key":"10.1016\/j.csl.2026.101990_b57","doi-asserted-by":"crossref","first-page":"588","DOI":"10.1044\/1092-4388(2009\/08-0138)","article-title":"Influence of sound immersion and communicative interaction on the Lombard effect","volume":"53","author":"Garnier","year":"2010","journal-title":"J. Speech Lang. Hear. Res.: JSLHR"},{"issue":"2","key":"10.1016\/j.csl.2026.101990_b58","doi-asserted-by":"crossref","first-page":"1059","DOI":"10.1121\/1.5051321","article-title":"Hyper-articulation in Lombard speech: An active communicative strategy to enhance visible speech cues?","volume":"144","author":"Garnier","year":"2018","journal-title":"J. Acoust. Soc. Am."},{"issue":"1","key":"10.1016\/j.csl.2026.101990_b59","doi-asserted-by":"crossref","first-page":"157","DOI":"10.1038\/s41598-024-84097-6","article-title":"Co-speech gestures influence the magnitude and stability of articulatory movements: evidence for coupling-based enhancement","volume":"15","author":"Garvin","year":"2025","journal-title":"Sci. Rep."},{"issue":"1","key":"10.1016\/j.csl.2026.101990_b60","doi-asserted-by":"crossref","first-page":"223","DOI":"10.1121\/1.381717","article-title":"Effect of speaking rate on vowel formant movements","volume":"63","author":"Gay","year":"1978","journal-title":"J. Acoust. Soc. Am."},{"key":"10.1016\/j.csl.2026.101990_b61","series-title":"<p>patterns of speech and gesture production in the communications of bilinguals and monolinguals: Do speaker proficiency and discourse context matter?<\/p>","author":"Ghobadi","year":"2025"},{"key":"10.1016\/j.csl.2026.101990_b62","doi-asserted-by":"crossref","first-page":"359","DOI":"10.1016\/j.cognition.2014.11.040","article-title":"The dual function of social gaze","volume":"136","author":"Gobel","year":"2015","journal-title":"Cognition"},{"key":"10.1016\/j.csl.2026.101990_b63","series-title":"INTERSPEECH","first-page":"1674","article-title":"Analysis and perception of speech under physical task stress","author":"Godin","year":"2008"},{"issue":"6","key":"10.1016\/j.csl.2026.101990_b64","doi-asserted-by":"crossref","first-page":"3992","DOI":"10.1121\/1.3647301","article-title":"Analysis of the effects of physical task stress on the speech signal","volume":"130","author":"Godin","year":"2011","journal-title":"J. Acoust. Soc. Am."},{"issue":"4","key":"10.1016\/j.csl.2026.101990_b65","doi-asserted-by":"crossref","first-page":"2156","DOI":"10.1109\/TAFFC.2022.3216993","article-title":"Robust audiovisual emotion recognition: Aligning modalities, capturing temporal information, and handling missing features","volume":"13","author":"Goncalves","year":"2022","journal-title":"IEEE Trans. Affect. Comput."},{"key":"10.1016\/j.csl.2026.101990_b66","series-title":"Why the child\u2019s theory of mind really is a theory","author":"Gopnik","year":"1992"},{"key":"10.1016\/j.csl.2026.101990_b67","series-title":"Proceedings of Fifth IEEE International Conference on Automatic Face Gesture Recognition","first-page":"396","article-title":"Visual prosody: Facial movements accompanying speech","author":"Graf","year":"2002"},{"key":"10.1016\/j.csl.2026.101990_b68","series-title":"ICMI Companion \u201924: Companion Proceedings of the 26th International Conference on Multimodal Interaction","first-page":"147","article-title":"Qualitative study of gesture annotation corpus : Challenges and perspectives","author":"Grondin-Verdon","year":"2024"},{"issue":"1","key":"10.1016\/j.csl.2026.101990_b69","doi-asserted-by":"crossref","first-page":"89","DOI":"10.1038\/s41597-025-04405-1","article-title":"The ECOLANG multimodal corpus of adult-child and adult-adult language","volume":"12","author":"Gu","year":"2025","journal-title":"Sci. Data"},{"issue":"3","key":"10.1016\/j.csl.2026.101990_b70","doi-asserted-by":"crossref","first-page":"280","DOI":"10.1016\/j.bandl.2005.06.001","article-title":"Neural modeling and imaging of the cortical interactions underlying syllable production","volume":"96","author":"Guenther","year":"2006","journal-title":"Brain Lang."},{"issue":"4","key":"10.1016\/j.csl.2026.101990_b71","doi-asserted-by":"crossref","first-page":"611","DOI":"10.1037\/0033-295X.105.4.611-633","article-title":"A theoretical investigation of reference frames for the planning of speech movements","volume":"105","author":"Guenther","year":"1998","journal-title":"Psychol Rev"},{"key":"10.1016\/j.csl.2026.101990_b72","series-title":"Gesture as a Communicative Strategy in Second Language Discourse: A study of Learners of French and Swedish","author":"Gullberg","year":"1998"},{"issue":"1","key":"10.1016\/j.csl.2026.101990_b73","doi-asserted-by":"crossref","first-page":"10451","DOI":"10.1038\/s41598-019-46416-0","article-title":"Speech, movement, and gaze behaviours during dyadic conversation in noise","volume":"9","author":"Hadley","year":"2019","journal-title":"Sci. Rep."},{"issue":"2019","key":"10.1016\/j.csl.2026.101990_b74","doi-asserted-by":"crossref","first-page":"271","DOI":"10.1146\/annurev-psych-010418-103145","article-title":"Nonverbal communication","volume":"70","author":"Hall","year":"2019","journal-title":"Annu. Rev. Psychol."},{"issue":"1","key":"10.1016\/j.csl.2026.101990_b75","doi-asserted-by":"crossref","first-page":"62","DOI":"10.1177\/0033688220966635","article-title":"Multimodal second-language communication: Research findings and pedagogical implications","volume":"52","author":"Hardison","year":"2021","journal-title":"RELC J."},{"issue":"4","key":"10.1016\/j.csl.2026.101990_b76","doi-asserted-by":"crossref","first-page":"1104","DOI":"10.1044\/2020_JSLHR-20-00376","article-title":"Effects of background noise on speech and language in Young adults","volume":"64","author":"Harmon","year":"2021","journal-title":"J. Speech Lang. Hear. Res."},{"issue":"4","key":"10.1016\/j.csl.2026.101990_b77","doi-asserted-by":"crossref","first-page":"2139","DOI":"10.1121\/1.3623753","article-title":"Acoustic-phonetic characteristics of speech produced with communicative intent to counter adverse listening conditions","volume":"130","author":"Hazan","year":"2011","journal-title":"J. Acoust. Soc. Am."},{"key":"10.1016\/j.csl.2026.101990_b78","doi-asserted-by":"crossref","first-page":"15","DOI":"10.1016\/j.csl.2017.09.002","article-title":"Audio-visual word prominence detection from clean and noisy speech","volume":"48","author":"Heckmann","year":"2018","journal-title":"Comput. Speech Lang."},{"issue":"6","key":"10.1016\/j.csl.2026.101990_b79","doi-asserted-by":"crossref","first-page":"3272","DOI":"10.1121\/1.4901705","article-title":"The interaction of lexical and phrasal prosody in whispered speech","volume":"136","author":"Heeren","year":"2014","journal-title":"J. Acoust. Soc. Am."},{"key":"10.1016\/j.csl.2026.101990_b80","doi-asserted-by":"crossref","unstructured":"Heo, S., Murdock, C., Proulx, M., Miller, C., 2025. Gaze-Enhanced Multimodal Turn-Taking Prediction in Triadic Conversations. In: Proceedings of Interspeech 2025.","DOI":"10.21437\/Interspeech.2025-167"},{"issue":"2","key":"10.1016\/j.csl.2026.101990_b81","doi-asserted-by":"crossref","first-page":"339","DOI":"10.1016\/0093-934X(89)90022-9","article-title":"Communicative skills in chronic and severe nonfluent aphasia","volume":"37","author":"Herrmann","year":"1989","journal-title":"Brain Lang."},{"issue":"8","key":"10.1016\/j.csl.2026.101990_b82","doi-asserted-by":"crossref","first-page":"750","DOI":"10.1016\/j.tics.2025.03.006","article-title":"Facial clues to conversational intentions","volume":"29","author":"Holler","year":"2025","journal-title":"Trends Cogn. Sci."},{"issue":"8","key":"10.1016\/j.csl.2026.101990_b83","doi-asserted-by":"crossref","first-page":"639","DOI":"10.1016\/j.tics.2019.05.006","article-title":"Multimodal language processing in human communication","volume":"23","author":"Holler","year":"2019","journal-title":"Trends Cogn. Sci."},{"key":"10.1016\/j.csl.2026.101990_b84","series-title":"2016 IEEE Conference on Computer Vision and Pattern Recognition","first-page":"3574","article-title":"Temporal multimodal learning in audiovisual speech recognition","author":"Hu","year":"2016"},{"issue":"3","key":"10.1016\/j.csl.2026.101990_b85","doi-asserted-by":"crossref","first-page":"323","DOI":"10.1016\/j.resp.2008.08.007","article-title":"Effects of utterance length and vocal loudness on speech breathing in older adults","volume":"164","author":"Huber","year":"2008","journal-title":"Respir. Physiol. Neurobiol."},{"key":"10.1016\/j.csl.2026.101990_b86","doi-asserted-by":"crossref","unstructured":"Hung, H., Jayagopi, D., Yeo, C., Friedland, G., Ba, S., Odobez, J.-M., Ramchandran, K., Mirghafori, N., Gatica-Perez, D., 2007. Using audio and video features to classify the most dominant person in a group meeting. In: Proceedings of the 15th ACM International Conference on Multimedia. pp. 835\u2013838.","DOI":"10.1145\/1291233.1291423"},{"issue":"1","key":"10.1016\/j.csl.2026.101990_b87","doi-asserted-by":"crossref","first-page":"22","DOI":"10.4218\/etrij.2023-0266","article-title":"Multimodal audiovisual speech recognition architecture using a three-feature multi-fusion method for noise-robust systems","volume":"46","author":"Jeon","year":"2024","journal-title":"ETRI J."},{"issue":"1","key":"10.1016\/j.csl.2026.101990_b88","doi-asserted-by":"crossref","first-page":"137","DOI":"10.1121\/10.0016820","article-title":"Non-native talkers and listeners and the perceptual benefits of clear speech","volume":"153","author":"Jung","year":"2023","journal-title":"J. Acoust. Soc. Am."},{"issue":"1","key":"10.1016\/j.csl.2026.101990_b89","doi-asserted-by":"crossref","first-page":"13","DOI":"10.1016\/S0167-6393(96)00041-6","article-title":"The influence of acoustics on speech production: A noise-induced stress phenomenon known as the Lombard reflex","volume":"20","author":"Junqua","year":"1996","journal-title":"Speech Commun."},{"key":"10.1016\/j.csl.2026.101990_b90","doi-asserted-by":"crossref","first-page":"22","DOI":"10.1016\/0001-6918(67)90005-4","article-title":"Some functions of gaze-direction in social interaction","volume":"26","author":"Kendon","year":"1967","journal-title":"Acta Psychol."},{"key":"10.1016\/j.csl.2026.101990_b91","series-title":"Looking in Conversation and the Regulation of Turns at Talk: A Comment on the Papers of G. Beattie and DR Rutter et al.","author":"Kendon","year":"1978"},{"key":"10.1016\/j.csl.2026.101990_b92","series-title":"Gesture: Visible Action as Utterance","author":"Kendon","year":"2004"},{"key":"10.1016\/j.csl.2026.101990_b93","doi-asserted-by":"crossref","first-page":"317","DOI":"10.1016\/j.specom.2013.06.003","article-title":"Tracking eyebrows and head gestures associated with spoken prosody","volume":"57","author":"Kim","year":"2014","journal-title":"Speech Commun."},{"key":"10.1016\/j.csl.2026.101990_b94","series-title":"Audio-Visual Speech Processing 2005","first-page":"17","article-title":"A visual concomitant of the Lombard reflex","author":"Kim","year":"2005"},{"key":"10.1016\/j.csl.2026.101990_b95","doi-asserted-by":"crossref","DOI":"10.3389\/fpsyg.2024.1324667","article-title":"Partner-directed gaze and co-speech hand gestures: effects of age, hearing loss and noise","volume":"15","author":"Kim","year":"2024","journal-title":"Front. Psychol."},{"issue":"2","key":"10.1016\/j.csl.2026.101990_b96","doi-asserted-by":"crossref","first-page":"417","DOI":"10.1044\/1092-4388(2010\/10-0020)","article-title":"An acoustic study of the relationships among neurologic disease, dysarthria type, and severity of dysarthria","volume":"54","author":"Kim","year":"2011","journal-title":"J. Speech Lang. Hear. Res.: JSLHR"},{"key":"10.1016\/j.csl.2026.101990_b97","series-title":"Language and Gesture","first-page":"162","article-title":"How representational gestures help speaking","author":"Kita","year":"2000"},{"key":"10.1016\/j.csl.2026.101990_b98","doi-asserted-by":"crossref","unstructured":"Koutsombogera, M., Vogel, C., 2018. Modelling Collaborative Multimodal Behavior in group dialogues: The MULTISIMO corpus. In: Proceedings of the Eleventh International Conference on Language Resources and Evaluation (LREC 2018). pp. 2945\u20132951.","DOI":"10.63317\/5ffknhrev5q8"},{"issue":"3","key":"10.1016\/j.csl.2026.101990_b99","doi-asserted-by":"crossref","first-page":"396","DOI":"10.1016\/j.jml.2007.06.005","article-title":"The effects of visual beats on prosodic prominence: Acoustic analyses, auditory perception and visual perception","volume":"57","author":"Krahmer","year":"2007","journal-title":"J. Mem. Lang."},{"key":"10.1016\/j.csl.2026.101990_b100","doi-asserted-by":"crossref","first-page":"600","DOI":"10.3758\/s13423-021-02009-5","article-title":"The role of iconic gestures and mouth movements in face-to-face communication","volume":"29","author":"Krason","year":"2022","journal-title":"Psychon. Bull. Rev."},{"issue":"5","key":"10.1016\/j.csl.2026.101990_b101","doi-asserted-by":"crossref","first-page":"837","DOI":"10.1037\/xlm0001399","article-title":"Understanding discourse in face-to-face settings: The impact of multimodal cues and listening conditions","volume":"51","author":"Krason","year":"2025","journal-title":"J. Exp. Psychol. Learn. Mem. Cogn."},{"issue":"5","key":"10.1016\/j.csl.2026.101990_b102","doi-asserted-by":"crossref","first-page":"2165","DOI":"10.1121\/1.1509432","article-title":"Investigating alternative forms of clear speech: the effects of speaking rate and speaking mode on intelligibility","volume":"112","author":"Krause","year":"2002","journal-title":"J. Acoust. Soc. Am."},{"issue":"3","key":"10.1016\/j.csl.2026.101990_b103","first-page":"1","article-title":"A kinematic study of prosodic structure in articulatory and manual gestures: Results from a novel method of data collection","volume":"8","author":"Krivokapi\u0107","year":"2017","journal-title":"Lab. Phonol."},{"key":"10.1016\/j.csl.2026.101990_b104","doi-asserted-by":"crossref","unstructured":"Kucherenko, T., Jonell, P., van Waveren, S., Henter, G.E., Alexandersson, S., Leite, I., Kjellstr\u00f6m, H., 2020. Gesticulator: A framework for semantically-aware speech-driven gesture generation. In: Proceedings of the 2020 International Conference on Multimodal Interaction. pp. 242\u2013250.","DOI":"10.1145\/3382507.3418815"},{"issue":"1","key":"10.1016\/j.csl.2026.101990_b105","doi-asserted-by":"crossref","first-page":"27","DOI":"10.3758\/s13414-017-1428-0","article-title":"Knowing when to respond: The role of visual information in conversational turn exchanges","volume":"80","author":"Latif","year":"2018","journal-title":"Atten. Percept. Psychophys."},{"issue":"10","key":"10.1016\/j.csl.2026.101990_b106","doi-asserted-by":"crossref","first-page":"1457","DOI":"10.1080\/01690965.2010.500218","article-title":"The temporal relation between beat gestures and speech","volume":"26","author":"Leonard","year":"2011","journal-title":"Lang. Cogn. Process."},{"issue":"1651","key":"10.1016\/j.csl.2026.101990_b107","doi-asserted-by":"crossref","DOI":"10.1098\/rstb.2013.0302","article-title":"The origin of human multi-modal communication","volume":"369","author":"Levinson","year":"2014","journal-title":"Phil. Trans. R. Soc. B"},{"issue":"6","key":"10.1016\/j.csl.2026.101990_b108","doi-asserted-by":"crossref","first-page":"431","DOI":"10.1037\/h0020279","article-title":"Perception of the speech code","volume":"74","author":"Liberman","year":"1967","journal-title":"Psychol Rev"},{"key":"10.1016\/j.csl.2026.101990_b109","series-title":"Predicting turn-taking and backchannel in human-machine conversations using linguistic, acoustic, and visual signals","author":"Lin","year":"2025"},{"key":"10.1016\/j.csl.2026.101990_b110","doi-asserted-by":"crossref","unstructured":"Lin, Y., Zheng, Y., Zeng, M., Shi, W., 2025b. Predicting Turn-Taking and Backchannel in Human-Machine Conversations Using Linguistic, Acoustic, and Visual Signals. In: Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics. pp. 15310\u201315322.","DOI":"10.18653\/v1\/2025.acl-long.743"},{"key":"10.1016\/j.csl.2026.101990_b111","doi-asserted-by":"crossref","first-page":"403","DOI":"10.1007\/978-94-009-2037-8_16","article-title":"Explaining phonetic variation: a sketch of the H&H theory","author":"Lindblom","year":"1990","journal-title":"Speech Prod. Speech Model."},{"issue":"37","key":"10.1016\/j.csl.2026.101990_b112","first-page":"101","article-title":"Le signe de l\u2019\u00e9l\u00e9vation de la voix","author":"Lombard","year":"1911","journal-title":"Ann. Mal. L\u2019Oreille Larynx"},{"issue":"5","key":"10.1016\/j.csl.2026.101990_b113","doi-asserted-by":"crossref","first-page":"3261","DOI":"10.1121\/1.2990705","article-title":"Speech production modifications produced by competing talkers, babble, and stationary noise","volume":"124","author":"Lu","year":"2008","journal-title":"J. Acoust. Soc. Am."},{"issue":"3","key":"10.1016\/j.csl.2026.101990_b114","doi-asserted-by":"crossref","first-page":"1495","DOI":"10.1121\/1.3179668","article-title":"Speech production modifications produced in the presence of low-pass and high-pass filtered noise","volume":"126","author":"Lu","year":"2009","journal-title":"J. Acoust. Soc. Am."},{"key":"10.1016\/j.csl.2026.101990_b115","article-title":"The relationship between different types of co-speech gestures and L2 speech performance","volume":"13","author":"Ma","year":"2022","journal-title":"Front. Psychol."},{"key":"10.1016\/j.csl.2026.101990_b116","series-title":"ICASSP 2021-2021 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"7613","article-title":"End-to-end audio-visual speech recognition with conformers","author":"Ma","year":"2021"},{"issue":"5","key":"10.1016\/j.csl.2026.101990_b117","doi-asserted-by":"crossref","first-page":"407","DOI":"10.3758\/BF03204884","article-title":"Influence of preceding liquid on stop-consonant perception","volume":"28","author":"Mann","year":"1980","journal-title":"Percept. Psychophys."},{"issue":"7\u20138","key":"10.1016\/j.csl.2026.101990_b118","doi-asserted-by":"crossref","first-page":"953","DOI":"10.1080\/01690965.2012.705006","article-title":"Speech recognition in adverse conditions: A review","volume":"27","author":"Mattys","year":"2012","journal-title":"Lang. Cogn. Process."},{"issue":"7","key":"10.1016\/j.csl.2026.101990_b119","doi-asserted-by":"crossref","first-page":"855","DOI":"10.1016\/S0378-2166(99)00079-X","article-title":"Linguistic functions of head movements in the context of speech","volume":"32","author":"McClave","year":"2000","journal-title":"J. Pragmat."},{"key":"10.1016\/j.csl.2026.101990_b120","unstructured":"McGurk, H., 1998. Developmental Psychology and the Vision of Speech (McGurk\u2019s Inaugural Lecture in 1988). In: Proc. AVSP 1998. pp. 3\u201320."},{"key":"10.1016\/j.csl.2026.101990_b121","series-title":"Hand and Mind: What Gestures Reveal About Thought","author":"McNeill","year":"1992"},{"key":"10.1016\/j.csl.2026.101990_b122","doi-asserted-by":"crossref","first-page":"1368","DOI":"10.1109\/TASLP.2021.3066303","article-title":"An overview of deep-learning-based audio-visual speech enhancement and separation","volume":"29","author":"Michelsanti","year":"2021","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"issue":"10","key":"10.1016\/j.csl.2026.101990_b123","doi-asserted-by":"crossref","first-page":"1694","DOI":"10.1109\/TMM.2015.2463722","article-title":"Multimodal multi-channel on-line speaker diarization using sensor fusion through SVM","volume":"17","author":"Minotto","year":"2015","journal-title":"IEEE Trans. Multimed."},{"issue":"1","key":"10.1016\/j.csl.2026.101990_b124","doi-asserted-by":"crossref","first-page":"40","DOI":"10.1121\/1.410492","article-title":"Interaction between duration, context, and speaking style in English stressed vowels","volume":"96","author":"Moon","year":"1994","journal-title":"J. Acoust. Soc. Am."},{"key":"10.1016\/j.csl.2026.101990_b125","unstructured":"Morett, L., Gibbs, R., MacWhinney, B., 2012. The Role of Gesture in Second Language Learning: Communication, Acquisition, and Retention. In: Proceedings of the Annual Meeting of the Cognitive Science Society. Vol. 34, ISBN: 1069-7977."},{"key":"10.1016\/j.csl.2026.101990_b126","series-title":"Proc. ACM Multimedia","first-page":"4878","article-title":"MultiMediate: Multi-modal group behaviour analysis for artificial mediation","author":"M\u00fcller","year":"2021"},{"issue":"2","key":"10.1016\/j.csl.2026.101990_b127","doi-asserted-by":"crossref","first-page":"133","DOI":"10.1111\/j.0963-7214.2004.01502010.x","article-title":"Visual prosody and speech intelligibility: Head movement improves auditory speech perception","volume":"15","author":"Munhall","year":"2004","journal-title":"Psychol. Sci."},{"key":"10.1016\/j.csl.2026.101990_b128","series-title":"Interspeech 2025","first-page":"1828","article-title":"Cocktail-party audio-visual speech recognition","author":"Nguyen","year":"2025"},{"issue":"1","key":"10.1016\/j.csl.2026.101990_b129","doi-asserted-by":"crossref","first-page":"79","DOI":"10.1109\/TPAMI.2011.47","article-title":"Multimodal speaker diarization","volume":"34","author":"Noulas","year":"2011","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.csl.2026.101990_b130","series-title":"Computer Graphics Forum","first-page":"569","article-title":"A comprehensive review of data-driven co-speech gesture generation","volume":"Vol. 42","author":"Nyatsanga","year":"2023"},{"issue":"1","key":"10.1016\/j.csl.2026.101990_b131","doi-asserted-by":"crossref","first-page":"26","DOI":"10.1044\/2024_JSLHR-24-00162","article-title":"Reassessing the benefits of audiovisual integration to speech perception and intelligibility","volume":"68","author":"O\u2019Hanlon","year":"2025","journal-title":"J. Speech Lang. Hear. Res."},{"issue":"1","key":"10.1016\/j.csl.2026.101990_b132","doi-asserted-by":"crossref","first-page":"1","DOI":"10.16995\/labphon.10900","article-title":"The encoding of prominence relations in supra-laryngeal articulation across speaking styles","volume":"15","author":"Pagel","year":"2024","journal-title":"Lab. Phonol."},{"key":"10.1016\/j.csl.2026.101990_b133","series-title":"ICPhS 2023","article-title":"A kinematic analysis of visual prosody: Head movements in habitual and loud speech","author":"Pagel","year":"2023"},{"key":"10.1016\/j.csl.2026.101990_b134","series-title":"Proceedings of the 2024 Joint International Conference on Computational Linguistics, Language Resources and Evaluation (LREC-COLING 2024)","first-page":"11890","article-title":"Multimodal behaviour in an online environment: The GEHM zoom corpus collection","author":"Paggio","year":"2024"},{"key":"10.1016\/j.csl.2026.101990_b135","doi-asserted-by":"crossref","unstructured":"Pelachaud, C., 2005. Multimodal expressive embodied conversational agents. In: Proceedings of the 13th Annual ACM International Conference on Multimedia. pp. 683\u2013689.","DOI":"10.1145\/1101149.1101301"},{"key":"10.1016\/j.csl.2026.101990_b136","series-title":"Proc. International Congress of Phonetic Sciences (ICPhS)","first-page":"3146","article-title":"Recovering implicit pitch contours from formants in whispered speech","author":"P\u00e9rez Zarazaga","year":"2023"},{"key":"10.1016\/j.csl.2026.101990_b137","doi-asserted-by":"crossref","DOI":"10.3389\/fpsyg.2024.1289637","article-title":"Investigating conversational dynamics in triads: Effects of noise, hearing impairment, and hearing aids","volume":"15","author":"Petersen","year":"2024","journal-title":"Front. Psychol."},{"issue":"1","key":"10.1016\/j.csl.2026.101990_b138","doi-asserted-by":"crossref","first-page":"96","DOI":"10.1044\/jshr.2801.96","article-title":"Speaking clearly for the hard of hearing I: Intelligibility differences between clear and conversational speech","volume":"28","author":"Picheny","year":"1985","journal-title":"J. Speech Hear. Res."},{"issue":"2","key":"10.1016\/j.csl.2026.101990_b139","doi-asserted-by":"crossref","first-page":"212","DOI":"10.1017\/S0140525X04450055","article-title":"The interactive-alignment model: Developments and refinements","volume":"27","author":"Pickering","year":"2004","journal-title":"Behav. Brain Sci."},{"issue":"4","key":"10.1016\/j.csl.2026.101990_b140","doi-asserted-by":"crossref","first-page":"515","DOI":"10.1017\/S0140525X00076512","article-title":"Does the chimpanzee have a theory of mind?","volume":"1","author":"Premack","year":"1978","journal-title":"Behav. Brain Sci."},{"issue":"3","key":"10.1016\/j.csl.2026.101990_b141","doi-asserted-by":"crossref","first-page":"171","DOI":"10.1145\/568513.568514","article-title":"Multimodal human discourse: gesture and speech","volume":"9","author":"Quek","year":"2002","journal-title":"ACM Trans. Comput.-Hum. Interact."},{"issue":"1","key":"10.1016\/j.csl.2026.101990_b142","doi-asserted-by":"crossref","first-page":"22","DOI":"10.1044\/jshr.2601.22","article-title":"Effects of physiological aging on selected acoustic characteristics of voice","volume":"26","author":"Ramig","year":"1983","journal-title":"J. Speech Lang. Hear. Res."},{"issue":"13","key":"10.1016\/j.csl.2026.101990_b143","doi-asserted-by":"crossref","first-page":"eadf3197","DOI":"10.1126\/sciadv.adf3197","article-title":"The CANDOR corpus: Insights from a large multimodal dataset of naturalistic conversation","volume":"9","author":"Reece","year":"2023","journal-title":"Sci. Adv."},{"key":"10.1016\/j.csl.2026.101990_b144","series-title":"Proceedings of the Thirteenth Language Resources and Evaluation Conference","doi-asserted-by":"crossref","first-page":"2517","DOI":"10.63317\/4z5a7ocx4enp","article-title":"RoomReader: A multimodal corpus of online multiparty conversational interactions","author":"Reverdy","year":"2022"},{"issue":"6","key":"10.1016\/j.csl.2026.101990_b145","doi-asserted-by":"crossref","first-page":"1507","DOI":"10.1044\/1092-4388(2008\/07-0173)","article-title":"The speech focus position effect on jaw-finger coordination in a pointing task","volume":"51","author":"Rochet-Capellan","year":"2008","journal-title":"J. Speech Lang. Hear. Res."},{"key":"10.1016\/j.csl.2026.101990_b146","series-title":"Interspeech 2018","first-page":"586","article-title":"Investigating speech features for continuous turn-taking prediction using LSTMs","author":"Roddy","year":"2018"},{"key":"10.1016\/j.csl.2026.101990_b147","doi-asserted-by":"crossref","unstructured":"Roddy, M., Skantze, G., Harte, N., 2018b. Multimodal continuous turn-taking prediction using multiscale rnns. In: Proceedings of the 20th ACM International Conference on Multimodal Interaction. pp. 186\u2013190.","DOI":"10.1145\/3242969.3242997"},{"key":"10.1016\/j.csl.2026.101990_b148","doi-asserted-by":"crossref","first-page":"411","DOI":"10.1007\/s41701-025-00197-2","article-title":"Multidimensional labeling of gesture in communication: the M3D proposal","volume":"9","author":"Rohrer","year":"2025","journal-title":"Corpus Pragmat."},{"issue":"2","key":"10.1016\/j.csl.2026.101990_b149","doi-asserted-by":"crossref","first-page":"S227","DOI":"10.1044\/1058-0360(2012\/12-0091)","article-title":"Releasing the constraints on aphasia therapy: the positive impact of gesture and multimodality treatments","volume":"22","author":"Rose","year":"2013","journal-title":"Am. J. Speech-Lang. Pathol."},{"issue":"9","key":"10.1016\/j.csl.2026.101990_b150","doi-asserted-by":"crossref","first-page":"1090","DOI":"10.1080\/02687038.2013.805726","article-title":"A systematic review of gesture treatments for post-stroke aphasia","volume":"27","author":"Rose","year":"2013","journal-title":"Aphasiology"},{"issue":"3","key":"10.1016\/j.csl.2026.101990_b151","doi-asserted-by":"crossref","DOI":"10.1016\/j.rmal.2024.100163","article-title":"What automatic speech recognition can and cannot do for conversational speech transcription","volume":"3","author":"Russell","year":"2024","journal-title":"Res. Methods Appl. Linguist."},{"key":"10.1016\/j.csl.2026.101990_b152","series-title":"Findings of the Association for Computational Linguistics: ACL 2025","first-page":"209","article-title":"Visual cues enhance predictive turn-taking for two-party human interaction","author":"Russell","year":"2025"},{"issue":"5","key":"10.1016\/j.csl.2026.101990_b153","doi-asserted-by":"crossref","first-page":"3793","DOI":"10.1121\/1.4824120","article-title":"Clarity in communication: \u201cClear\u201d speech authenticity and lexical neighborhood density effects in speech production and perception","volume":"134","author":"Scarborough","year":"2013","journal-title":"J. Acoust. Soc. Am."},{"key":"10.1016\/j.csl.2026.101990_b154","doi-asserted-by":"crossref","DOI":"10.1016\/j.csl.2023.101534","article-title":"An experimental review of speaker diarization methods with application to two-speaker conversational telephone speech recordings","volume":"82","author":"Serafini","year":"2023","journal-title":"Comput. Speech Lang."},{"key":"10.1016\/j.csl.2026.101990_b155","doi-asserted-by":"crossref","unstructured":"Skantze, G., 2017. Towards a general, continuous model of turn-taking in spoken dialogue using LSTM recurrent neural networks. In: Proceedings of the 18th Annual SIGdial Meeting on Discourse and Dialogue. pp. 220\u2013230.","DOI":"10.18653\/v1\/W17-5527"},{"key":"10.1016\/j.csl.2026.101990_b156","doi-asserted-by":"crossref","DOI":"10.1016\/j.csl.2020.101178","article-title":"Turn-taking in conversational systems and human-robot interaction: a review","volume":"67","author":"Skantze","year":"2021","journal-title":"Comput. Speech Lang."},{"key":"10.1016\/j.csl.2026.101990_b157","series-title":"See What I Mean: Hearing Loss, Gaze and Repair in Conversation","author":"Skelt","year":"2006"},{"key":"10.1016\/j.csl.2026.101990_b158","series-title":"Seminars in Hearing","first-page":"116","article-title":"\u201cAre you looking at me?\u201d The influence of gaze on frequent conversation partners\u2019 management of interaction with adults with acquired hearing impairment","volume":"Vol. 31","author":"Skelt","year":"2010"},{"key":"10.1016\/j.csl.2026.101990_b159","doi-asserted-by":"crossref","DOI":"10.3389\/fpsyg.2025.1584937","article-title":"Adaptions in eye-movement behavior during face-to-face communication in noise","volume":"16","author":"Slomianka","year":"2025","journal-title":"Front. Psychol."},{"issue":"1","key":"10.1016\/j.csl.2026.101990_b160","doi-asserted-by":"crossref","first-page":"236","DOI":"10.1111\/j.1749-818X.2008.00112.x","article-title":"Speaking and hearing clearly: Talker and listener factors in speaking style changes","volume":"3","author":"Smiljani\u0107","year":"2009","journal-title":"Lang. Linguist. Compass"},{"issue":"6","key":"10.1016\/j.csl.2026.101990_b161","doi-asserted-by":"crossref","first-page":"4020","DOI":"10.1121\/1.3652882","article-title":"Bidirectional clear speech perception benefit for native and high-proficiency non-native talkers and listeners: Intelligibility and accentedness","volume":"130","author":"Smiljani\u0107","year":"2011","journal-title":"J. Acoust. Soc. Am."},{"issue":"4","key":"10.1016\/j.csl.2026.101990_b162","article-title":"Multimodal language in child-directed versus adult-directed speech","volume":"77","author":"Song\u00fcl","year":"2024","journal-title":"Q. J. Exp. Psychol. (2006)"},{"key":"10.1016\/j.csl.2026.101990_b163","doi-asserted-by":"crossref","DOI":"10.1016\/j.jcomdis.2020.106030","article-title":"Gesture, communication, and adult acquired hearing loss","volume":"87","author":"Sparrow","year":"2020","journal-title":"J. Commun. Disord."},{"key":"10.1016\/j.csl.2026.101990_b164","doi-asserted-by":"crossref","first-page":"1052","DOI":"10.1109\/TASLP.2020.2980436","article-title":"How to teach DNNs to pay attention to the visual modality in speech recognition","volume":"28","author":"Sterpu","year":"2020","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"issue":"1","key":"10.1016\/j.csl.2026.101990_b165","doi-asserted-by":"crossref","first-page":"3","DOI":"10.1016\/S0095-4470(19)31520-7","article-title":"On the quantal nature of speech","volume":"17","author":"Stevens","year":"1989","journal-title":"J. Phon."},{"issue":"5","key":"10.1016\/j.csl.2026.101990_b166","doi-asserted-by":"crossref","first-page":"1358","DOI":"10.1121\/1.382102","article-title":"Invariant cues for place of articulation in stop consonants","volume":"64","author":"Stevens","year":"1978","journal-title":"J. Acoust. Soc. Am."},{"key":"10.1016\/j.csl.2026.101990_b167","doi-asserted-by":"crossref","first-page":"212","DOI":"10.1121\/1.1907309","article-title":"Visual contribution to speech intelligibility in noise","volume":"26","author":"Sumby","year":"1954","journal-title":"J. Acoust. Soc. Am."},{"key":"10.1016\/j.csl.2026.101990_b168","first-page":"3","article-title":"Some preliminaries to a compre hensive account of audio-visual speech perception","author":"Summerfield","year":"1987","journal-title":"Hear. Eye: Psychol. Lip-Read."},{"key":"10.1016\/j.csl.2026.101990_b169","doi-asserted-by":"crossref","DOI":"10.1016\/j.parkreldis.2023.105487","article-title":"Compensatory articulatory mechanisms preserve intelligibility in prodromal Parkinson\u2019s disease","volume":"112","author":"Thies","year":"2023","journal-title":"Parkinsonism Rel. Disord."},{"issue":"11","key":"10.1016\/j.csl.2026.101990_b170","doi-asserted-by":"crossref","first-page":"2163","DOI":"10.1037\/xlm0000939","article-title":"The scope of audience design in child-directed speech: Parents\u2019 tailoring of word lengths for adult versus child listeners.","volume":"46","author":"Tippenhauer","year":"2020","journal-title":"J. Exp. Psychol. [Learn. Mem. Cogn.]"},{"issue":"5","key":"10.1016\/j.csl.2026.101990_b171","doi-asserted-by":"crossref","first-page":"1557","DOI":"10.1109\/TASL.2006.878256","article-title":"An overview of automatic speaker diarization systems","volume":"14","author":"Tranter","year":"2006","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"issue":"4","key":"10.1016\/j.csl.2026.101990_b172","doi-asserted-by":"crossref","DOI":"10.1098\/rsos.211489","article-title":"A multi-scale investigation of the human communication system\u2019s response to visual disruption","volume":"9","author":"Trujillo","year":"2022","journal-title":"R. Soc. Open Sci."},{"issue":"1","key":"10.1016\/j.csl.2026.101990_b173","doi-asserted-by":"crossref","first-page":"16721","DOI":"10.1038\/s41598-021-95791-0","article-title":"Speakers exhibit a multimodal Lombard effect in noise","volume":"11","author":"Trujillo","year":"2021","journal-title":"Sci. Rep."},{"issue":"5","key":"10.1016\/j.csl.2026.101990_b174","doi-asserted-by":"crossref","DOI":"10.1002\/wcs.1557","article-title":"Speech aging: Production and perception","volume":"12","author":"Tucker","year":"2021","journal-title":"WIREs Cogn. Sci."},{"issue":"1841","key":"10.1016\/j.csl.2026.101990_b175","doi-asserted-by":"crossref","DOI":"10.1098\/rstb.2020.0398","article-title":"Speech modifications in interactive speech: effects of age, sex and noise type","volume":"377","author":"Tuomainen","year":"2022","journal-title":"Phil. Trans. R. Soc. B"},{"key":"10.1016\/j.csl.2026.101990_b176","doi-asserted-by":"crossref","first-page":"207","DOI":"10.1002\/9780470757024.ch9","article-title":"Clear speech","author":"Uchanski","year":"2005","journal-title":"Handb. Speech Percept."},{"issue":"3","key":"10.1016\/j.csl.2026.101990_b177","doi-asserted-by":"crossref","first-page":"509","DOI":"10.1109\/TMM.2012.2233724","article-title":"A multimodal approach to speaker diarization on TV talk-shows","volume":"15","author":"Vallet","year":"2012","journal-title":"IEEE Trans. Multimed."},{"issue":"3","key":"10.1016\/j.csl.2026.101990_b178","doi-asserted-by":"crossref","first-page":"917","DOI":"10.1121\/1.396660","article-title":"Effects of noise on speech production: Acoustic and perceptual analyses","volume":"84","author":"Van Summers","year":"1988","journal-title":"J. Acoust. Soc. Am."},{"key":"10.1016\/j.csl.2026.101990_b179","doi-asserted-by":"crossref","unstructured":"Vercherand, G., 2011. Perceptual level of intonation in whispered voice. In: Proc. ISCA Workshop on Experimental Linguistics. Paris, France.","DOI":"10.36505\/ExLing-2011\/04\/0037\/000206"},{"issue":"1651","key":"10.1016\/j.csl.2026.101990_b180","doi-asserted-by":"crossref","DOI":"10.1098\/rstb.2013.0292","article-title":"Language as a multimodal phenomenon: implications for language learning, processing and evolution","volume":"369","author":"Vigliocco","year":"2014","journal-title":"Phil. Trans. R. Soc. B"},{"issue":"4","key":"10.1016\/j.csl.2026.101990_b181","doi-asserted-by":"crossref","first-page":"220","DOI":"10.1159\/000078344","article-title":"Occupational safety and health aspects of voice and speech professions","volume":"56","author":"Vilkman","year":"2004","journal-title":"Folia Phoniatr. Logop.: Off. Organ Int. Assoc. Logop. Phoniatr. (IALP)"},{"issue":"Supplement C","key":"10.1016\/j.csl.2026.101990_b182","doi-asserted-by":"crossref","first-page":"209","DOI":"10.1016\/j.specom.2013.09.008","article-title":"Gesture and speech in interaction: An overview","volume":"57","author":"Wagner","year":"2014","journal-title":"Speech Commun."},{"issue":"1","key":"10.1016\/j.csl.2026.101990_b183","doi-asserted-by":"crossref","first-page":"6","DOI":"10.1186\/s40469-015-0006-9","article-title":"Evaluating embodied conversational agents in multimodal interfaces","volume":"1","author":"Weiss","year":"2015","journal-title":"Comput. Cogn. Sci."},{"issue":"4","key":"10.1016\/j.csl.2026.101990_b184","doi-asserted-by":"crossref","first-page":"2896","DOI":"10.1121\/10.0004774","article-title":"Conversational distance adaptation in noise and its effect on signal-to-noise ratio in realistic listening environments","volume":"149","author":"Weisser","year":"2021","journal-title":"J. Acoust. Soc. Am."},{"key":"10.1016\/j.csl.2026.101990_b185","doi-asserted-by":"crossref","first-page":"708","DOI":"10.3389\/fpsyg.2017.00708","article-title":"Respiratory constraints in verbal and non-verbal communication","volume":"8","author":"W\u0142odarczak","year":"2017","journal-title":"Front. Psychol."},{"issue":"1","key":"10.1016\/j.csl.2026.101990_b186","doi-asserted-by":"crossref","first-page":"23","DOI":"10.1016\/S0167-6393(98)00048-X","article-title":"Quantitative association of vocal-tract and facial behavior","volume":"26","author":"Yehia","year":"1998","journal-title":"Speech Commun."},{"key":"10.1016\/j.csl.2026.101990_b187","doi-asserted-by":"crossref","DOI":"10.3389\/fcomp.2024.1384252","article-title":"Linguistic analysis of human-computer interaction","volume":"6","author":"Zellou","year":"2024","journal-title":"Front. Comput. Sci."},{"issue":"1","key":"10.1016\/j.csl.2026.101990_b188","article-title":"The role of multimodal cues in second language comprehension","volume":"13","author":"Zhang","year":"2023","journal-title":"Sci. Rep."},{"issue":"1","key":"10.1016\/j.csl.2026.101990_b189","doi-asserted-by":"crossref","first-page":"613","DOI":"10.1121\/10.0015251","article-title":"Communicative constraints affect oro-facial gestures and acoustics: Whispered vs normal speech","volume":"153","author":"\u017bygis","year":"2023","journal-title":"J. Acoust. Soc. Am."}],"container-title":["Computer Speech &amp; Language"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0885230826000537?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0885230826000537?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T08:13:36Z","timestamp":1783152816000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0885230826000537"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2027,1]]},"references-count":189,"alternative-id":["S0885230826000537"],"URL":"https:\/\/doi.org\/10.1016\/j.csl.2026.101990","relation":{},"ISSN":["0885-2308"],"issn-type":[{"value":"0885-2308","type":"print"}],"subject":[],"published":{"date-parts":[[2027,1]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"A multimodal perspective on adaptive communication: Extending the hyper- and hypo-articulation theory","name":"articletitle","label":"Article Title"},{"value":"Computer Speech & Language","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.csl.2026.101990","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 The Authors. Published by Elsevier Ltd.","name":"copyright","label":"Copyright"}],"article-number":"101990"}}