<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Neurosci.</journal-id>
<journal-title>Frontiers in Neuroscience</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Neurosci.</abbrev-journal-title>
<issn pub-type="epub">1662-453X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fnins.2024.1379988</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Neuroscience</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Synthetic faces generated with the facial action coding system or deep neural networks improve speech-in-noise perception, but not as much as real faces</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Yu</surname> <given-names>Yingjia</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Lado</surname> <given-names>Anastasia</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Zhang</surname> <given-names>Yue</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name><surname>Magnotti</surname> <given-names>John F.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn0005"><sup>&#x2020;</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes" equal-contrib="yes">
<name><surname>Beauchamp</surname> <given-names>Michael S.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<xref ref-type="author-notes" rid="fn0005"><sup>&#x2020;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/1021/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Department of Neurosurgery, Perelman School of Medicine, University of Pennsylvania</institution>, <addr-line>Philadelphia, PA</addr-line>, <country>United States</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of Neurosurgery, Baylor College of Medicine</institution>, <addr-line>Houston, TX</addr-line>, <country>United States</country></aff>
<author-notes>
<fn fn-type="edited-by" id="fn0006">
<p>Edited by: Argiro Vatakis, Panteion University, Greece</p>
</fn>
<fn fn-type="edited-by" id="fn0007">
<p>Reviewed by: Patrick Bruns, University of Hamburg, Germany</p>
<p>Benjamin A. Rowland, Wake Forest University, United States</p>
<p>Pierre M&#x00E9;gevand, University of Geneva, Switzerland</p>
</fn>
<corresp id="c001">&#x002A;Correspondence: Michael S. Beauchamp, <email>beaucha@upenn.edu</email></corresp>
<fn fn-type="equal" id="fn0005">
<p><sup>&#x2020;</sup>These authors have contributed equally to this work</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>09</day>
<month>05</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>18</volume>
<elocation-id>1379988</elocation-id>
<history>
<date date-type="received">
<day>31</day>
<month>01</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>11</day>
<month>04</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2024 Yu, Lado, Zhang, Magnotti and Beauchamp.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Yu, Lado, Zhang, Magnotti and Beauchamp</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>The prevalence of synthetic talking faces in both commercial and academic environments is increasing as the technology to generate them grows more powerful and available. While it has long been known that seeing the face of the talker improves human perception of speech-in-noise, recent studies have shown that synthetic talking faces generated by deep neural networks (DNNs) are also able to improve human perception of speech-in-noise. However, in previous studies the benefit provided by DNN synthetic faces was only about half that of real human talkers. We sought to determine whether synthetic talking faces generated by an alternative method would provide a greater perceptual benefit. The facial action coding system (FACS) is a comprehensive system for measuring visually discernible facial movements. Because the action units that comprise FACS are linked to specific muscle groups, synthetic talking faces generated by FACS might have greater verisimilitude than DNN synthetic faces which do not reference an explicit model of the facial musculature. We tested the ability of human observers to identity speech-in-noise accompanied by a blank screen; the real face of the talker; and synthetic talking faces generated either by DNN or FACS. We replicated previous findings of a large benefit for seeing the face of a real talker for speech-in-noise perception and a smaller benefit for DNN synthetic faces. FACS faces also improved perception, but only to the same degree as DNN faces. Analysis at the phoneme level showed that the performance of DNN and FACS faces was particularly poor for phonemes that involve interactions between the teeth and lips, such as /f/, /v/, and /th/. Inspection of single video frames revealed that the characteristic visual features for these phonemes were weak or absent in synthetic faces. Modeling the real vs. synthetic difference showed that increasing the realism of a few phonemes could substantially increase the overall perceptual benefit of synthetic faces.</p>
</abstract>
<kwd-group>
<kwd>audiovisual</kwd>
<kwd>multisensory</kwd>
<kwd>speech</kwd>
<kwd>face</kwd>
<kwd>speech-in-noise (SIN)</kwd>
<kwd>deep neural</kwd>
</kwd-group>
<counts>
<fig-count count="4"/>
<table-count count="0"/>
<equation-count count="1"/>
<ref-count count="38"/>
<page-count count="11"/>
<word-count count="7485"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Auditory Cognitive Neuroscience</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec1">
<title>Introduction</title>
<p>Recent advances in computer graphics have made it much easier to create realistic, synthetic talking faces, spurring adoption in commercial and academic communities. For companies, agents created by pairing a synthetic talking face with the output of large language models provide an always-available simulacrum of a real human representative (<xref ref-type="bibr" rid="ref22">Perry et al., 2023</xref>). In academia, the use of synthetic talking faces in studies of speech perception provides more precise control over the visual features of experimental stimuli than is possible with videos of real human talkers (<xref ref-type="bibr" rid="ref31">Th&#x00E9;z&#x00E9; et al., 2020a</xref>).</p>
<p>Of particular interest is the long-standing observation that humans understand speech-in-noise much better when it is paired with a video of the talker&#x2019;s face (<xref ref-type="bibr" rid="ref30">Sumby and Pollack, 1954</xref>). The ability to rapidly generate a synthetic face saying arbitrary words suggests the possibility of an &#x201C;audiovisual hearing aid&#x201D; that displays a synthetic talking face to improve comprehension. This possibility received support from two recent studies that used deep neural networks (DNNs) to generate realistic, synthetic talking faces (<xref ref-type="bibr" rid="ref28">Shan et al., 2022</xref>; <xref ref-type="bibr" rid="ref36">Varano et al., 2022</xref>). Both studies found that viewing synthetic faces significantly improved speech-in-noise perception, but the benefit was only about half as much as viewing a real human talker.</p>
<p>The substantial disadvantage of synthetic faces raises the question of whether alternative techniques for generating synthetic faces might provide a greater perceptual benefit. DNNs associate given speech sounds with visual features in their training dataset, but do not contain any explicit models of the facial musculature. In contrast, the facial action coding system (FACS) uses 46 basic action units to represent all possible movements of the facial musculature that are visually discernable (<xref ref-type="bibr" rid="ref10">Ekman and Friesen, 1976</xref>, <xref ref-type="bibr" rid="ref11">1978</xref>; <xref ref-type="bibr" rid="ref20">Parke and Waters, 2008</xref>). Unlike DNNs, the FACS scheme is built on an understanding of the physical relationship between speech and facial anatomy, potentially resulting in more accurate representations of speech movements. To test this idea, we undertook a behavioral study to compare the perception of speech-in-noise on its own; speech-in-noise with real faces (to serve as a benchmark); and speech-in-noise presented with two types of synthetic faces. The first synthetic face type was generated by a deep neural network, as in the studies of (<xref ref-type="bibr" rid="ref28">Shan et al., 2022</xref>; <xref ref-type="bibr" rid="ref36">Varano et al., 2022</xref>). The second synthetic face type was generated using FACS, as implemented in the commercial software package JALI (<xref ref-type="bibr" rid="ref9">Edwards et al., 2016</xref>; <xref ref-type="bibr" rid="ref38">Zhou et al., 2018</xref>). For comparison with previous studies, we performed a word-level analysis in which each response was scored as correct or incorrect. To facilitate more fine-grained comparisons between the different face types, we also analyzed data using the phonemic content of each stimulus word.</p>
</sec>
<sec sec-type="methods" id="sec2">
<title>Methods</title>
<sec id="sec3">
<title>Participant recruitment and testing</title>
<p>All experiments were approved by the Institutional Review Board of the University of Pennsylvania, Philadelphia, PA. Participants were recruited and tested using Amazon Mechanical Turk,<xref ref-type="fn" rid="fn0001"><sup>1</sup></xref> an online platform that provides access to an on-demand workforce. Only &#x201C;master workers&#x201D; were recruited, classified as such by Amazon based on their high performance and location in the United States. Sixty-two participants completed the main experiment (median time to complete: 12&#x2009;min) and received $5 reimbursement. The participants answered the questions &#x201C;Do you have a hearing impairment that would make it difficult to understand words embedded in background noise?&#x201D; and &#x201C;Do you have an uncorrected vision impairment that would make it difficult to watch a video of a person talking?&#x201D; One participant was excluded because of a reported hearing impairment, leaving 61 participants whose data is reported here. There were 25 females and 36 males, mean age 46&#x2009;years, range 30&#x2013;72.</p>
</sec>
<sec id="sec4">
<title>Overview</title>
<p>Workers were asked to enroll in the experiment only if they were using a desktop or laptop PC or a tablet (but not a phone) and information about the user&#x2019;s system was collected to verify compliance. At the beginning of the experiment, participants viewed an instructional video (recorded by author MSB) that explained the task and presented examples of the different experimental stimuli. The instructional video was accompanied by text instructions stating &#x201C;Please adjust your window size and audio volume so that you can see and hear everything clearly.&#x201D;</p>
<p>Following completion of the instructional video, participants identified 73 words presented in five different formats (<xref ref-type="fig" rid="fig1">Figure 1</xref>). Sixty-four of the words contained added auditory noise to make identifying them more difficult and increase the importance of visual speech. There were four formats of noisy words: auditory-only (An); with a talking face (audiovisual; AnV) that was either the real face of the talker (AnV:<italic>Real</italic>); a synthetic face created using the facial action coding system (AnV:<italic>FACS</italic>); or a synthetic face created using a deep neural network (AnV:<italic>DNN</italic>). To prevent perceptual learning, each word was only presented once to each participant, 16 words in each of the four formats. Within participants, the order of words and face formats was randomized, and across participants, the format of each word was cycled to ensure that every word was presented in every format. To assess participant compliance, the remaining nine words presented were clear audiovisual words (AV:<italic>catch_trials</italic>). The catch trials sampled all face types (3 Real, 3 FACS, and 3 DNN) and the talkers and words differed from those presented in the noisy trials to prevent learning. Accuracy for catch trials was very high (mean of 98%) demonstrating attention and task engagement. All data was analyzed in R, primarily using mixed effects models. See <xref ref-type="supplementary-material" rid="SM1">Supplementary Material</xref> for all data and an R markdown document that contains all analysis code and results.</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption>
<p><bold>(A)</bold> The main stimulus set consisted of 64 single words with auditory noise added, 32 recorded by a male talker and 32 recorded by a female talker. The noisy words were presented in four formats. The first format consisted of the noisy auditory recordings paired with a video of the actual talker (AnV:<italic>Real</italic>). <bold>(B)</bold> The second format consisted of the recordings paired with gender-matched synthetic face movies generated by a facial action coding system model (AnV:<italic>FACS</italic>). <bold>(C)</bold> The third format consisted of the recordings paired with synthetic face movies generated by a deep neural network (AnV:<italic>DNN</italic>). The gender of the synthetic face matched the gender of the voice. <bold>(D)</bold> The fourth format consisted of the recordings presented with a blank screen (An). Note that for <bold>(A)</bold>&#x2013;<bold>(D)</bold>, the auditory component of the stimulus was identical, only the visual component differed. <bold>(E)</bold> In catch trials, an additional stimulus set was presented consisting of recordings of 9 audiovisual words without added noise (AV) paired with gender-matched real, FACS, or DNN faces (three words each). The words, faces and voices were different than in the main stimulus set to prevent any interference. <bold>(F)</bold> Each participant was presented with 73 words (64 noisy and 9 clear) in random order. Within participants, each noisy word from the main stimulus set was presented only once, in one of the four formats. Counterbalancing was used to present every noisy word in each of the four formats shown in <bold>(A)</bold>&#x2013;<bold>(D)</bold>. For instance, for participant 1, the word <italic>spout</italic> was presented in AnV:<italic>DNN</italic> format, while for participant 2, <italic>spout</italic> was presented in AnV:<italic>Real</italic> format, etc. Following the presentation of each word, participants typed the word in a text box.</p>
</caption>
<graphic xlink:href="fnins-18-1379988-g001.tif"/>
</fig>
</sec>
<sec id="sec5">
<title>Subject responses and scoring: word-level</title>
<p>Following presentation of a word, participants were instructed to &#x201C;type the word&#x201D; into a text box; the next trial did not begin until a response was entered. If a participant&#x2019;s response matched the stimulus word, the trial was scored as &#x201C;correct,&#x201D; otherwise the trial was scored as &#x201C;incorrect.&#x201D; For example, the stimulus word <italic>wormhole</italic> and the response <italic>wormhole</italic> was correct, while the stimulus word <italic>booth</italic> and the response <italic>boot</italic> was incorrect. Misspellings were not considered incorrect (e.g., stimulus <italic>echos</italic> and response <italic>echoes</italic> was correct) nor were homophones (e.g., stimulus <italic>wore</italic> and response <italic>war</italic> was correct). The analysis was performed separately for each condition (An, AnV:<italic>Real</italic>, AnV:<italic>DNN</italic>, AnV:<italic>FACS</italic>, AV:<italic>catch_trials</italic>). Mean accuracy for each condition was calculated per participant, and then averaged across participants. See <xref ref-type="supplementary-material" rid="SM1">Supplementary Material</xref> for a complete list of the words, response and scores.</p>
</sec>
<sec id="sec6">
<title>Phonemic analysis</title>
<p>In addition to the binary word-level accuracy measure, a continuous accuracy measure was calculated for each trial based on the overlap in the phonemes in the stimulus word and the response. Phoneme composition was determined using the Carnegie Mellon University (CMU) pronouncing dictionary.<xref ref-type="fn" rid="fn0002"><sup>2</sup></xref> Lexical stress markers were removed from vowels. For responses that contained multiple words, the phonemes for all response words were extracted from the CMU dictionary and entered into the calculation. Repetitions of the same phoneme were also entered into the calculation. The accuracy measure was the Jaccard index: the number of phonemes in common between the stimulus and response divided by the total number of phonemes in the stimulus and response. The measure ranged from 0 (no phonemes in common between stimulus and response) to 1 (identical phonemes in stimulus and response) and was calculated as</p>
<disp-formula id="E1">
<mml:math id="M1">
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mi>J</mml:mi>
<mml:mfenced open="(" close=")" separators=",">
<mml:mi mathvariant="italic">stimulus</mml:mi>
<mml:mi mathvariant="italic">response</mml:mi>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mfenced open="|" close="|">
<mml:mrow>
<mml:mi mathvariant="italic">stimulus</mml:mi>
<mml:mo>&#x2229;</mml:mo>
<mml:mi mathvariant="italic">response</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mfenced open="|" close="|">
<mml:mrow>
<mml:mi mathvariant="italic">stimulus</mml:mi>
<mml:mo>&#x222A;</mml:mo>
<mml:mi mathvariant="italic">response</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mfrac>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mfenced open="|" close="|">
<mml:mrow>
<mml:mi mathvariant="italic">stimulus</mml:mi>
<mml:mo>&#x2229;</mml:mo>
<mml:mi mathvariant="italic">response</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mrow>
<mml:mo stretchy="true">|</mml:mo>
<mml:mi mathvariant="italic">stimulus</mml:mi>
<mml:mo stretchy="true">|</mml:mo>
<mml:mo>+</mml:mo>
<mml:mo stretchy="true">|</mml:mo>
<mml:mi mathvariant="italic">response</mml:mi>
<mml:mo stretchy="true">|</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mo>&#x2223;</mml:mo>
<mml:mi mathvariant="italic">stimulus</mml:mi>
<mml:mo>&#x2229;</mml:mo>
<mml:mi mathvariant="italic">response</mml:mi>
<mml:mo>&#x2223;</mml:mo>
</mml:mrow>
</mml:mfrac>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<p>For example, the stimulus word <italic>polish</italic> contains the phonemes P, AA, L, IH, SH. One participant&#x2019;s response was <italic>policy</italic>, containing the phonemes P, AA, L, AH, S, IY. The intersection contains three phonemes (P, AA, L) while the union contains 8 phonemes (P, AA, L, IH, SH, AH, S, IY) for a Jaccard index of 3/8&#x2009;=&#x2009;38%. In another example, the stimulus word <italic>ethic</italic> contains the phonemes EH, TH, IH, K while the response <italic>essay</italic> contains the phonemes EH, S, EY. The intersection contained 1 phoneme and the union contained 6 phonemes, for a Jaccard index 1/6&#x2009;=&#x2009;17%. See <xref ref-type="supplementary-material" rid="SM1">Supplementary material</xref> for a complete list of Jaccard indices.</p>
<p>The analysis was performed separately for each stimulus condition. For participant-level analysis, the mean accuracy across all trials was calculated for each participant, and then averaged across participants.</p>
<p>For the phoneme-specific analysis, the dependent variable was the number of times a phoneme was successfully identified vs. the total number of times that phoneme was presented across all words, for each participant, for each movie type (Real, DNN, and FACS). Because participants were not shown all possible phonemes for all movie types, we plot the estimated marginal means and standard errors derived from the GLME.</p>
<sec id="sec7">
<title>Additional stimulus details</title>
<p>The original stimulus material consisted of 32 audiovisual words recorded by a female talker and 32 words recorded by a male talker. Pink noise was added to the auditory track of each recording at a signal-to-noise ratio (SNR) of &#x2212;12&#x2009;dB. For the An format, only the noisy auditory recording was played with no visual stimulus. For the AnV:<italic>Real</italic> format, the audio recording was accompanied by the original video recording. For the synthetic faces, gender-matched synthetic faces roughly approximating the appearance of the real talkers were created.</p>
<p>During the online testing procedure, videos were presented using a custom JavaScript routine that ensured that all stimuli were presented with the same dimensions (height: 490 pixels; width: 872 pixels) regardless of the participant&#x2019;s device.</p>
<p>AnV:<italic>FACS</italic> words were created using JALI software (<xref ref-type="bibr" rid="ref9">Edwards et al., 2016</xref>).<xref ref-type="fn" rid="fn0003"><sup>3</sup></xref> The text transcript was manually tuned to create more pronounced mouth movements (e.g., AWLTHOH for although; see <xref ref-type="supplementary-material" rid="SM1">Supplementary material</xref> for a complete list of the phonetic spelling). Following JALI animation, the mouth movements were manually adjusted with MAYA&#x2019;s graph editor to better match the mouth movements in the real videos. The edited animation sequence was imported into Unreal Engine 5.10 and rendered as 16:9 images at 50&#x2009;mm focal length and anti-aliasing with 16 spatial sample count. Image sequences were assembled into mp4 format and aligned with the original audio track in Adobe Premiere. The video frame rates was 24 fps.</p>
<p>AnV:<italic>DNN</italic> words were created with the API for D-ID studio.<xref ref-type="fn" rid="fn0004"><sup>4</sup></xref> Two artificial faces included with D-ID studio were used as the base face. Clear audio files from the real speakers were uploaded, and a static driver face was used to minimize head and neck movements. Eye blinking and watermarks in each movie were removed using Adobe Premiere and exported in mp4 format at 24 fps.</p>
</sec>
</sec>
<sec id="sec8">
<title>Power analysis</title>
<p>In a pilot study, the audiovisual benefit for real faces and synthetic (FACS) faces was measured in 34 participants using 60 words from the main experiment. Accuracy for real faces was 23% (<italic>SD</italic>&#x2009;=&#x2009;13%) greater than synthetic faces. Using this effect size (1.68) in a power analysis with G&#x002A;Power software estimated that only six participants would be required for 90% power to detect a real vs. synthetic difference (<italic>t</italic>-test, difference between two dependent means). However, because we expected a smaller difference between the two synthetic face types (FACS and DNN), a more conservative effect size estimate of 0.5 was substituted. A corrected-alpha level of 0.0167 (0.05/3, to account for three comparisons) resulted in an estimate of 57 participants for 90% power. In anticipation of excluding some participants, five additional participants were recruited, for a total of 62. Only one participant was excluded, resulting in a final sample size of 61.</p>
</sec>
</sec>
<sec sec-type="results" id="sec9">
<title>Results</title>
<sec id="sec10">
<title>Participant-level analysis: word accuracy</title>
<p>In the first analysis, responses were scored as correct if they exactly matched the stimulus word and incorrect otherwise. Seeing the face of the talker improved the intelligibility of noisy auditory words. For real faces, accuracy increased from 10% in the auditory-only condition (An) to 59% in the audiovisual condition (AnV:<italic>Real</italic>), averaged across words and participants. There was also an improvement, albeit smaller, for synthetic faces (<xref ref-type="fig" rid="fig2">Figure 2A</xref>). From the auditory-only baseline of 10%, accuracy improved to 29% with faces generated by the facial action coding system (AnV: <italic>FACS</italic>). Accuracy was 30% for faces generated by a deep neural network (AnV:<italic>DNN</italic>). While there was a range of accuracies across participants, accuracy was higher for the real face format than the auditory-only format in 60 of 61 participants and higher for real faces than synthetic faces (<italic>Synthetic;</italic> average across <italic>DNN</italic> and <italic>FACS</italic>) in 59 of 61 participants (<xref ref-type="fig" rid="fig2">Figure 2B</xref>).</p>
<fig position="float" id="fig2">
<label>Figure 2</label>
<caption>
<p><bold>(A)</bold> For word-level scoring, the response was assessed as correct if it exactly matched the stimulus word and incorrect if it did not. Each bar shows the mean accuracy for each stimulus format (error bars show the standard error of the mean across participants). <bold>(B)</bold> Variability across participants assessed with word-level scoring (one symbol per participant). The <italic>y</italic>-axis shows the perceptual benefit of real faces (real minus auditory-only accuracy). The <italic>x</italic>-axis shows the perceptual benefit of synthetic faces (average of <italic>FACS</italic> and <italic>DNN</italic> accuracies minus auditory-only). Participants above the dashed line show a benefit for real faces compared with auditory-only. Participants above the solid identity line show a greater benefit for real faces than synthetic faces. <bold>(C)</bold> For phoneme-level scoring, the phonemic content of the stimulus and response were compared and the percentage of correct phonemes calculated for each stimulus format. <bold>(D)</bold> Variability across participants assessed with phoneme-level scoring (one symbol per participant).</p>
</caption>
<graphic xlink:href="fnins-18-1379988-g002.tif"/>
</fig>
<p>To estimate statistical significance, a mixed-effects model was constructed with a dependent variable of accuracy; fixed effect of word format; and random effects of word, participant and participant batch (complete model specification and output in <xref ref-type="supplementary-material" rid="SM1">Supplementary material</xref>). There was a main effect of stimulus format (<italic>&#x03C7;</italic><sup>2</sup><italic>
<sub>3</sub>
</italic>&#x2009;=&#x2009;768, <italic>p</italic>&#x2009;&#x003C;&#x2009;10<sup>&#x2212;16</sup>) and <italic>post hoc</italic> pair-wise comparisons showed that words accompanied by a visual face (real or synthetic) were perceived more accurately than auditory-only words (all <italic>p</italic>&#x2009;&#x003C;&#x2009;10<sup>&#x2212;16</sup>). The accuracy for real faces was higher than for synthetic faces (<italic>Real</italic> vs. <italic>DNN</italic>; <italic>t</italic>&#x2009;=&#x2009;&#x2212;17, <italic>p</italic>&#x2009;&#x003C;&#x2009;10<sup>&#x2212;16</sup>; <italic>Real</italic> vs. <italic>FACS</italic>; <italic>t</italic>&#x2009;=&#x2009;&#x2212;17, <italic>p</italic>&#x2009;&#x003C;&#x2009;10<sup>&#x2212;16</sup>) but accuracy for the two synthetic face formats was equivalent (<italic>DNN</italic> vs. <italic>FACS</italic>; <italic>t</italic>&#x2009;=&#x2009;0.2, <italic>p</italic>&#x2009;=&#x2009;0.99).</p>
</sec>
<sec id="sec11">
<title>Participant-level analysis: phoneme accuracy</title>
<p>In a second analysis, instead of classifying each response as either correct or incorrect, partial credit was given if the response contained phonemes that matched those in the stimulus word. This scoring method generated a phoneme accuracy score for each condition in each participant. The pattern of results were very similar to the word accuracy analysis (<xref ref-type="fig" rid="fig2">Figure 2C</xref> <xref ref-type="supplementary-material" rid="SM1">Supplementary material</xref>). There was a main effect of stimulus format (<italic>&#x03C7;</italic><sup>2</sup><italic>
<sub>3</sub>
</italic>&#x2009;=&#x2009;986, <italic>p</italic>&#x2009;&#x003C;&#x2009;10<sup>&#x2212;16</sup>) and <italic>post hoc</italic> pair-wise comparisons showed that words accompanied by a visual face (real or synthetic) were perceived more accurately than auditory-only words (all <italic>p</italic>&#x2009;&#x003C;&#x2009;10<sup>&#x2212;16</sup>). The accuracy for real faces was higher than for synthetic faces (<italic>Real</italic> vs. <italic>DNN</italic>; <italic>t</italic>&#x2009;=&#x2009;&#x2212;17, <italic>p</italic>&#x2009;&#x003C;&#x2009;10<sup>&#x2212;16</sup>; <italic>Real</italic> vs. <italic>FACS</italic>; <italic>t</italic>&#x2009;=&#x2009;&#x2212;18, <italic>p</italic>&#x2009;&#x003C;&#x2009;10<sup>&#x2212;16</sup>) but equivalent for the two synthetic face formats (<italic>DNN</italic> vs. <italic>FACS</italic>; <italic>t</italic>&#x2009;=&#x2009;0.9, <italic>p</italic>&#x2009;=&#x2009;0.82). Accuracy was higher for real faces than synthetic faces in every participant (<xref ref-type="fig" rid="fig2">Figure 2D</xref>).</p>
</sec>
<sec id="sec12">
<title>Phoneme-level analysis</title>
<p>In a third analysis, the accuracy difference between real and synthetic faces was examined separately for each phoneme (<xref ref-type="fig" rid="fig3">Figure 3A</xref>). For 38 of 39 phonemes, accuracy was higher for real faces than synthetic faces, and this difference was significant for 19 of 39 phonemes (after Bonferroni-correction for multiple comparisons). For four phonemes (<italic>/th/</italic>, <italic>/dh/</italic>, <italic>/f/</italic>, <italic>/v/</italic>) the accuracy advantage of real faces was especially pronounced (<italic>Real</italic> &#x003E;&#x003E; <italic>Synthetic</italic>). This observation could arise because these four phonemes had high accuracy for real faces, low accuracy for synthetic faces, or both. To distinguish these possibilities, we calculated the mean <italic>Real</italic> accuracy for (<italic>/th/</italic>, <italic>/dh/</italic>, <italic>/f/</italic>, <italic>/v/</italic>) compared with other phonemes, and found little difference (78% vs. 78%, <italic>t</italic>&#x2009;=&#x2009;&#x2212;0.2, <italic>p</italic>&#x2009;=&#x2009;0.81, paired samples <italic>t</italic>-test). In contrast, the mean <italic>Synthetic</italic> accuracy was significantly lower for (<italic>/th/</italic>, <italic>/dh/</italic>, <italic>/f/</italic>, <italic>/v/</italic>) than for other phonemes (28% vs. 61%, <italic>t</italic>&#x2009;=&#x2009;&#x2212;25, <italic>p</italic>&#x2009;=&#x2009;10<sup>&#x2212;16</sup>). Thus, the greater real-synthetic difference for <italic>/th/</italic>, <italic>/dh/</italic>, <italic>/f/</italic>, <italic>/v/</italic> than other phonemes (50% vs. 17%, <italic>t</italic>&#x2009;=&#x2009;11, <italic>p</italic>&#x2009;=&#x2009;10<sup>&#x2212;16</sup>) was attributable to particularly low <italic>Synthetic</italic> accuracy (<xref ref-type="fig" rid="fig3">Figure 3B</xref>).</p>
<fig position="float" id="fig3">
<label>Figure 3</label>
<caption>
<p><bold>(A)</bold> For each phoneme, the perceptual accuracy was calculated separately for each stimulus format across all participants. The accuracy for audiovisual synthetic faces (average of <italic>DNN</italic> and <italic>FACS</italic>) was subtracted from the accuracy for audiovisual real faces to generate a single value for each phoneme. For plotting, phonemes were sorted by the difference value. Four phonemes (boldface labels and left four bars colored dark green) showed the largest real-synthetic difference. Phonemes marked with asterisk showed a significantly greater accuracy for real faces after correction for multiple comparisons. <bold>(B)</bold> Four phonemes (<italic>/th/</italic>, <italic>/dh/</italic>, <italic>/v/</italic>, <italic>/f/</italic>) showed the largest real-synthetic difference phonemes (left plot). Average of these four phonemes shown by dark green bar, average of all other phonemes shown by light green bar, error bar shows SEM. This was not due to differences in the real face condition (middle plot) but rather to low accuracy for the top four phonemes in the synthetic face condition (right plot). <bold>(C)</bold> Enlargement of the mouth region for the three different face formats for words containing (<italic>/th/</italic>, <italic>/dh/</italic>, <italic>/v/</italic>, <italic>/f/</italic>). Enlargement for illustration only, participants viewed the entire face, as shown in <xref ref-type="fig" rid="fig1">Figure 1</xref>. Top row: video frame 18 from the word <italic>thank</italic>. Second row: video frame 30 from the word <italic>loathe</italic>. Third row: video frame 27 from the word <italic>chief</italic>. Fourth row: Video frame 16 from the word <italic>voice</italic>.</p>
</caption>
<graphic xlink:href="fnins-18-1379988-g003.tif"/>
</fig>
<p>The poor accuracy for synthetic <italic>/th/</italic>, <italic>/dh/</italic>, <italic>/f/</italic>, <italic>/v/</italic> suggested that some key visual features might be missing (<xref ref-type="fig" rid="fig3">Figure 3C</xref>). For <italic>/th/</italic> and <italic>/dh/</italic>, the salient visual feature is the tongue sandwiched between the teeth. This feature was clearly visible in the real face videos but was absent from the <italic>DNN</italic> and <italic>FACS</italic> face videos. For <italic>/f/</italic> and <italic>/v/</italic>, the salient visual feature is the upper teeth pressed onto the lower lip. This feature was obvious in the real face videos but not in the synthetic face videos.</p>
</sec>
<sec id="sec13">
<title>Modeling the effects of improving the four phonemes</title>
<p>Advances in computer graphics will make it possible to create synthetic faces that more accurately depict <italic>/th/</italic>, <italic>/dh/</italic>, <italic>/f/</italic>, <italic>/v/</italic> and thereby increase the synthetic face benefit for words containing these phonemes. To estimate this increase, we constructed a logistic model that predicted the word accuracy based on the presence or absence of every different phoneme in the word. The fitted model contained one coefficient for each synthetic face phone me and one coefficient for each real face phoneme, and was a good fit to the data (<italic>r</italic><sup>2</sup>&#x2009;=&#x2009;0.85, <italic>p</italic>&#x2009;&#x003C;&#x2009;10<sup>&#x2212;16</sup>). In the hypothetical best case, the improved synthetic versions of <italic>/th/</italic>, <italic>/dh/</italic>, <italic>/f/</italic>, <italic>/v/</italic> would be as good as the real face versions. This was simulated in the model by replacing the synthetic face coefficients for these phonemes with the real face coefficients. With this adjustment, the model predicted a word accuracy of 43%, compared with 29% for the original versions of <italic>/th/</italic>, <italic>/dh/</italic>, <italic>/f/</italic>, <italic>/v/</italic>, suggesting that improving the quality of the synthetic faces for the four phonemes with very low synthetic benefit could significantly boost overall accuracy.</p>
</sec>
<sec id="sec14">
<title>Effects of alternative stimulus material</title>
<p>The overall phoneme accuracy rates shown in <xref ref-type="fig" rid="fig2">Figure 2C</xref> were determined by the phonemic content of the 64 tested words. This raises the question of how the results might differ for a much larger corpus of words. With a large enough corpus, the frequency of phonemes should match their overall prevalence in the English language. To simulate an experiment with a large corpus, we weighted the real-synthetic difference for each phoneme by its prevalence in the English language (<xref ref-type="bibr" rid="ref14">Hayden, 1950</xref>). This procedure predicted an overall auditory-only phoneme accuracy of 40%, compared with 37% for the actual stimulus set. The predicted synthetic face phoneme accuracy was 60%, compared with 58% for the actual stimulus set, and the predicted real face accuracy was 78%, compared with 79% for the tested words. The predicted real-synthetic difference was 18% compared with 21% for the actual stimulus set. Testing with a large corpus of words (or a smaller set of words that matched overall English-language phoneme frequency) should lead to only small changes in accuracy.</p>
</sec>
<sec id="sec15">
<title>Differences between synthetic face types</title>
<p>The accuracy for DNN and FACS faces was similar for the word analysis (<xref ref-type="fig" rid="fig4">Figure 4A</xref>; <italic>t</italic>&#x2009;=&#x2009;0.2, <italic>p</italic>&#x2009;=&#x2009;0.99) and the phoneme analysis (<xref ref-type="fig" rid="fig4">Figure 4B</xref>; <italic>t</italic>&#x2009;=&#x2009;0.9, <italic>p</italic>&#x2009;=&#x2009;0.82) leading us to combine both conditions into a single &#x201C;synthetic face&#x201D; accuracy for the analyses presented above. To search for more subtle differences between the synthetic face types, we compared DNN and FACS accuracy for individual phonemes (<xref ref-type="fig" rid="fig4">Figure 4C</xref>). 51% (20 out of 39) of the phonemes had higher accuracy for DNN faces vs. FACS faces, but after correction for multiple comparisons, the difference was significant for only three phonemes (<italic>/b/</italic>, <italic>/p/</italic> and <italic>/aw/</italic>), all with higher accuracy for DNN faces. An examination of word videos containing these phonemes did not reveal any obvious differences between the mouth movements of FACS and DNN faces.</p>
<fig position="float" id="fig4">
<label>Figure 4</label>
<caption>
<p><bold>(A)</bold> Variability across participants for synthetic faces assessed with word-level scoring (one symbol per participant). The <italic>y</italic>-axis shows the perceptual benefit of synthetic DNN faces (DNN minus auditory-only accuracy). The <italic>x</italic>-axis shows the perceptual benefit of synthetic FACS faces (FACS minus auditory-only). <bold>(B)</bold> Variability across participants for synthetic faces assessed with phoneme-level scoring (one symbol per participant). For phoneme-level scoring, the phonemic content of the stimulus and response were compared and the percentage of correct phonemes calculated. <bold>(C)</bold> For each phoneme, the perceptual accuracy was calculated separately for DNN and FACS synthetic faces and subtracted (positive values indicate higher accuracy for DNN than FACS, negative values the opposite). Three phonemes (boldface labels, and star) showed significant difference in accuracy between formats (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.05, corrected for multiple comparisons). Phoneme order identical to <xref ref-type="fig" rid="fig3">Figure 3A</xref> to facilitate comparison.</p>
</caption>
<graphic xlink:href="fnins-18-1379988-g004.tif"/>
</fig>
</sec>
<sec id="sec16">
<title>Individual differences</title>
<p>There was substantial variability in the benefit of visual speech to noisy speech perception across individuals. To determine if these individual differences were consistent, we correlated the visual benefit for different face types. For word accuracy (<xref ref-type="fig" rid="fig2">Figure 2B</xref>), there was a strong correlation across participants between the benefit of real and synthetic faces (<italic>r</italic> =&#x2009;0.49, <italic>p</italic> =&#x2009;10<sup>&#x2212;4</sup>). The same was true for phoneme accuracy (<xref ref-type="fig" rid="fig2">Figure 2C</xref>; <italic>r</italic> =&#x2009;0.71, <italic>p</italic> =&#x2009;10<sup>&#x2212;10</sup>). Comparing the two types of synthetic faces, there was a positive correlation between the benefit of DNN and JALI faces for both word accuracy (<xref ref-type="fig" rid="fig4">Figure 4A</xref>; <italic>r</italic> =&#x2009;0.39, <italic>p</italic> =&#x2009;0.002) and phoneme accuracy (<xref ref-type="fig" rid="fig4">Figure 4B</xref>; <italic>r</italic> =&#x2009;0.47, <italic>p</italic> =&#x2009;10<sup>&#x2212;4</sup>).</p>
</sec>
</sec>
<sec sec-type="discussion" id="sec17">
<title>Discussion</title>
<p>Our study replicates decades of research by showing that seeing the face of a real talker improves speech-in-noise perception (<xref ref-type="bibr" rid="ref30">Sumby and Pollack, 1954</xref>; <xref ref-type="bibr" rid="ref21">Peelle and Sommers, 2015</xref>). Our study also confirms two recent reports that viewing a synthetic face generated by a deep neural network (DNN) significantly improves speech-in-noise perception (<xref ref-type="bibr" rid="ref28">Shan et al., 2022</xref>; <xref ref-type="bibr" rid="ref36">Varano et al., 2022</xref>). Both the present study and these previous reports found that the improvement from viewing DNN faces was only about half that provided by viewing real faces.</p>
<p>To determine if some idiosyncrasy of DNN faces was responsible for their poor performance relative to real faces, we also tested synthetic faces generated with a completely different technique, the facial action coding system (FACS). Since FACS explicitly models the relationship between the facial musculature and visual speech movements, we anticipated that it might provide more benefit to speech perception than DNN faces. Instead, the perceptual benefit of FACS faces was very similar to that of DNN faces.</p>
<p>Across participants, there was a positive correlation between the benefit of real and synthetic faces, leading us to infer that observers extract similar visual speech information from both kinds of faces, but that less visual speech information is available in synthetic faces. To better understand the real-synthetic difference, we decomposed stimulus words and participant responses into their component phonemes. This analysis revealed variability across phonemes. Four phonemes (<italic>/th/</italic>, <italic>/dh/</italic>, <italic>/f/</italic>, <italic>/v/</italic>) had an especially large real-synthetic difference, driven by low performance for both DNN and FACS faces. Examining single frames of the real and synthetic videos for words containing these phonemes revealed an obvious cause for the reduced benefit of synthetic faces. The synthetic videos were missing the interactions between teeth, lip and tongue that are the diagnostic visual feature for <italic>/th/</italic>, <italic>/dh/</italic>, <italic>/f/</italic>, <italic>/v/</italic>. Without these visual cues, participants did not receive the visual information beneficial for detecting these phonemes in speech-in-noise.</p>
<sec id="sec18">
<title>The concept of visemes</title>
<p>Visemes can be defined as the set of mouth configurations used to pronounce the phonemes in a language and may be shared across different phonemes. For instance, (<italic>/f/</italic>, <italic>/v/</italic>) are visually similar labiodental fricatives that require talkers to place the top teeth on the lower lip. The acoustic difference is generated by voicing (<italic>/f/</italic> is unvoiced while <italic>/v/</italic> is voiced); this voicing difference does not provide a visible cue to an observer. Similarly, (<italic>/th/</italic>, <italic>/dh/</italic>) are dental fricatives, articulated with the tongue against the upper teeth, with <italic>/th/</italic> unvoiced and <italic>/dh/</italic> voiced. While there is no generally agreed upon set of English visemes, five common viseme classifications all place (<italic>/f/</italic>, <italic>/v/</italic>) in one viseme category and (<italic>/th/</italic>, <italic>/dh/</italic>) in a different viseme category (<xref ref-type="bibr" rid="ref7">Cappelletta and Harte, 2012</xref>). Our results confirm the veracity of this grouping.</p>
</sec>
<sec id="sec19">
<title>How to improve synthetic faces</title>
<p>For the FACS faces, it would be possible to manually control teeth and tongue positioning using the underlying 3D face models, although this would be a time-consuming process. Alternately, the automated software used to animate the 3D face models (JALI)(<xref ref-type="bibr" rid="ref9">Edwards et al., 2016</xref>) could be modified to automatically code teeth and tongue positioning. For DNN faces, it is less straightforward to incorporate dental and labial interactions. The DNN models are trained on thousands or millions of examples of auditory and visual speech, and the network learns the correspondence between particular sounds and visual features. It may be that dental and labial interactions are highly variable across talkers, or not easily visible in the videos used for training, resulting in their absence in the final output. A common step in neural network model creation is fine-tuning. Incorporating training data that explicitly includes dental and labial features, such as from electromagnetic articulography (<xref ref-type="bibr" rid="ref26">Sch&#x00F6;nle et al., 1987</xref>) or MRI (<xref ref-type="bibr" rid="ref3">Baer et al., 1987</xref>), would improve the DNN&#x2019;s ability to depict these features.</p>
</sec>
<sec id="sec20">
<title>Other factors</title>
<p>While four phonemes showed an especially large real-synthetic difference (mean of 48% for <italic>/th/</italic>, <italic>/dh/</italic>, <italic>/f/</italic>, <italic>/v/</italic>) there was also a substantial difference for the remaining phonemes (mean of 16%). The origin of this difference is likely to be multi-faceted. One likely contributing factor is that just as for <italic>/th/</italic>, <italic>/dh/</italic>, <italic>/f/</italic>, <italic>/v/</italic>, the diagnostic mouth features for other phonemes are not as accurately depicted or as obvious in the synthetic faces as they are in the real face videos. This possibility could be tested by showing the most diagnostic single frame from each type of video and asking participants to guess the phoneme being spoken. The prediction is that, even for single video frames, synthetic performance would be worse than real face performance. Differences between DNN faces and real faces when pronouncing particular phonemes has been proposed as a method to detect deep-fake videos (<xref ref-type="bibr" rid="ref1">Agarwal et al., 2020</xref>) although the phonemes examined in their study (<italic>/m/</italic>, <italic>/b/</italic>, <italic>/p/</italic>) were not those that showed the largest real-synthetic difference in the present study.</p>
<p>For real talkers, information about speech content is available throughout the face. Since synthetic face generation concentrates on the mouth and lip region, this decreases the information about speech content available to observers (<xref ref-type="bibr" rid="ref19">Munhall and Vatikiotis-Bateson, 2004</xref>).</p>
<p>Another contributing factor for the real-synthetic difference could be temporal asynchrony, since the temporal alignment between auditory and visual speech contributes to causal inference and other perceptual processing underlying speech perception (<xref ref-type="bibr" rid="ref15">Magnotti et al., 2013</xref>; <xref ref-type="bibr" rid="ref6">Bhat et al., 2015</xref>). Synthetic faces were aligned to the auditory speech recording using a software processing pipeline; the JALI software used to create the FACS faces claims an audiovisual alignment accuracy of 15&#x2009;ms (<xref ref-type="bibr" rid="ref9">Edwards et al., 2016</xref>). Asynchrony produced by dubbing synthetic faces and real voices could be replicated for the real face condition by dubbing real faces and auditory recordings from a different talker or a synthesized voice.</p>
</sec>
<sec id="sec21">
<title>Relevance to experimental studies of audiovisual speech perception</title>
<p>An important reason for creating synthetic talking faces is to investigate the perceptual and neural properties of audiovisual speech perception (<xref ref-type="bibr" rid="ref31">Th&#x00E9;z&#x00E9; et al., 2020a</xref>,<xref ref-type="bibr" rid="ref32">b</xref>). In the well-known illusion known as the McGurk effect, incongruent auditory and visual speech leads to unexpected percepts (<xref ref-type="bibr" rid="ref17">McGurk and MacDonald, 1976</xref>). However, different McGurk stimuli vary widely in their efficacy. For instance, <xref ref-type="bibr" rid="ref4">Basu Mallick et al. (2015)</xref> tested 12 different McGurk stimuli used in published studies, and found that the strongest evoked the illusion on 58% of trials while the weakest stimulus evoked the illusion on only 17% of trials. The causes of high inter-stimulus variability are difficult to study, as talkers are limited in their ability to control the visual aspects of speech production. In contrast, synthetic faces, especially those created with FACS and related techniques, provide the ability to experimentally manipulate visual speech (<xref ref-type="bibr" rid="ref31">Th&#x00E9;z&#x00E9; et al., 2020a</xref>; <xref ref-type="bibr" rid="ref28">Shan et al., 2022</xref>; <xref ref-type="bibr" rid="ref36">Varano et al., 2022</xref>) making them a key tool for improving our understanding of the McGurk effect other incongruent audiovisual speech (<xref ref-type="bibr" rid="ref8">Dias et al., 2016</xref>; <xref ref-type="bibr" rid="ref27">Shahin, 2019</xref>).</p>
</sec>
<sec id="sec22">
<title>Individual differences</title>
<p>Two findings of the present study are consistent with decades of research. First, that seeing the face of the talker is beneficial for noisy speech perception and second, that the degree of benefit varies widely between individuals (<xref ref-type="bibr" rid="ref30">Sumby and Pollack, 1954</xref>; <xref ref-type="bibr" rid="ref12">Erber, 1975</xref>; <xref ref-type="bibr" rid="ref13">Grant et al., 1998</xref>; <xref ref-type="bibr" rid="ref33">Tye-Murray et al., 2008</xref>; <xref ref-type="bibr" rid="ref34">Van Engen et al., 2014</xref>, <xref ref-type="bibr" rid="ref35">2017</xref>; <xref ref-type="bibr" rid="ref21">Peelle and Sommers, 2015</xref>; <xref ref-type="bibr" rid="ref29">Sommers et al., 2020</xref>). Individual differences are observed even when participants&#x2019; eye movements are monitored, ruling out the trivial explanation that participants with low visual benefit fail to look at the visual display (<xref ref-type="bibr" rid="ref24">Rennig et al., 2020</xref>). However, eye movements to particular parts of the talker&#x2019;s face (specifically, a preference for foveating the mouth of the talker when viewing clear speech), combined with recognition performance during auditory-only noisy speech, explain about 10% of the variability across individuals (<xref ref-type="bibr" rid="ref24">Rennig et al., 2020</xref>). At the neural level, fMRI response patterns in superior temporal cortex to clear and noisy speech are more similar in participants with a larger benefit from seeing the face of the talker (<xref ref-type="bibr" rid="ref23">Rennig and Beauchamp, 2022</xref>; <xref ref-type="bibr" rid="ref37">Zhang et al., 2023</xref>).</p>
</sec>
<sec id="sec23">
<title>Limitations of the present study</title>
<p>The present study has a number of limitations. In order to maximize the number of tested words and minimize experimental time, only a single noise level was tested, as in a previous study of DNN faces (<xref ref-type="bibr" rid="ref36">Varano et al., 2022</xref>). A high level of noise (&#x2212;12&#x2009;dB) was selected to maximize the benefit of visual speech (<xref ref-type="bibr" rid="ref25">Ross et al., 2007</xref>; <xref ref-type="bibr" rid="ref24">Rennig et al., 2020</xref>). Another previous study of DNN faces tested multiple noise levels and found a lawful relationship between different noise levels and perception (<xref ref-type="bibr" rid="ref28">Shan et al., 2022</xref>). As the amount of added auditory noise decreased, accuracy increased for the no-face, real face and DNN face conditions in parallel, converging at ceiling accuracy for all three conditions when no auditory noise was added. We would expect a similar pattern if our experiments were repeated with different levels of auditory noise. Another experimental approach is to present visual-only speech without any auditory input, although this condition differs from most real-world situations with the exception of profound deafness. The ability to extract information about speech from the face of the talker (known as lipreading or speechreading) varies widely across individuals, and it may be possible to improve this ability through training (<xref ref-type="bibr" rid="ref2">Auer and Bernstein, 2007</xref>; <xref ref-type="bibr" rid="ref5">Bernstein et al., 2022</xref>).</p>
<p>While our study only examined speech perception, a similar approach could be taken to compare real and synthetic faces in other domains, such as emotions and looking behavior (<xref ref-type="bibr" rid="ref18">Miller et al., 2023</xref>).</p>
</sec>
</sec>
<sec sec-type="conclusions" id="sec24">
<title>Conclusion</title>
<p><xref ref-type="bibr" rid="ref16">Massaro and Cohen (1995)</xref> pioneered the use of synthetic faces to examine audiovisual speech perception and recent advances in computer graphics and deep neural faces show that synthetic faces offer a promising tool for both research and practical applications to help patients with deficits in speech perception.</p>
</sec>
<sec sec-type="data-availability" id="sec25">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="supplementary-material" rid="SM1">Supplementary material</xref>, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec sec-type="ethics-statement" id="sec26">
<title>Ethics statement</title>
<p>The studies involving humans were approved by Institutional Review Board of the University of Pennsylvania, Philadelphia, PA. The studies were conducted in accordance with the local legislation and institutional requirements. The ethics committee/institutional review board waived the requirement of written informed consent for participation from the participants or the participants&#x2019; legal guardians/next of kin because Participants were recruited and tested online. Written informed consent was obtained from the individual(s) for the publication of any potentially identifiable images or data included in this article.</p>
</sec>
<sec sec-type="author-contributions" id="sec27">
<title>Author contributions</title>
<p>YY: Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. AL: Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. YZ: Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. JM: Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. MB: Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing.</p>
</sec>
</body>
<back>
<sec sec-type="funding-information" id="sec28">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This research was supported by NIH R01NS065395 and U01NS113339. The funders had no role in study design, data collection and analysis, decision to publish, or preparation of the manuscript.</p>
</sec>
<sec sec-type="COI-statement" id="sec29">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="sec100" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="sec30">
<title>Supplementary material</title>
<p>The Supplementary material for this article can be found online at: <ext-link xlink:href="https://www.frontiersin.org/articles/10.3389/fnins.2024.1379988/full#supplementary-material" ext-link-type="uri">https://www.frontiersin.org/articles/10.3389/fnins.2024.1379988/full#supplementary-material</ext-link></p>
<supplementary-material xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="Table_1.XLSX" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"/>
<supplementary-material xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="Data_Sheet_1.zip" id="SM2" mimetype="application/zip"/>
</sec>
<fn-group>
<fn id="fn0001">
<p><sup>1</sup><ext-link xlink:href="https://www.mturk.com/" ext-link-type="uri">https://www.mturk.com/</ext-link>
</p>
</fn>
<fn id="fn0002">
<p><sup>2</sup><ext-link xlink:href="http://www.speech.cs.cmu.edu/cgi-bin/cmudict" ext-link-type="uri">http://www.speech.cs.cmu.edu/cgi-bin/cmudict</ext-link>
</p>
</fn>
<fn id="fn0003">
<p><sup>3</sup><ext-link xlink:href="https://jaliresearch.com/" ext-link-type="uri">https://jaliresearch.com/</ext-link>
</p>
</fn>
<fn id="fn0004">
<p><sup>4</sup><ext-link xlink:href="https://www.d-id.com/" ext-link-type="uri">https://www.d-id.com/</ext-link>
</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="ref1"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Agarwal</surname> <given-names>S.</given-names></name> <name><surname>Farid</surname> <given-names>H.</given-names></name> <name><surname>Fried</surname> <given-names>O.</given-names></name> <name><surname>Agrawala</surname> <given-names>M.</given-names></name></person-group>, (<year>2020</year>). <article-title>Detecting deep-fake videos from phoneme-Viseme mismatches</article-title>, in: <conf-name>2020 IEEE/CVF Conference on Computer Vision and Pattern Recognition Workshops (CVPRW). Presented at the 2020 IEEE/CVF Conference on Computer Vision and Pattern Recognition Workshops (CVPRW)</conf-name>, <fpage>2814</fpage>&#x2013;<lpage>2822</lpage>.</citation></ref>
<ref id="ref2"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Auer</surname> <given-names>E. T.</given-names> <suffix>Jr.</suffix></name> <name><surname>Bernstein</surname> <given-names>L. E.</given-names></name></person-group> (<year>2007</year>). <article-title>Enhanced visual speech perception in individuals with early-onset hearing impairment</article-title>. <source>J. Speech Lang. Hear. Res.</source> <volume>50</volume>, <fpage>1157</fpage>&#x2013;<lpage>1165</lpage>. doi: <pub-id pub-id-type="doi">10.1044/1092-4388(2007/080)</pub-id>, PMID: <pub-id pub-id-type="pmid">17905902</pub-id></citation></ref>
<ref id="ref3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Baer</surname> <given-names>T.</given-names></name> <name><surname>Gore</surname> <given-names>J. C.</given-names></name> <name><surname>Boyce</surname> <given-names>S.</given-names></name> <name><surname>Nye</surname> <given-names>P. W.</given-names></name></person-group> (<year>1987</year>). <article-title>Application of MRI to the analysis of speech production</article-title>. <source>Magn. Reson. Imaging</source> <volume>5</volume>, <fpage>1</fpage>&#x2013;<lpage>7</lpage>. doi: <pub-id pub-id-type="doi">10.1016/0730-725x(87)90477-2</pub-id>, PMID: <pub-id pub-id-type="pmid">3586868</pub-id></citation></ref>
<ref id="ref4"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Basu Mallick</surname> <given-names>D.</given-names></name> <name><surname>Magnotti</surname> <given-names>J. F.</given-names></name> <name><surname>Beauchamp</surname> <given-names>M. S.</given-names></name></person-group> (<year>2015</year>). <article-title>Variability and stability in the McGurk effect: contributions of participants, stimuli, time, and response type</article-title>. <source>Psychon. Bull. Rev.</source> <volume>22</volume>, <fpage>1299</fpage>&#x2013;<lpage>1307</lpage>. doi: <pub-id pub-id-type="doi">10.3758/s13423-015-0817-4</pub-id>, PMID: <pub-id pub-id-type="pmid">25802068</pub-id></citation></ref>
<ref id="ref5"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bernstein</surname> <given-names>L. E.</given-names></name> <name><surname>Jordan</surname> <given-names>N.</given-names></name> <name><surname>Auer</surname> <given-names>E. T.</given-names></name> <name><surname>Eberhardt</surname> <given-names>S. P.</given-names></name></person-group> (<year>2022</year>). <article-title>Lipreading: a review of its continuing importance for speech recognition with an acquired hearing loss and possibilities for effective training</article-title>. <source>Am. J. Audiol.</source> <volume>31</volume>, <fpage>453</fpage>&#x2013;<lpage>469</lpage>. doi: <pub-id pub-id-type="doi">10.1044/2021_AJA-21-00112</pub-id>, PMID: <pub-id pub-id-type="pmid">35316072</pub-id></citation></ref>
<ref id="ref6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bhat</surname> <given-names>J.</given-names></name> <name><surname>Miller</surname> <given-names>L. M.</given-names></name> <name><surname>Pitt</surname> <given-names>M. A.</given-names></name> <name><surname>Shahin</surname> <given-names>A. J.</given-names></name></person-group> (<year>2015</year>). <article-title>Putative mechanisms mediating tolerance for audiovisual stimulus onset asynchrony</article-title>. <source>J. Neurophysiol.</source> <volume>113</volume>, <fpage>1437</fpage>&#x2013;<lpage>1450</lpage>. doi: <pub-id pub-id-type="doi">10.1152/jn.00200.2014</pub-id>, PMID: <pub-id pub-id-type="pmid">25505102</pub-id></citation></ref>
<ref id="ref7"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Cappelletta</surname> <given-names>L.</given-names></name> <name><surname>Harte</surname> <given-names>N.</given-names></name></person-group>, (<year>2012</year>). <article-title>Phoneme-to-Viseme mapping for visual speech recognition</article-title>. In <conf-name>Presented at the proceedings of the 1st international conference on pattern recognition applications and methods</conf-name>, <publisher-name>SciTePress</publisher-name>, <fpage>322</fpage>&#x2013;<lpage>329</lpage>.</citation></ref>
<ref id="ref8"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dias</surname> <given-names>J. W.</given-names></name> <name><surname>Cook</surname> <given-names>T. C.</given-names></name> <name><surname>Rosenblum</surname> <given-names>L. D.</given-names></name></person-group> (<year>2016</year>). <article-title>Influences of selective adaptation on perception of audiovisual speech</article-title>. <source>J. Phon.</source> <volume>56</volume>, <fpage>75</fpage>&#x2013;<lpage>84</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.wocn.2016.02.004</pub-id>, PMID: <pub-id pub-id-type="pmid">27041781</pub-id></citation></ref>
<ref id="ref9"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Edwards</surname> <given-names>P.</given-names></name> <name><surname>Landreth</surname> <given-names>C.</given-names></name> <name><surname>Fiume</surname> <given-names>E.</given-names></name> <name><surname>Singh</surname> <given-names>K.</given-names></name></person-group> (<year>2016</year>). <article-title>JALI: an animator-centric viseme model for expressive lip synchronization</article-title>. <source>ACM Trans. Graph.</source> <volume>35</volume>, <fpage>1</fpage>&#x2013;<lpage>11</lpage>. doi: <pub-id pub-id-type="doi">10.1145/2897824.2925984</pub-id></citation></ref>
<ref id="ref10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ekman</surname> <given-names>P.</given-names></name> <name><surname>Friesen</surname> <given-names>W. V.</given-names></name></person-group> (<year>1976</year>). <article-title>Measuring facial movement</article-title>. <source>J. Nonverbal Behav.</source> <volume>1</volume>, <fpage>56</fpage>&#x2013;<lpage>75</lpage>. doi: <pub-id pub-id-type="doi">10.1007/BF01115465</pub-id></citation></ref>
<ref id="ref11"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Ekman</surname> <given-names>P.</given-names></name> <name><surname>Friesen</surname> <given-names>W. V.</given-names></name></person-group>, (<year>1978</year>). <source>Facial Action Coding System (FACS) [Database record]</source>. APA PsycTests.</citation></ref>
<ref id="ref12"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Erber</surname> <given-names>N. P.</given-names></name></person-group> (<year>1975</year>). <article-title>Auditory-visual perception of speech</article-title>. <source>J. Speech Hear. Disord.</source> <volume>40</volume>, <fpage>481</fpage>&#x2013;<lpage>492</lpage>. doi: <pub-id pub-id-type="doi">10.1044/jshd.4004.481</pub-id></citation></ref>
<ref id="ref13"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Grant</surname> <given-names>K. W.</given-names></name> <name><surname>Walden</surname> <given-names>B. E.</given-names></name> <name><surname>Seitz</surname> <given-names>P. F.</given-names></name></person-group> (<year>1998</year>). <article-title>Auditory-visual speech recognition by hearing-impaired subjects: consonant recognition, sentence recognition, and auditory-visual integration</article-title>. <source>J. Acoust. Soc. Am.</source> <volume>103</volume>, <fpage>2677</fpage>&#x2013;<lpage>2690</lpage>. doi: <pub-id pub-id-type="doi">10.1121/1.422788</pub-id>, PMID: <pub-id pub-id-type="pmid">9604361</pub-id></citation></ref>
<ref id="ref14"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hayden</surname> <given-names>R. E.</given-names></name></person-group> (<year>1950</year>). <article-title>The relative frequency of phonemes in general-American English</article-title>. <source>Word</source> <volume>6</volume>, <fpage>217</fpage>&#x2013;<lpage>223</lpage>. doi: <pub-id pub-id-type="doi">10.1080/00437956.1950.11659381</pub-id></citation></ref>
<ref id="ref15"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Magnotti</surname> <given-names>J. F.</given-names></name> <name><surname>Ma</surname> <given-names>W. J.</given-names></name> <name><surname>Beauchamp</surname> <given-names>M. S.</given-names></name></person-group> (<year>2013</year>). <article-title>Causal inference of asynchronous audiovisual speech</article-title>. <source>Front. Psychol.</source> <volume>4</volume>:<fpage>798</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fpsyg.2013.00798</pub-id>, PMID: <pub-id pub-id-type="pmid">24294207</pub-id></citation></ref>
<ref id="ref16"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Massaro</surname> <given-names>D. W.</given-names></name> <name><surname>Cohen</surname> <given-names>M. M.</given-names></name></person-group> (<year>1995</year>). <article-title>Perceiving talking faces</article-title>. <source>Curr. Dir. Psychol. Sci.</source> <volume>4</volume>, <fpage>104</fpage>&#x2013;<lpage>109</lpage>. doi: <pub-id pub-id-type="doi">10.1111/1467-8721.ep10772401</pub-id></citation></ref>
<ref id="ref17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>McGurk</surname> <given-names>H.</given-names></name> <name><surname>MacDonald</surname> <given-names>J.</given-names></name></person-group> (<year>1976</year>). <article-title>Hearing lips and seeing voices</article-title>. <source>Nature</source> <volume>264</volume>, <fpage>746</fpage>&#x2013;<lpage>748</lpage>. doi: <pub-id pub-id-type="doi">10.1038/264746a0</pub-id>, PMID: <pub-id pub-id-type="pmid">1012311</pub-id></citation></ref>
<ref id="ref18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Miller</surname> <given-names>E. J.</given-names></name> <name><surname>Foo</surname> <given-names>Y. Z.</given-names></name> <name><surname>Mewton</surname> <given-names>P.</given-names></name> <name><surname>Dawel</surname> <given-names>A.</given-names></name></person-group> (<year>2023</year>). <article-title>How do people respond to computer-generated versus human faces? A systematic review and meta-analyses</article-title>. <source>Comput. Hum. Behav. Rep.</source> <volume>10</volume>:<fpage>100283</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.chbr.2023.100283</pub-id></citation></ref>
<ref id="ref19"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Munhall</surname> <given-names>K. G.</given-names></name> <name><surname>Vatikiotis-Bateson</surname> <given-names>E.</given-names></name></person-group> (<year>2004</year>). &#x201C;<article-title>Spatial and temporal constraints on audiovisual speech perception</article-title>&#x201D; in <source>The handbook of multisensory processes</source>. eds. <person-group person-group-type="editor"><name><surname>Calvert</surname> <given-names>G. A.</given-names></name> <name><surname>Spence</surname> <given-names>C.</given-names></name> <name><surname>Stein</surname> <given-names>B. E.</given-names></name></person-group> (<publisher-name>MIT Press</publisher-name>), <fpage>177</fpage>&#x2013;<lpage>188</lpage>.</citation></ref>
<ref id="ref20"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Parke</surname> <given-names>F. I.</given-names></name> <name><surname>Waters</surname> <given-names>K.</given-names></name></person-group>, (<year>2008</year>). <source>Computer facial animation</source>, <edition>2nd Edn</edition>, <publisher-name>A K Peters</publisher-name>, <publisher-loc>Wellesley, Mass</publisher-loc>.</citation></ref>
<ref id="ref21"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Peelle</surname> <given-names>J. E.</given-names></name> <name><surname>Sommers</surname> <given-names>M. S.</given-names></name></person-group> (<year>2015</year>). <article-title>Prediction and constraint in audiovisual speech perception</article-title>. <source>Cortex</source> <volume>68</volume>, <fpage>169</fpage>&#x2013;<lpage>181</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.cortex.2015.03.006</pub-id>, PMID: <pub-id pub-id-type="pmid">25890390</pub-id></citation></ref>
<ref id="ref22"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Perry</surname> <given-names>G.</given-names></name> <name><surname>Blondheim</surname> <given-names>S.</given-names></name> <name><surname>Kuta</surname> <given-names>E.</given-names></name></person-group>, (<year>2023</year>). A web app that lets you video chat with an AI on human terms. [WWW document]. D-ID. Available at: <ext-link xlink:href="https://www.d-id.com/chat/" ext-link-type="uri">https://www.d-id.com/chat/</ext-link> (Accessed January 20, 2024).</citation></ref>
<ref id="ref23"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rennig</surname> <given-names>J.</given-names></name> <name><surname>Beauchamp</surname> <given-names>M. S.</given-names></name></person-group> (<year>2022</year>). <article-title>Intelligibility of audiovisual sentences drives multivoxel response patterns in human superior temporal cortex</article-title>. <source>NeuroImage</source> <volume>247</volume>:<fpage>118796</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neuroimage.2021.118796</pub-id>, PMID: <pub-id pub-id-type="pmid">34906712</pub-id></citation></ref>
<ref id="ref24"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rennig</surname> <given-names>J.</given-names></name> <name><surname>Wegner-Clemens</surname> <given-names>K.</given-names></name> <name><surname>Beauchamp</surname> <given-names>M. S.</given-names></name></person-group> (<year>2020</year>). <article-title>Face viewing behavior predicts multisensory gain during speech perception</article-title>. <source>Psychon. Bull. Rev.</source> <volume>27</volume>, <fpage>70</fpage>&#x2013;<lpage>77</lpage>. doi: <pub-id pub-id-type="doi">10.3758/s13423-019-01665-y</pub-id>, PMID: <pub-id pub-id-type="pmid">31845209</pub-id></citation></ref>
<ref id="ref25"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ross</surname> <given-names>L. A.</given-names></name> <name><surname>Saint-Amour</surname> <given-names>D.</given-names></name> <name><surname>Leavitt</surname> <given-names>V. M.</given-names></name> <name><surname>Javitt</surname> <given-names>D. C.</given-names></name> <name><surname>Foxe</surname> <given-names>J. J.</given-names></name></person-group> (<year>2007</year>). <article-title>Do you see what I am saying? Exploring visual enhancement of speech comprehension in noisy environments</article-title>. <source>Cereb. Cortex</source> <volume>17</volume>, <fpage>1147</fpage>&#x2013;<lpage>1153</lpage>. doi: <pub-id pub-id-type="doi">10.1093/cercor/bhl024</pub-id>, PMID: <pub-id pub-id-type="pmid">16785256</pub-id></citation></ref>
<ref id="ref26"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sch&#x00F6;nle</surname> <given-names>P. W.</given-names></name> <name><surname>Gr&#x00E4;be</surname> <given-names>K.</given-names></name> <name><surname>Wenig</surname> <given-names>P.</given-names></name> <name><surname>H&#x00F6;hne</surname> <given-names>J.</given-names></name> <name><surname>Schrader</surname> <given-names>J.</given-names></name> <name><surname>Conrad</surname> <given-names>B.</given-names></name></person-group> (<year>1987</year>). <article-title>Electromagnetic articulography: use of alternating magnetic fields for tracking movements of multiple points inside and outside the vocal tract</article-title>. <source>Brain Lang.</source> <volume>31</volume>, <fpage>26</fpage>&#x2013;<lpage>35</lpage>. doi: <pub-id pub-id-type="doi">10.1016/0093-934x(87)90058-7</pub-id>, PMID: <pub-id pub-id-type="pmid">3580838</pub-id></citation></ref>
<ref id="ref27"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shahin</surname> <given-names>A. J.</given-names></name></person-group> (<year>2019</year>). <article-title>Neural evidence accounting for interindividual variability of the McGurk illusion</article-title>. <source>Neurosci. Lett.</source> <volume>707</volume>:<fpage>134322</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neulet.2019.134322</pub-id>, PMID: <pub-id pub-id-type="pmid">31181299</pub-id></citation></ref>
<ref id="ref28"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shan</surname> <given-names>T.</given-names></name> <name><surname>Wenner</surname> <given-names>C. E.</given-names></name> <name><surname>Xu</surname> <given-names>C.</given-names></name> <name><surname>Duan</surname> <given-names>Z.</given-names></name> <name><surname>Maddox</surname> <given-names>R. K.</given-names></name></person-group> (<year>2022</year>). <article-title>Speech-in-noise comprehension is improved when viewing a deep-neural-network-generated talking face</article-title>. <source>Trends Hear.</source> <volume>26</volume>:<fpage>23312165221136934</fpage>. doi: <pub-id pub-id-type="doi">10.1177/23312165221136934</pub-id>, PMID: <pub-id pub-id-type="pmid">36384325</pub-id></citation></ref>
<ref id="ref29"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sommers</surname> <given-names>M. S.</given-names></name> <name><surname>Spehar</surname> <given-names>B.</given-names></name> <name><surname>Tye-Murray</surname> <given-names>N.</given-names></name> <name><surname>Myerson</surname> <given-names>J.</given-names></name> <name><surname>Hale</surname> <given-names>S.</given-names></name></person-group> (<year>2020</year>). <article-title>Age differences in the effects of speaking rate on auditory, visual, and auditory-visual speech perception</article-title>. <source>Ear Hear.</source> <volume>41</volume>, <fpage>549</fpage>&#x2013;<lpage>560</lpage>. doi: <pub-id pub-id-type="doi">10.1097/AUD.0000000000000776</pub-id>, PMID: <pub-id pub-id-type="pmid">31453875</pub-id></citation></ref>
<ref id="ref30"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sumby</surname> <given-names>W. H.</given-names></name> <name><surname>Pollack</surname> <given-names>I.</given-names></name></person-group> (<year>1954</year>). <article-title>Visual contribution to speech intelligibility in noise</article-title>. <source>J. Acoust. Soc. Am.</source> <volume>26</volume>, <fpage>212</fpage>&#x2013;<lpage>215</lpage>. doi: <pub-id pub-id-type="doi">10.1121/1.1907309</pub-id></citation></ref>
<ref id="ref31"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Th&#x00E9;z&#x00E9;</surname> <given-names>R.</given-names></name> <name><surname>Gadiri</surname> <given-names>M. A.</given-names></name> <name><surname>Albert</surname> <given-names>L.</given-names></name> <name><surname>Provost</surname> <given-names>A.</given-names></name> <name><surname>Giraud</surname> <given-names>A.-L.</given-names></name> <name><surname>M&#x00E9;gevand</surname> <given-names>P.</given-names></name></person-group> (<year>2020a</year>). <article-title>Animated virtual characters to explore audio-visual speech in controlled and naturalistic environments</article-title>. <source>Sci. Rep.</source> <volume>10</volume>:<fpage>15540</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s41598-020-72375-y</pub-id>, PMID: <pub-id pub-id-type="pmid">32968127</pub-id></citation></ref>
<ref id="ref32"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Th&#x00E9;z&#x00E9;</surname> <given-names>R.</given-names></name> <name><surname>Giraud</surname> <given-names>A.-L.</given-names></name> <name><surname>M&#x00E9;gevand</surname> <given-names>P.</given-names></name></person-group> (<year>2020b</year>). <article-title>The phase of cortical oscillations determines the perceptual fate of visual cues in naturalistic audiovisual speech</article-title>. <source>Sci. Adv.</source> <volume>6</volume>:<fpage>eabc6348</fpage>. doi: <pub-id pub-id-type="doi">10.1126/sciadv.abc6348</pub-id>, PMID: <pub-id pub-id-type="pmid">33148648</pub-id></citation></ref>
<ref id="ref33"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tye-Murray</surname> <given-names>N.</given-names></name> <name><surname>Sommers</surname> <given-names>M.</given-names></name> <name><surname>Spehar</surname> <given-names>B.</given-names></name> <name><surname>Myerson</surname> <given-names>J.</given-names></name> <name><surname>Hale</surname> <given-names>S.</given-names></name> <name><surname>Rose</surname> <given-names>N. S.</given-names></name></person-group> (<year>2008</year>). <article-title>Auditory-visual discourse comprehension by older and young adults in favorable and unfavorable conditions</article-title>. <source>Int. J. Audiol.</source> <volume>47</volume>, <fpage>S31</fpage>&#x2013;<lpage>S37</lpage>. doi: <pub-id pub-id-type="doi">10.1080/14992020802301662</pub-id>, PMID: <pub-id pub-id-type="pmid">19012110</pub-id></citation></ref>
<ref id="ref34"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Van Engen</surname> <given-names>K. J.</given-names></name> <name><surname>Phelps</surname> <given-names>J. E. B.</given-names></name> <name><surname>Smiljanic</surname> <given-names>R.</given-names></name> <name><surname>Chandrasekaran</surname> <given-names>B.</given-names></name></person-group> (<year>2014</year>). <article-title>Enhancing speech intelligibility: interactions among context, modality, speech style, and masker</article-title>. <source>J. Speech Lang. Hear. Res.</source> <volume>57</volume>, <fpage>1908</fpage>&#x2013;<lpage>1918</lpage>. doi: <pub-id pub-id-type="doi">10.1044/JSLHR-H-13-0076</pub-id>, PMID: <pub-id pub-id-type="pmid">24687206</pub-id></citation></ref>
<ref id="ref35"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Van Engen</surname> <given-names>K. J.</given-names></name> <name><surname>Xie</surname> <given-names>Z.</given-names></name> <name><surname>Chandrasekaran</surname> <given-names>B.</given-names></name></person-group> (<year>2017</year>). <article-title>Audiovisual sentence recognition not predicted by susceptibility to the McGurk effect</article-title>. <source>Atten. Percept. Psychophys.</source> <volume>79</volume>, <fpage>396</fpage>&#x2013;<lpage>403</lpage>. doi: <pub-id pub-id-type="doi">10.3758/s13414-016-1238-9</pub-id>, PMID: <pub-id pub-id-type="pmid">27921268</pub-id></citation></ref>
<ref id="ref36"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Varano</surname> <given-names>E.</given-names></name> <name><surname>Vougioukas</surname> <given-names>K.</given-names></name> <name><surname>Ma</surname> <given-names>P.</given-names></name> <name><surname>Petridis</surname> <given-names>S.</given-names></name> <name><surname>Pantic</surname> <given-names>M.</given-names></name> <name><surname>Reichenbach</surname> <given-names>T.</given-names></name></person-group> (<year>2022</year>). <article-title>Speech-driven facial animations improve speech-in-noise comprehension of humans</article-title>. <source>Front. Neurosci.</source> <volume>15</volume>:<fpage>781196</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fnins.2021.781196</pub-id>, PMID: <pub-id pub-id-type="pmid">35069100</pub-id></citation></ref>
<ref id="ref37"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Rennig</surname> <given-names>J.</given-names></name> <name><surname>Magnotti</surname> <given-names>J. F.</given-names></name> <name><surname>Beauchamp</surname> <given-names>M. S.</given-names></name></person-group> (<year>2023</year>). <article-title>Multivariate fMRI responses in superior temporal cortex predict visual contributions to, and individual differences in, the intelligibility of noisy speech</article-title>. <source>NeuroImage</source> <volume>278</volume>:<fpage>120271</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neuroimage.2023.120271</pub-id>, PMID: <pub-id pub-id-type="pmid">37442310</pub-id></citation></ref>
<ref id="ref38"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>Y.</given-names></name> <name><surname>Xu</surname> <given-names>Z.</given-names></name> <name><surname>Landreth</surname> <given-names>C.</given-names></name> <name><surname>Kalogerakis</surname> <given-names>E.</given-names></name> <name><surname>Maji</surname> <given-names>S.</given-names></name> <name><surname>Singh</surname> <given-names>K.</given-names></name></person-group> (<year>2018</year>). <article-title>Visemenet: audio-driven animator-centric speech animation</article-title>. <source>ACM Trans. Graph.</source> <volume>37</volume>, <fpage>1</fpage>&#x2013;<lpage>10</lpage>. doi: <pub-id pub-id-type="doi">10.1145/3197517.3201292</pub-id></citation></ref>
</ref-list>
</back>
</article>