<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Neurosci.</journal-id>
<journal-title>Frontiers in Neuroscience</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Neurosci.</abbrev-journal-title>
<issn pub-type="epub">1662-453X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fnins.2021.744959</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Neuroscience</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Relative Weights of Temporal Envelope Cues in Different Frequency Regions for Mandarin Vowel, Consonant, and Lexical Tone Recognition</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Zheng</surname> <given-names>Zhong</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1037188/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Li</surname> <given-names>Keyi</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1392048/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Feng</surname> <given-names>Gang</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1391152/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Guo</surname> <given-names>Yang</given-names></name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/696877/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Li</surname> <given-names>Yinan</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1564798/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Xiao</surname> <given-names>Lili</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1171379/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Liu</surname> <given-names>Chengqi</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1233798/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>He</surname> <given-names>Shouhuan</given-names></name>
<xref ref-type="aff" rid="aff6"><sup>6</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1243414/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Zhang</surname> <given-names>Zhen</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1550373/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Qian</surname> <given-names>Di</given-names></name>
<xref ref-type="aff" rid="aff7"><sup>7</sup></xref>
<xref ref-type="corresp" rid="c002"><sup>&#x002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1550630/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Feng</surname> <given-names>Yanmei</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="c003"><sup>&#x002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1240669/overview"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Department of Otolaryngology-Head and Neck Surgery, Shanghai Jiao Tong University Affiliated Sixth People&#x2019;s Hospital</institution>, <addr-line>Shanghai</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>Shanghai Key Laboratory of Sleep Disordered Breathing</institution>, <addr-line>Shanghai</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>Sydney Institute of Language and Commerce, Shanghai University</institution>, <addr-line>Shanghai</addr-line>, <country>China</country></aff>
<aff id="aff4"><sup>4</sup><institution>Department of Graduate, The First Affiliated Hospital of Jinzhou Medical University</institution>, <addr-line>Jinzhou</addr-line>, <country>China</country></aff>
<aff id="aff5"><sup>5</sup><institution>Ear, Nose, and Throat Institute and Otorhinolaryngology Department, Eye and ENT Hospital of Fudan University</institution>, <addr-line>Shanghai</addr-line>, <country>China</country></aff>
<aff id="aff6"><sup>6</sup><institution>Department of Otolaryngology, Qingpu Branch of Zhongshan Hospital Affiliated to Fudan University</institution>, <addr-line>Shanghai</addr-line>, <country>China</country></aff>
<aff id="aff7"><sup>7</sup><institution>Department of Otolaryngology, Shenzhen Longhua District People&#x2019;s Hospital</institution>, <addr-line>Shenzhen</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Lin Chen, University of Science and Technology of China, China</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Yi-Wen Liu, National Tsing Hua University, Taiwan; Chenghui Jiang, Nanjing Medical University, China</p></fn>
<corresp id="c001">&#x002A;Correspondence: Zhen Zhang, <email>zhangzhen1994s@outlook.com</email></corresp>
<corresp id="c002">Di Qian, <email>skeayqd@sina.com</email></corresp>
<corresp id="c003">Yanmei Feng, <email>ymfeng@sjtu.edu.cn</email></corresp>
<fn fn-type="other" id="fn004"><p>This article was submitted to Auditory Cognitive Neuroscience, a section of the journal Frontiers in Neuroscience</p></fn>
</author-notes>
<pub-date pub-type="epub">
<day>02</day>
<month>12</month>
<year>2021</year>
</pub-date>
<pub-date pub-type="collection">
<year>2021</year>
</pub-date>
<volume>15</volume>
<elocation-id>744959</elocation-id>
<history>
<date date-type="received">
<day>22</day>
<month>07</month>
<year>2021</year>
</date>
<date date-type="accepted">
<day>15</day>
<month>11</month>
<year>2021</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2021 Zheng, Li, Feng, Guo, Li, Xiao, Liu, He, Zhang, Qian and Feng.</copyright-statement>
<copyright-year>2021</copyright-year>
<copyright-holder>Zheng, Li, Feng, Guo, Li, Xiao, Liu, He, Zhang, Qian and Feng</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p><bold>Objectives:</bold> Mandarin-speaking users of cochlear implants (CI) perform poorer than their English counterpart. This may be because present CI speech coding schemes are largely based on English. This study aims to evaluate the relative contributions of temporal envelope (E) cues to Mandarin phoneme (including vowel, and consonant) and lexical tone recognition to provide information for speech coding schemes specific to Mandarin.</p>
<p><bold>Design:</bold> Eleven normal hearing subjects were studied using acoustic temporal E cues that were extracted from 30 continuous frequency bands between 80 and 7,562 Hz using the Hilbert transform and divided into five frequency regions. Percent-correct recognition scores were obtained with acoustic E cues presented in three, four, and five frequency regions and their relative weights calculated using the least-square approach.</p>
<p><bold>Results:</bold> For stimuli with three, four, and five frequency regions, percent-correct scores for vowel recognition using E cues were 50.43&#x2013;84.82%, 76.27&#x2013;95.24%, and 96.58%, respectively; for consonant recognition 35.49&#x2013;63.77%, 67.75&#x2013;78.87%, and 87.87%; for lexical tone recognition 60.80&#x2013;97.15%, 73.16&#x2013;96.87%, and 96.73%. For frequency region 1 to frequency region 5, the mean weights in vowel recognition were 0.17, 0.31, 0.22, 0.18, and 0.12, respectively; in consonant recognition 0.10, 0.16, 0.18, 0.23, and 0.33; in lexical tone recognition 0.38, 0.18, 0.14, 0.16, and 0.14.</p>
<p><bold>Conclusion:</bold> Regions that contributed most for vowel recognition was Region 2 (502&#x2013;1,022 Hz) that contains first formant (<italic>F</italic>1) information; Region 5 (3,856&#x2013;7,562 Hz) contributed most to consonant recognition; Region 1 (80&#x2013;502 Hz) that contains fundamental frequency (F0) information contributed most to lexical tone recognition.</p>
</abstract>
<kwd-group>
<kwd>temporal envelope cues</kwd>
<kwd>frequency region</kwd>
<kwd>Mandarin</kwd>
<kwd>vowel</kwd>
<kwd>consonant</kwd>
<kwd>tone</kwd>
</kwd-group>
<counts>
<fig-count count="3"/>
<table-count count="1"/>
<equation-count count="0"/>
<ref-count count="61"/>
<page-count count="9"/>
<word-count count="7242"/>
</counts>
</article-meta>
</front>
<body>
<sec id="S1" sec-type="intro">
<title>Introduction</title>
<p>Hearing loss is a common sensory disorder and has become an important global health problem due to the increasing prevalence and its negative impact on quality of life. <xref ref-type="bibr" rid="B54">World Health Organization [WHO] (2020)</xref> estimates that 466 million people suffer from hearing loss, with sensorineural hearing loss (SNHL) being the most common. Cochlear implant (CI) is currently the only effective method for patients with severe-to-profound SNHL (<xref ref-type="bibr" rid="B58">Zeng, 2004</xref>). Plenty of past research has been conducted to figure out the best strategies for encoding speech. <xref ref-type="bibr" rid="B36">Roy et al. (2015)</xref> programmed with either fine structure processing or high-definition continuous interleaved sampling strategy for CI users, and found that fine structure processing strategy offers better musical sound quality discrimination for CI users with respect to fundamental frequency perception. <xref ref-type="bibr" rid="B44">Tabibi et al. (2020)</xref> implemented a bio-inspired coding strategy for better representation of spectral and temporal information with 11 CI users, and significantly better performance was observed for bio-inspired coding strategy compared to the advanced combination encoder strategy. Recently, there are many studies for tonal language pitch encoding. Temporal limits encoder, optimized pitch, and language strategy has recently been proposed that can provide a significant benefit to perception of speech intonation (<xref ref-type="bibr" rid="B30">Meng et al., 2016</xref>; <xref ref-type="bibr" rid="B46">Vandali et al., 2019</xref>). The mainstream CI speech processing strategies, such as advanced combination encoder (<xref ref-type="bibr" rid="B34">Psarros et al., 2002</xref>), SPEAK (<xref ref-type="bibr" rid="B41">Skinner et al., 2002</xref>), and n-of-m (<xref ref-type="bibr" rid="B61">Ziese et al., 2000</xref>; <xref ref-type="bibr" rid="B6">Buechner et al., 2009</xref>) are based on the continuous interleaved sampling strategy (<xref ref-type="bibr" rid="B53">Wilson et al., 1991</xref>; <xref ref-type="bibr" rid="B5">Bo&#x00EB;x et al., 1996</xref>). For the continuous interleaved sampling speech processing strategy, the electrode array is successively spaced with a single stimulus, that is, only one electrode is emitting the stimulus current at a time, and the interference and diffusion of the stimulus current between two electrodes are prevented by alternating stimulation (<xref ref-type="bibr" rid="B59">Zeng et al., 2008</xref>). Although contemporary CI has up to 22 intracochlear electrodes, the capacity of patients to use multiple channels typically asymptotes at around 8 channels or less (<xref ref-type="bibr" rid="B33">Pfingst et al., 2011</xref>; <xref ref-type="bibr" rid="B29">Macherey and Carlyon, 2014</xref>). Therefore, even as the most successful neural implants in the world, there is still much to be studied and improved in signal processing strategies.</p>
<p>Speech acoustic signals can be regarded as the temporal envelope (E) cues with slow change and the temporal fine structure (TFS) information with fast change based on the Hilbert transform (<xref ref-type="bibr" rid="B22">Kong and Zeng, 2006</xref>). The TFS information is the pre-dominant cues for lexical tone perception in NH listeners (<xref ref-type="bibr" rid="B56">Xu and Pfingst, 2003</xref>) but in hearing-impaired listeners and in noise environment, envelope cue plays an increasingly important role for lexical tone perception (<xref ref-type="bibr" rid="B48">Wang S. et al., 2011</xref>; <xref ref-type="bibr" rid="B35">Qi et al., 2017</xref>). The temporal E cues represents the amplitude of the waveform changing with time phase, which usually includes the duration information and amplitude E cues of the speech signal (<xref ref-type="bibr" rid="B22">Kong and Zeng, 2006</xref>). Perceptual research has shown that E cues are important for speech perception in quiet conditions (<xref ref-type="bibr" rid="B42">Smith et al., 2002</xref>; <xref ref-type="bibr" rid="B56">Xu and Pfingst, 2003</xref>). Different frequency regions of speech signals contain different information with varying functions, making it necessary to evaluate the relative importance of temporal information with different frequency regions in speech recognition. Past research methods on the role of temporal information in different frequency regions for speech recognition include removing a specific spectral information (<xref ref-type="bibr" rid="B39">Shannon et al., 2002</xref>), correlation analysis (<xref ref-type="bibr" rid="B1">Apoux and Bacon, 2004</xref>), lowpass- and highpass- filtration (<xref ref-type="bibr" rid="B4">Ardoint and Lorenzi, 2010</xref>), and band-pass filtration (<xref ref-type="bibr" rid="B3">Ardoint et al., 2011</xref>).</p>
<p>Previous research was mostly based on English, a non-tonal language. Mandarin, a tonal and most common spoken language in the world is significantly different from English. Mandarin includes 24 finals, 23 consonants, and 4 lexical tones. The 24 finals include 6 monophthongs, 9 diphthongs, and triphthongs, and 9 nasal finals. The 23 consonants always occur as &#x201C;initials.&#x201D; Phonemes that include vowels and consonants are important signals for the auditory system because they have great contribution to speech intelligibility across languages (<xref ref-type="bibr" rid="B21">Kewley-Port et al., 2007</xref>). The four lexical tones include Lexical tone 1- (high-level), Lexical tone 2/(rising), Lexical tone 3 v (falling-rising), and Lexical tone 4 \(falling). In Mandarin, the same words with different lexical tones can represent many different meanings (<xref ref-type="bibr" rid="B31">Nissen et al., 2005</xref>). Previous studies have shown that compared with normal-hearing listeners, Mandarin-speaking CI users have shown poor performance in lexical tone recognition (<xref ref-type="bibr" rid="B51">Wei et al., 2004</xref>; <xref ref-type="bibr" rid="B49">Wang W. et al., 2011</xref>). One of the most important reasons is that most Chinese CI wearers have imported devices where the language processing strategy is calibrated toward non-tonal languages. Our previous studies have shown that the acoustic temporal E cues in frequency regions 1 (80&#x2013;502 Hz) and 3 (1,022&#x2013;1,913 Hz) significantly contributed to Mandarin sentence recognition in quiet (<xref ref-type="bibr" rid="B16">Guo et al., 2017</xref>). Given that speech perception involves both bottom-up and top-down processes, sentence recognition are heavily influenced by phonemic, lexical tone, and context in top-down condition (<xref ref-type="bibr" rid="B17">Hickok and Poeppel, 2007</xref>). This was the basis for our investigation into the different contributions of frequency regions in Mandarin phonemic and lexical tone recognition.</p>
</sec>
<sec id="S2" sec-type="materials|methods">
<title>Materials and Methods</title>
<sec id="S2.SS1">
<title>Subjects</title>
<p>A group of 11 listeners (5 males, 6 females) from graduates of Shanghai Jiao Tong University were recruited in this study. Their ages ranged from 22 to 27 (average = 24.6) years with no reported history of ear disease or hearing difficulty. They were all native Mandarin Chinese speakers with normal audiometric thresholds (&#x003C;20 dB HL), bilaterally, at frequencies between 0.25 and 8 kHz. Pure-tone audiometric thresholds were recorded using a GSI-61 audiometer (Grason-Stadler, Madison, WI, United States) with standard audiometric procedures. All subjects had no preceding exposure to the speech materials. Before the experiment, all subjects had signed a consent form and were compensated hourly. All procedures performed in studies involving human participants were approved and in accordance with the Ethics Committee of the Sixth People&#x2019;s Hospital affiliated to Shanghai Jiao Tong University (ChiCTR-ROC-17013460) and with the 1964 Declaration of Helsinki and its later amendments.</p>
</sec>
<sec id="S2.SS2">
<title>Signal Processing</title>
<p>The speech test program named Angel Sound<sup><xref ref-type="fn" rid="footnote1">1</xref></sup> developed by Qian-Jie Fu at the House Ear Institute (Los Angeles, CA, United States) was used for Mandarin phoneme and lexical tone tests (<xref ref-type="bibr" rid="B55">Wu et al., 2007</xref>). All speech materials were sourced from the language database of University of Science and Technology of China, which includes the phonemes and lexical tones most frequently used in Mandarin. All the materials were recorded by one male and one female native Mandarin speaker. All speech stimuli were sampled at a 22-kHz sampling rate, without high-frequency pre-emphasis. The test ensures that only one phoneme is different. For example, to test for vowel, combinations of the same consonant and lexical tone that carry the most vowel options were selected. For lexical tone, given the maximum possibility is four, phoneme combinations fewer than four lexical tones were excluded (<xref ref-type="bibr" rid="B55">Wu et al., 2007</xref>). Lexical tone duration was normalized for lexical tone tokens to minimize bias on tonal perception (<xref ref-type="bibr" rid="B19">Jing et al., 2017</xref>). The speech materials were filtered into 30 contiguous frequency bands using zero-phase, third-order Butterworth filters (18 dB/oct slopes), ranging from 80 to 7,562 Hz (<xref ref-type="bibr" rid="B25">Li et al., 2016</xref>; <xref ref-type="bibr" rid="B16">Guo et al., 2017</xref>; <xref ref-type="bibr" rid="B60">Zheng et al., 2021</xref>). Each frequency band was an equivalent rectangular bandwidth for normal people, which simulates the frequency selection of normal auditory system (<xref ref-type="bibr" rid="B15">Glasberg and Moore, 1990</xref>). E information was extracted from each band using the Hilbert decomposition and low-pass filter at 64 Hz using third-order Butterworth filters. Then E was used to modulate the amplitude of a white noise. The envelope-modulated noise was bandpass-filtered using the same filter parameters as before. This study focuses on the parameters used in the present CI strategy in low frequency (&#x003C;500 Hz), medium low frequency (500&#x2013;1,000 Hz), medium frequency (1,000&#x2013;2,000 Hz), medium high frequency (2,000&#x2013;4,000 Hz), and high frequency (4,000&#x2013;8,000 Hz) bands. Given the cut-off frequency of each frequency band is close to 500, 1,000, 2,000, 4,000, and 8,000 Hz, the modulated noise bands were summed across frequency bands to produce the frequency regions of acoustic E cues as follows: Bands 1&#x2013;8, 9&#x2013;13, 14&#x2013;18, 19&#x2013;24, and 25&#x2013;30 were summed to form Frequency Regions 1&#x2013;5, respectively (<xref ref-type="table" rid="T1">Table 1</xref>). To prevent subjects from using the E cues of the adjacent boundary band (<xref ref-type="bibr" rid="B50">Warren et al., 2004</xref>; <xref ref-type="bibr" rid="B24">Li et al., 2015</xref>), the frequency region containing the E cues was combined with complementary frequency regions containing noise masker that was presented at a speech-to-noise ratio of +16 dB. The speech-to-noise ratio was determined prior to signal processing using a full range of speech and noise stimuli. Masking noise was low-pass and high-pass filtered so that the final long-term power spectrum did not overlap the processed speech signals as the previous study (<xref ref-type="bibr" rid="B3">Ardoint et al., 2011</xref>). To investigate the role of different frequency regions for Mandarin phoneme and lexical tone recognition, the E cues from three frequency regions (10 conditions including &#x201C;Region 123,&#x201D; &#x201C;Region 124,&#x201D; &#x201C;Region 125,&#x201D; &#x201C;Region 134,&#x201D; &#x201C;Region 135,&#x201D; &#x201C;Region 145,&#x201D; &#x201C;Region 234,&#x201D; &#x201C;Region 235,&#x201D; &#x201C;Region 245,&#x201D; and &#x201C;Region 345&#x201D;), four frequency regions (five conditions including &#x201C;Region 1234,&#x201D; &#x201C;Region 1345,&#x201D; &#x201C;Region 1245,&#x201D; &#x201C;Region 1235,&#x201D; and &#x201C;Region 2345&#x201D;) and five frequency regions (one condition, &#x201C;Region 12345&#x201D;) were presented to subjects. For example, the condition of &#x201C;Region 123&#x201D; meant the stimulus presented to the subject contained the E cues of frequency regions 1, 2, and 3 with noise of the remaining frequency regions 4 and 5. Similarly, in the test condition of &#x201C;Region 124,&#x201D; the stimulus sound contains the E cues of frequency regions 1, 2, and 4 while other frequency bands (band 3 and 5) are white noise. In the test of full band region &#x201C;Region 12345,&#x201D; the stimulus sound contains the E cues information of all frequency bands, and there is no other noise.</p>
<table-wrap position="float" id="T1">
<label>TABLE 1</label>
<caption><p>Cut-off frequency for extracting temporal envelope information in different frequency regions.</p></caption>
<table cellspacing="5" cellpadding="5" frame="hsides" rules="groups">
<thead>
<tr>
<td valign="top" align="left">Frequency regions</td>
<td valign="top" align="center">Bands</td>
<td valign="top" align="center">Lower frequency (Hz)</td>
<td valign="top" align="center">Upper frequency (Hz)</td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">1</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">80</td>
<td valign="top" align="center">115</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">2</td>
<td valign="top" align="center">115</td>
<td valign="top" align="center">154</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">3</td>
<td valign="top" align="center">154</td>
<td valign="top" align="center">198</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">4</td>
<td valign="top" align="center">198</td>
<td valign="top" align="center">246</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">5</td>
<td valign="top" align="center">246</td>
<td valign="top" align="center">300</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">6</td>
<td valign="top" align="center">300</td>
<td valign="top" align="center">360</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">7</td>
<td valign="top" align="center">360</td>
<td valign="top" align="center">427</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">8</td>
<td valign="top" align="center">427</td>
<td valign="top" align="center">502</td>
</tr>
<tr>
<td valign="top" align="left">2</td>
<td valign="top" align="center">9</td>
<td valign="top" align="center">502</td>
<td valign="top" align="center">585</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">10</td>
<td valign="top" align="center">585</td>
<td valign="top" align="center">677</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">11</td>
<td valign="top" align="center">677</td>
<td valign="top" align="center">780</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">12</td>
<td valign="top" align="center">780</td>
<td valign="top" align="center">894</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">13</td>
<td valign="top" align="center">894</td>
<td valign="top" align="center">1,022</td>
</tr>
<tr>
<td valign="top" align="left">3</td>
<td valign="top" align="center">14</td>
<td valign="top" align="center">1,022</td>
<td valign="top" align="center">1,164</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">15</td>
<td valign="top" align="center">1,164</td>
<td valign="top" align="center">1,322</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">16</td>
<td valign="top" align="center">1,322</td>
<td valign="top" align="center">1,499</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">17</td>
<td valign="top" align="center">1,499</td>
<td valign="top" align="center">1,695</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">18</td>
<td valign="top" align="center">1,695</td>
<td valign="top" align="center">1,913</td>
</tr>
<tr>
<td valign="top" align="left">4</td>
<td valign="top" align="center">19</td>
<td valign="top" align="center">1,913</td>
<td valign="top" align="center">2,157</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">20</td>
<td valign="top" align="center">2,157</td>
<td valign="top" align="center">2,428</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">21</td>
<td valign="top" align="center">2,428</td>
<td valign="top" align="center">2,729</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">22</td>
<td valign="top" align="center">2,729</td>
<td valign="top" align="center">3,066</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">23</td>
<td valign="top" align="center">3,066</td>
<td valign="top" align="center">3,440</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">24</td>
<td valign="top" align="center">3,440</td>
<td valign="top" align="center">3,856</td>
</tr>
<tr>
<td valign="top" align="left">5</td>
<td valign="top" align="center">25</td>
<td valign="top" align="center">3,856</td>
<td valign="top" align="center">4,321</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">26</td>
<td valign="top" align="center">4,321</td>
<td valign="top" align="center">4,837</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">27</td>
<td valign="top" align="center">4,837</td>
<td valign="top" align="center">5,413</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">28</td>
<td valign="top" align="center">5,413</td>
<td valign="top" align="center">6,054</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">29</td>
<td valign="top" align="center">6,054</td>
<td valign="top" align="center">6,767</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">30</td>
<td valign="top" align="center">6,767</td>
<td valign="top" align="center">7,562</td>
</tr>
</tbody>
</table></table-wrap>
</sec>
<sec id="S2.SS3">
<title>Test Procedure</title>
<p>None of the subjects had participated in the perception experiments testing acoustic temporal E cues before. The experiments were conducted in a double-walled, soundproof room. All test stimuli were delivered through Sennheiser HD205 II circumaural headphones. The stimuli were determined according to the most comfortable level of the subjects, generally around 65 dB SPL.</p>
<p>Before the formal test, &#x223C;30 min of practice were provided. The speech material was presented under &#x201C;full Region&#x201D; conditions initially, and then presented in the same way as the test condition stimulus. Feedback was given during the practice. To familiarize the subjects with the test material, they can repeatedly listen to a word indefinitely and move on after they feel they have reached a stable state.</p>
<p>In the formal test, we randomly selected test sounds from different conditions and allowed subjects to hear the same test sound multiple times. Subjects were required to focus on repeating the keywords as accurately as possible, and we encouraged them to guess the uncertain words. Our observation was that most participants listened to each word two or three times before moving on. Vowel and consonant recognition were measured using a 16-alternative identification paradigm. The response buttons were labeled using vowel syllables for the vowel recognition task, consonant context with common finals for the consonant recognition task. Lexical tone recognition was measured using a four-alternative identification paradigm, and &#x201C;Lexical tone 1,&#x201D; &#x201C;Lexical tone 2,&#x201D; &#x201C;Lexical tone 3,&#x201D; and &#x201C;Lexical tone 4&#x201D; for the Mandarin lexical tone recognition task. No feedback was given during the formal test. Each word was rated as correct or incorrect, and then the percentage of correct words was recorded under different conditions. Subjects can take a rest at any time to minimize fatigue during testing. The complete test time for each participant is &#x223C;1.5&#x2013;2 h.</p>
</sec>
<sec id="S2.SS4">
<title>Least-Squares Approach</title>
<p>To evaluate the weight of the five frequency regions in Mandarin phoneme and lexical tone recognition using acoustic temporal E cues, we calculated the weight of each frequency region using the least-squares approach previously used in other research (<xref ref-type="bibr" rid="B20">Kasturi et al., 2002</xref>). The strength of each frequency region was defined as a binary value of 0 or 1, depending on whether the frequency region was presented or not. Then, the weight of each frequency region was calculated by predicting the subject&#x2019;s response as a linear combination of the contribution of each frequency region. The initial weights of each subject&#x2019;s five frequency regions were converted to relative weights by summing them up, and the weights of each frequency region were expressed as the initial weight divided by the sum of all five frequency regions weights. Therefore, the weights of the five frequency regions add up to 1.0 (For more details, please see <xref ref-type="supplementary-material" rid="DS1">Supplementary Material</xref>).</p>
</sec>
<sec id="S2.SS5">
<title>Statistical Analysis</title>
<p>The Statistical Package for Social Sciences (SPSS) version 24.0 (IBM Corp., Armonk, NY, United States) was used for statistical analysis. The one-way analysis of variance (ANOVA) with repeated measures was used for the results from different test conditions for phoneme and lexical tone recognition. The <italic>post hoc</italic> analysis (Tukey&#x2019;s test) was used for pairwise comparison. The least-squares approach was used to calculate the relative weights of the five frequency regions. The independent samples <italic>t</italic>-test was used to compare the relative weights of five frequency regions in Mandarin phoneme and lexical tone recognition. The figures were generated by GraphPad Prism 8.0 (GraphPad Software, San Diego, CA, United States). Statistical significance was set at <italic>p</italic> &#x003C; 0.05.</p>
</sec>
</sec>
<sec id="S3" sec-type="results">
<title>Results</title>
<sec id="S3.SS1">
<title>Scores for Mandarin Phoneme and Lexical Tone Recognition Across Conditions Using Temporal E Cues</title>
<p>As shown in <xref ref-type="fig" rid="F1">Figure 1A</xref>, the vowel recognition scores ranged from 50.43 to 84.82% when the E cues were presented in three frequency regions. The Region 234 condition score was the highest, &#x223C;84.82%, while Region 135 was lowest, &#x223C;50.43%. A one-way repeated-measures ANOVA of different test conditions with three frequency regions showed significant differences in vowel recognition scores among different frequency regions combinations [<italic>F</italic><sub>(9,100)</sub> = 13.559, <italic>p</italic> &#x003C; 0.05]. The Tukey&#x2019;s test revealed that the score under the Region 135 and Region 145 conditions was significantly lower than the scores under all other conditions with three frequency regions (<italic>p</italic> &#x003C; 0.05). The consonant recognition scores ranged from 35.49 to 63.77% when the E cues were presented in three frequency regions (see in <xref ref-type="fig" rid="F1">Figure 1B</xref>). The Region 345 condition score was the highest, &#x223C;63.77%, while Region 123 was lowest, &#x223C;35.49%. A one-way repeated-measures ANOVA of different test conditions with three frequency regions showed significant differences in consonant recognition scores among different frequency region combinations [<italic>F</italic><sub>(9,100)</sub> = 11.622, <italic>p</italic> &#x003C; 0.05]. The Tukey&#x2019;s test revealed that the scores obtained from conditions combined with Frequency Region 5 would be higher than those obtained from conditions combined without Region 5 (Region 123, Region 124, Region 134, and Region 234) (<italic>p</italic> &#x003C; 0.05). The lexical tone recognition scores ranged from 60.80 to 97.15% when the E cues were presented in three frequency regions (see in <xref ref-type="fig" rid="F1">Figure 1C</xref>). The Region 124 condition score was the highest, &#x223C;97.15%, while Region 345 was lowest, &#x223C;60.80%. A one-way repeated-measures ANOVA of different test conditions with three frequency regions showed significant differences in lexical tone recognition scores among different frequency region combinations [<italic>F</italic><sub>(9,100)</sub> = 46.910, <italic>p</italic> &#x003C; 0.05]. The Tukey&#x2019;s test revealed that the scores obtained from conditions combined with Frequency Region 1 would be higher than those obtained from conditions combined without Region 1 (Region 234, Region 235, Region 245, and Region 345) (<italic>p</italic> &#x003C; 0.05).</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption><p>Averaged percent-correct scores for Mandarin phoneme and lexical tone recognition using acoustic temporal envelope with three frequency regions conditions. The error bars represent standard errors. <bold>(A)</bold> Averaged scores for Mandarin vowel recognition with envelope cues in three frequency regions conditions. <bold>(B)</bold> Averaged scores for Mandarin consonant recognition with envelope cues in three frequency regions conditions. <bold>(C)</bold> Averaged scores for Mandarin lexical tone recognition with envelope cues in three frequency regions conditions.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-15-744959-g001.tif"/>
</fig>
<p>As shown in <xref ref-type="fig" rid="F2">Figure 2A</xref>, the vowel recognition scores ranged from 76.27 to 96.58% when the E cues were presented in four frequency regions. The Region 1234 condition score was the highest, &#x223C;95.24%, while Region 1345 was the lowest, &#x223C;76.27%. When stimulus presented in full frequency regions, the score raised to 96.58%. A one-way repeated-measures ANOVA of different test conditions with four and five frequency regions showed significant differences in vowel recognition scores among different frequency region combinations [<italic>F</italic><sub>(5</sub>,<sub>60)</sub> = 27.674, <italic>p</italic> &#x003C; 0.05]. The Tukey&#x2019;s test revealed that the score under the Region 1345 condition was significantly lower than the score under all other conditions (<italic>p</italic> &#x003C; 0.05). The consonant recognition scores ranged from 67.75 to 78.87% when the E cues were presented in four frequency regions (see in <xref ref-type="fig" rid="F2">Figure 2B</xref>). The Region 2345 condition score was the highest, &#x223C;78.87%, while Region 1234 was the lowest, &#x223C;67.75%. When stimulus presented in full frequency regions, the score raised to 87.87%. A one-way repeated-measures ANOVA of different test conditions with four and five frequency regions showed significant differences in consonant recognition scores among different frequency region combinations [<italic>F</italic><sub>(5,60)</sub> = 6.462, <italic>p</italic> &#x003C; 0.05]. The Tukey&#x2019;s test revealed that the difference in the five conditions with four frequency regions was not significant (<italic>p</italic> = 0.063). However, the consonant recognition scores with full frequency regions were significantly higher than that in four frequency regions combinations (<italic>p</italic> &#x003C; 0.05). The lexical tone recognition scores ranged from 73.16 to 96.87% when the E cues were presented in four frequency regions (see in <xref ref-type="fig" rid="F2">Figure 2C</xref>). The Region 1234 condition score was the highest, &#x223C;96.87%, while Region 2345 was lowest, &#x223C;73.16%. When stimulus presented in full frequency regions, the score raised to 96.73%. A one-way repeated-measures ANOVA of different test conditions with four and five frequency regions showed significant differences in consonant recognition scores among different frequency region combinations [<italic>F</italic><sub>(5,60)</sub> = 30.802, <italic>p</italic> &#x003C; 0.05]. The Tukey&#x2019;s test revealed that the score under the Region 2345 condition was significantly lower than the score under all other conditions with four frequency regions (<italic>p</italic> &#x003C; 0.05).</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption><p>Averaged percent-correct scores for Mandarin phoneme and lexical tone recognition using acoustic temporal envelope with four and five frequency regions conditions. The error bars represent standard errors. <bold>(A)</bold> Averaged scores for Mandarin vowel recognition with envelope cues in four frequency regions conditions. <bold>(B)</bold> Averaged scores for Mandarin consonant recognition with envelope cues in four frequency regions conditions. <bold>(C)</bold> Averaged scores for Mandarin lexical tone recognition with envelope cues in four frequency regions conditions.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-15-744959-g002.tif"/>
</fig>
</sec>
<sec id="S3.SS2">
<title>Relative Weights of the Five Frequency Regions in Mandarin Phoneme and Lexical Tone Recognition</title>
<p>As shown in <xref ref-type="fig" rid="F3">Figure 3</xref>, the mean weights of frequency region 1&#x2013;5 for vowel recognition were 0.17, 0.31, 0.22, 0.18, and 0.12, respectively. The one-way ANOVA showed a significant main effect of region on weight for vowel recognition [<italic>F</italic><sub>(4,50)</sub> = 41.117, <italic>p</italic> &#x003C; 0.05]. The Tukey&#x2019;s test revealed that the relative weight of Region 2 was highest than all other regions while the relative weight of Region 5 was lowest than all other regions (<italic>p</italic> &#x003C; 0.05). The mean weights of frequency region 1&#x2013;5 for consonant recognition were 0.10, 0.16, 0.18, 0.23, and 0.33, respectively. The one-way ANOVA showed a significant main effect of region on weight for consonant recognition [<italic>F</italic><sub>(4,50)</sub> = 40.459, <italic>p</italic> &#x003C; 0.05]. The Tukey&#x2019;s test revealed that the relative weight of Region 5 was highest than all other regions while the relative weight of Region 1 was lowest than all other regions (<italic>p</italic> &#x003C; 0.05). The mean weights of frequency region 1&#x2013;5 for lexical tone recognition were 0.38, 0.18, 0.14, 0.16, and 0.14, respectively. The one-way ANOVA showed a significant main effect of region on weight for lexical tone recognition [<italic>F</italic><sub>(4,50)</sub> = 176.725, <italic>p</italic> &#x003C; 0.05]. The Tukey&#x2019;s test revealed that the relative weight of Region 1 was highest than all other regions (<italic>p</italic> &#x003C; 0.05).</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption><p>The relative weights of different frequency regions for Mandarin phoneme and lexical tone recognition using acoustic temporal envelope. The error bars represent standard errors.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-15-744959-g003.tif"/>
</fig>
</sec>
</sec>
<sec id="S4" sec-type="discussion">
<title>Discussion</title>
<p>This study was designed to explore the relative importance of acoustic E cues across different frequency regions for Mandarin phoneme and lexical tone recognition. Then we calculated the weight of each frequency region in Mandarin phoneme and lexical tone recognition by the least-squares approach as shown in <xref ref-type="fig" rid="F3">Figure 3</xref>. Region 2 (502&#x2013;1,022 Hz), Region 5 (3,856&#x2013;7,562 Hz), and Region 1 (80&#x2013;502 Hz) significantly contributed to Mandarin vowel, consonant, and lexical tone recognition, respectively.</p>
<p>Previous reports suggested that Mandarin phonemes and sentence recognition improved dramatically when the number of frequency regions increased from one to four (<xref ref-type="bibr" rid="B14">Fu et al., 1998</xref>), which was in line with results found in English speech recognition (<xref ref-type="bibr" rid="B40">Shannon et al., 1995</xref>). Our results affirm these findings, and that when presented in full region, the temporal E cues are sufficient to code for the recognition of Mandarin phoneme and lexical tone.</p>
<p>Vowels are important to the power of speech that is characterized by open vocal tract with sustained vocalization, low-frequency energy, and long duration (<xref ref-type="bibr" rid="B10">Chen et al., 2013</xref>; <xref ref-type="bibr" rid="B8">Chen and Chan, 2016</xref>). Sounds typically have four or five formants when it passes through the vocal tract. The first two formants determine the quality of vowels, while the last three formants determine the individual&#x2019;s unique timbre and influence the individual&#x2019;s vocal characteristics. When vowels are pronounced, the height of the tongue position corresponds to the first formant (F1), and the front and back of the tongue position correspond to the second formant (F2) (<xref ref-type="bibr" rid="B26">Lopes et al., 2018</xref>). Formant frequencies F1 and F2 have long been known to be crucial for encoding the phonetic identity of vowels (<xref ref-type="bibr" rid="B7">Carney et al., 2015</xref>; <xref ref-type="bibr" rid="B13">Fogerty, 2015</xref>). <xref ref-type="bibr" rid="B18">Hillenbrand et al. (1995)</xref> analyzed the data obtained from 45 men, 48 women, and 46 children and revealed that F1 (342&#x2013;1,022 Hz) and F2 (910&#x2013;3,081 Hz) were sufficient for vowel classification. It seems that the locations of the formants are dispersed optimally in the F1&#x2013;F2 space, as described in dispersion theory (<xref ref-type="bibr" rid="B38">Schwartz et al., 1997</xref>). With the increase of the size of vowel systems, this dispersion leads to the consistencies among linguistic vowel systems in the appearance of vowel contrasts. While a study (<xref ref-type="bibr" rid="B32">Parikh and Loizou, 2005</xref>) reported that vowel recognition in noise is supported mainly by information about F1 along with some information about F2, another study (<xref ref-type="bibr" rid="B57">Xu et al., 2005</xref>) analyzed the information transmitted for acoustic features of vowels and found that the duration and F1 cues rather than F2 cues contributed substantially to vowel recognition. This is closer to a report (<xref ref-type="bibr" rid="B45">Traunm&#x00FC;ller, 1981</xref>) that suggested the simultaneous distance between F1 and the fundamental frequency (F0) is the primary determinant of perceived vowel height. In our research, we found that the scores under the conditions without Region 2 (Region 135, Region 145, and Region 1345) were significantly lower than the score in other conditions (seen in <xref ref-type="fig" rid="F1">Figures 1A</xref>, <xref ref-type="fig" rid="F2">2A</xref>). The mean weights of frequency region 1&#x2013;5 for vowel recognition were 0.17, 0.31, 0.22, 0.18, and 0.12, respectively. The relative weight of Region 2 (502&#x2013;1,022 Hz) was highest across all other regions (seen in <xref ref-type="fig" rid="F3">Figure 3</xref>). This is consistent with previously reported study (<xref ref-type="bibr" rid="B20">Kasturi et al., 2002</xref>) that channels 1, 3, and 4, centered at 393, 1,037, and 1,685 Hz, respectively, received the largest weight for vowels recognition and will lead to decrease in listener&#x2019;s performance if removed.</p>
<p>Different from vowels, consonants are characterized by complete or partial vocal tract constriction with high-frequency energy and short duration that are important to speech intelligibility (<xref ref-type="bibr" rid="B10">Chen et al., 2013</xref>; <xref ref-type="bibr" rid="B8">Chen and Chan, 2016</xref>). For consonants, many of these phonemes are characterized by rapid, instantaneous changes in amplitude, for instance, those caused by burst noise (<xref ref-type="bibr" rid="B43">Stevens, 2002</xref>). Therefore, high frequencies phoneme level modulation may be particularly important for conveying the consonant cues necessary for intelligibility. These high-frequency bands are characterized by having fast rate E modulations. A previous research (<xref ref-type="bibr" rid="B12">Fogerty, 2014</xref>) replaced consonant and vowel segments with noise matched to speech spectrum and found that consonants contain higher frequency components compared to vowels. We found that high-frequency region (3,856&#x2013;7,562 Hz) of E cues plays a crucial role in consonant recognition (seen in <xref ref-type="fig" rid="F3">Figure 3</xref>), and the scores obtained from conditions combined with Region 5 would be higher than those obtained from conditions combined without Region 5 (seen in <xref ref-type="fig" rid="F1">Figure 1B</xref>). However, our result of relative weight for consonant recognition is different from previous findings (<xref ref-type="bibr" rid="B20">Kasturi et al., 2002</xref>). Here, the relative weight for the consonants was quite flat and all channels (the frequencies ranged from 300 to 4,444 Hz) are equally important for consonant recognition. In contrast, a study (<xref ref-type="bibr" rid="B2">Apoux and Bacon, 2008</xref>) reported that consonant recognition was not affected by removing E cues above 4 Hz in the low- and high-frequency bands, while the consonant recognition decreased as the cutoff frequency was decreased in the mid-frequency region from 16 to 4 Hz. Possible reasons for the different weight for consonant recognition include: (1) differences in speech materials, for the type of speech material may have a strong impact on the value of acoustic information (<xref ref-type="bibr" rid="B27">Lunner et al., 2012</xref>); (2) the different methods of processing stimuli and different cut-off frequencies used in different experiments; (3) finally and most importantly the difference may come from differences in languages. The plosive bursts in Mandarin consonants produce a strongly synchronized burst of energy across the frequency spectrum related to high frequency regions (<xref ref-type="bibr" rid="B11">Drullman et al., 1994</xref>).</p>
<p>Lexical tone is also called pitch or the height of the sound. Most ordinary sounds can be analyzed as a sum of sinusoidal components with harmonic frequencies and evoke a pitch corresponding to their F0 (<xref ref-type="bibr" rid="B37">Santurette and Dau, 2011</xref>). As previous studies reported, F0 cues are important for Chinese lexical tone recognition (<xref ref-type="bibr" rid="B28">Luo and Fu, 2004</xref>; <xref ref-type="bibr" rid="B9">Chen et al., 2014</xref>; <xref ref-type="bibr" rid="B47">Vandali et al., 2015</xref>). There are four lexical tones in Mandarin including Lexical tone 1- (high-level), Lexical tone 2/(rising), Lexical tone 3 v (falling-rising), and Lexical tone 4 \(falling). As a tonal language, the same phonetic segment carries a different meaning when produced with different lexical tones (<xref ref-type="bibr" rid="B10">Chen et al., 2013</xref>). Lexical tones of Mandarin have been studied by many researchers. As previous research reported, Lexical tone 1 is associated with a flat F0 contour and Lexical tone 2 with a rising F0 contour (<xref ref-type="bibr" rid="B52">Whalen and Xu, 1992</xref>), Lexical tone 3 has the lowest intensity and longest duration while Lexical tone 4 is usually the strongest and shortest lasting pitch (<xref ref-type="bibr" rid="B23">Kuo et al., 2008</xref>). Although these four lexical tones are mainly distinguished by the F0 cues, other characteristics including overall intensity and duration vary systematically with lexical tone (<xref ref-type="bibr" rid="B23">Kuo et al., 2008</xref>). A study (<xref ref-type="bibr" rid="B9">Chen et al., 2014</xref>) examined the effects of lexical tone on the intelligibility of Mandarin revealed that the F0 contour is particularly important in tonal recognition in noisy environments. Another study (<xref ref-type="bibr" rid="B47">Vandali et al., 2015</xref>) reported that training with a single cue (F0 and center frequency) can improve the recognition ability of pitch and timbre without other cues variations. This is similar to another finding (<xref ref-type="bibr" rid="B28">Luo and Fu, 2004</xref>) that found modifying the amplitude E to make it closer to the F0 contour may be an effective method to improve the lexical tone recognition of Chinese CI wearers. Our findings are consistent with these previous studies, where Region 1 (80&#x2013;502 Hz) significantly contributes to Mandarin lexical tone recognition (seen in <xref ref-type="fig" rid="F3">Figure 3</xref>).</p>
<p>There are some limitations in our study. First, the age of the participants ranged from 21 to 27, so the result has limited explanation for other age groups including the infant and the elderly. Secondly, the subjects in this study received higher education. Larger-scale studies are needed to verify the results, and for further research. Thirdly, the preliminary results our study currently obtained will be used to guide more in-depth research, however, how signal processing or stimulation strategies for future CI systems will be influenced by current results is unclear. Perhaps CI users, in addition to normal hearing listeners, could further be recruited to increase the impact of this research.</p>
</sec>
<sec id="S5" sec-type="conclusion">
<title>Conclusion</title>
<list list-type="simple">
<list-item>
<label>(1)</label>
<p>For Mandarin vowel recognition, Region 2 (502&#x2013;1,022 Hz) which contained the first formant (<italic>F</italic>1) information contributed more than other regions.</p>
</list-item>
<list-item>
<label>(2)</label>
<p>For Mandarin consonant recognition, Region 5 (3,856&#x2013;7,562 Hz) contributed more than other regions.</p>
</list-item>
<list-item>
<label>(3)</label>
<p>For Mandarin lexical tone recognition, Region 1 (80&#x2013;502 Hz) which contained fundamental frequency (F0) information contributed more than other regions.</p>
</list-item>
</list>
</sec>
<sec id="S6" sec-type="data-availability">
<title>Data Availability Statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="supplementary-material" rid="DS1">Supplementary Material</xref>, further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec id="S7">
<title>Ethics Statement</title>
<p>The studies involving human participants were reviewed and approved by the Shanghai Jiaotong University Affiliated Sixth People&#x2019;s Hospital Ethics Committee. The patients/participants provided their written informed consent to participate in this study.</p>
</sec>
<sec id="S8">
<title>Author Contributions</title>
<p>ZZhe, KL, ZZha, DQ, and YF: conceptualization. KL, YG, and SH: methodology. KL, GF, and YG: data curation. YL, LX, CL, and SH: investigation. ZZhe: writing&#x2014;original draft preparation. ZZha, DQ, and YF: writing&#x2014;review and editing. YF: funding acquisition. ZZhe and YF: resources. ZZha: supervision. All authors contributed to the article and approved the submitted version.</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="pudiscl1" sec-type="disclaimer">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
</body>
<back>
<sec id="S9" sec-type="funding-information">
<title>Funding</title>
<p>This research was funded by the National Natural Science Foundation of China (No. 81771015) and the Shanghai Municipal Commission of Science and Technology (Grant No. 18DZ2260200) and the International Cooperation and Exchange of the National Natural Science Foundation of China (Grant No. 81720108010).</p>
</sec>
<ack>
<p>We would like to thank Qian-Jie Fu for software support, and we also would like to thank the participants of the study.</p>
</ack>
<sec id="S11" sec-type="supplementary-material"><title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fnins.2021.744959/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fnins.2021.744959/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Data_Sheet_1.XLSX" id="DS1" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Data_Sheet_2.docx" id="DS2" mimetype="application/application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Apoux</surname> <given-names>F.</given-names></name> <name><surname>Bacon</surname> <given-names>S. P.</given-names></name></person-group> (<year>2004</year>). <article-title>Relative importance of temporal information in various frequency regions for consonant identification in quiet and in noise.</article-title> <source><italic>J. Acoust. Soc. Am.</italic></source> <volume>116</volume> <fpage>1671</fpage>&#x2013;<lpage>1680</lpage>. <pub-id pub-id-type="doi">10.1121/1.1781329</pub-id></citation></ref>
<ref id="B2"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Apoux</surname> <given-names>F.</given-names></name> <name><surname>Bacon</surname> <given-names>S. P.</given-names></name></person-group> (<year>2008</year>). <article-title>Differential contribution of envelope fluctuations across frequency to consonant identification in quiet.</article-title> <source><italic>J. Acoust. Soc. Am.</italic></source> <volume>123</volume> <issue>2792</issue>. <pub-id pub-id-type="doi">10.1121/1.2897916</pub-id></citation></ref>
<ref id="B3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ardoint</surname> <given-names>M.</given-names></name> <name><surname>Agus</surname> <given-names>T.</given-names></name> <name><surname>Sheft</surname> <given-names>S.</given-names></name> <name><surname>Lorenzi</surname> <given-names>C.</given-names></name></person-group> (<year>2011</year>). <article-title>Importance of temporal-envelope speech cues in different spectral regions.</article-title> <source><italic>J. Acoust. Soc. Am.</italic></source> <volume>130</volume> <fpage>El115</fpage>&#x2013;<lpage>El121</lpage>. <pub-id pub-id-type="doi">10.1121/1.3602462</pub-id></citation></ref>
<ref id="B4"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ardoint</surname> <given-names>M.</given-names></name> <name><surname>Lorenzi</surname> <given-names>C.</given-names></name></person-group> (<year>2010</year>). <article-title>Effects of lowpass and highpass filtering on the intelligibility of speech based on temporal fine structure or envelope cues.</article-title> <source><italic>Hear. Res.</italic></source> <volume>260</volume> <fpage>89</fpage>&#x2013;<lpage>95</lpage>. <pub-id pub-id-type="doi">10.1016/j.heares.2009.12.002</pub-id> <pub-id pub-id-type="pmid">19963053</pub-id></citation></ref>
<ref id="B5"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bo&#x00EB;x</surname> <given-names>C.</given-names></name> <name><surname>Pelizzone</surname> <given-names>M.</given-names></name> <name><surname>Montandon</surname> <given-names>P.</given-names></name></person-group> (<year>1996</year>). <article-title>Speech recognition with a CIS strategy for the ineraid multichannel cochlear implant.</article-title> <source><italic>Am. J. Otol.</italic></source> <volume>17</volume> <fpage>61</fpage>&#x2013;<lpage>68</lpage>.</citation></ref>
<ref id="B6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Buechner</surname> <given-names>A.</given-names></name> <name><surname>Frohne-Buechner</surname> <given-names>C.</given-names></name> <name><surname>Boyle</surname> <given-names>P.</given-names></name> <name><surname>Battmer</surname> <given-names>R. D.</given-names></name> <name><surname>Lenarz</surname> <given-names>T.</given-names></name></person-group> (<year>2009</year>). <article-title>A high rate n-of-m speech processing strategy for the first generation Clarion cochlear implant.</article-title> <source><italic>Intern. J. Audiol.</italic></source> <volume>48</volume> <fpage>868</fpage>&#x2013;<lpage>875</lpage>. <pub-id pub-id-type="doi">10.3109/14992020903095783</pub-id> <pub-id pub-id-type="pmid">20017683</pub-id></citation></ref>
<ref id="B7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Carney</surname> <given-names>L. H.</given-names></name> <name><surname>Li</surname> <given-names>T.</given-names></name> <name><surname>McDonough</surname> <given-names>J. M.</given-names></name></person-group> (<year>2015</year>). <article-title>Speech coding in the brain: representation of vowel formants by midbrain meurons tuned to sound fluctuations.</article-title> <source><italic>eNeuro</italic></source> <volume>2</volume>:<issue>ENEURO.0004-15.2015</issue>. <pub-id pub-id-type="doi">10.1523/eneuro.0004-15.2015</pub-id> <pub-id pub-id-type="pmid">26464993</pub-id></citation></ref>
<ref id="B8"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>F.</given-names></name> <name><surname>Chan</surname> <given-names>F. W.</given-names></name></person-group> (<year>2016</year>). <article-title>Understanding frequency-compressed Mandarin sentences: role of vowels.</article-title> <source><italic>J. Acoust. Soc. Am.</italic></source> <volume>139</volume> <fpage>1204</fpage>&#x2013;<lpage>1213</lpage>. <pub-id pub-id-type="doi">10.1121/1.4944037</pub-id></citation></ref>
<ref id="B9"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>F.</given-names></name> <name><surname>Wong</surname> <given-names>L. L.</given-names></name> <name><surname>Hu</surname> <given-names>Y.</given-names></name></person-group> (<year>2014</year>). <article-title>Effects of lexical tone contour on Mandarin sentence intelligibility.</article-title> <source><italic>J. Speech Lang. Hear. Res.</italic></source> <volume>57</volume> <fpage>338</fpage>&#x2013;<lpage>345</lpage>. <pub-id pub-id-type="doi">10.1044/1092-4388(2013/12-0324)</pub-id></citation></ref>
<ref id="B10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>F.</given-names></name> <name><surname>Wong</surname> <given-names>L. L.</given-names></name> <name><surname>Wong</surname> <given-names>E. Y.</given-names></name></person-group> (<year>2013</year>). <article-title>Assessing the perceptual contributions of vowels and consonants to Mandarin sentence intelligibility.</article-title> <source><italic>J. Acoust. Soc. Am.</italic></source> <volume>134</volume> <fpage>El178</fpage>&#x2013;<lpage>El184</lpage>. <pub-id pub-id-type="doi">10.1121/1.4812820</pub-id></citation></ref>
<ref id="B11"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Drullman</surname> <given-names>R.</given-names></name> <name><surname>Festen</surname> <given-names>J. M.</given-names></name> <name><surname>Plomp</surname> <given-names>R.</given-names></name></person-group> (<year>1994</year>). <article-title>Effect of temporal envelope smearing on speech reception.</article-title> <source><italic>J. Acoust. Soc. Am.</italic></source> <volume>95</volume> <fpage>1053</fpage>&#x2013;<lpage>1064</lpage>. <pub-id pub-id-type="doi">10.1121/1.408467</pub-id></citation></ref>
<ref id="B12"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fogerty</surname> <given-names>D.</given-names></name></person-group> (<year>2014</year>). <article-title>Importance of envelope modulations during consonants and vowels in segmentally interrupted sentences.</article-title> <source><italic>J. Acoust. Soc. Am.</italic></source> <volume>135</volume> <fpage>1568</fpage>&#x2013;<lpage>1576</lpage>. <pub-id pub-id-type="doi">10.1121/1.4863652</pub-id></citation></ref>
<ref id="B13"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fogerty</surname> <given-names>D.</given-names></name></person-group> (<year>2015</year>). <article-title>Indexical properties influence time-varying amplitude and fundamental frequency contributions of vowels to sentence intelligibility.</article-title> <source><italic>J. Phonet.</italic></source> <volume>52</volume> <fpage>89</fpage>&#x2013;<lpage>104</lpage>. <pub-id pub-id-type="doi">10.1016/j.wocn.2015.06.005</pub-id> <pub-id pub-id-type="pmid">26543276</pub-id></citation></ref>
<ref id="B14"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fu</surname> <given-names>Q. J.</given-names></name> <name><surname>Zeng</surname> <given-names>F. G.</given-names></name> <name><surname>Shannon</surname> <given-names>R. V.</given-names></name> <name><surname>Soli</surname> <given-names>S. D.</given-names></name></person-group> (<year>1998</year>). <article-title>Importance of tonal envelope cues in Chinese speech recognition.</article-title> <source><italic>J. Acoust. Soc. Am.</italic></source> <volume>104</volume> <fpage>505</fpage>&#x2013;<lpage>510</lpage>. <pub-id pub-id-type="doi">10.1121/1.423251</pub-id></citation></ref>
<ref id="B15"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Glasberg</surname> <given-names>B. R.</given-names></name> <name><surname>Moore</surname> <given-names>B. C.</given-names></name></person-group> (<year>1990</year>). <article-title>Derivation of auditory filter shapes from notched-noise data.</article-title> <source><italic>Hear. Res.</italic></source> <volume>47</volume> <fpage>103</fpage>&#x2013;<lpage>138</lpage>. <pub-id pub-id-type="doi">10.1016/0378-5955(90)90170-t</pub-id></citation></ref>
<ref id="B16"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Guo</surname> <given-names>Y.</given-names></name> <name><surname>Sun</surname> <given-names>Y.</given-names></name> <name><surname>Feng</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Yin</surname> <given-names>S.</given-names></name></person-group> (<year>2017</year>). <article-title>The relative weight of temporal envelope cues in different frequency regions for Mandarin sentence recognition.</article-title> <source><italic>Neural Plast.</italic></source> <volume>2017</volume>:<issue>7416727</issue>. <pub-id pub-id-type="doi">10.1155/2017/7416727</pub-id> <pub-id pub-id-type="pmid">28203463</pub-id></citation></ref>
<ref id="B17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hickok</surname> <given-names>G.</given-names></name> <name><surname>Poeppel</surname> <given-names>D.</given-names></name></person-group> (<year>2007</year>). <article-title>The cortical organization of speech processing.</article-title> <source><italic>Nat. Rev. Neurosci.</italic></source> <volume>8</volume> <fpage>393</fpage>&#x2013;<lpage>402</lpage>. <pub-id pub-id-type="doi">10.1038/nrn2113</pub-id> <pub-id pub-id-type="pmid">17431404</pub-id></citation></ref>
<ref id="B18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hillenbrand</surname> <given-names>J.</given-names></name> <name><surname>Getty</surname> <given-names>L. A.</given-names></name> <name><surname>Clark</surname> <given-names>M. J.</given-names></name> <name><surname>Wheeler</surname> <given-names>K.</given-names></name></person-group> (<year>1995</year>). <article-title>Acoustic characteristics of American English vowels.</article-title> <source><italic>J. Acoust. Soc. Am.</italic></source> <volume>97</volume>(<issue>5 Pt 1</issue>), <fpage>3099</fpage>&#x2013;<lpage>3111</lpage>. <pub-id pub-id-type="doi">10.1121/1.411872</pub-id></citation></ref>
<ref id="B19"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jing</surname> <given-names>Y.</given-names></name> <name><surname>Yu</surname> <given-names>Z.</given-names></name> <name><surname>Li</surname> <given-names>A.</given-names></name> <name><surname>Li</surname> <given-names>X.</given-names></name></person-group> (<year>2017</year>). &#x201C;<article-title>On the duration of mandarin tones</article-title>,&#x201D; in <source><italic>Proceedings of the Interspeech 2017</italic></source>, <publisher-loc>Stockholm</publisher-loc>.</citation></ref>
<ref id="B20"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kasturi</surname> <given-names>K.</given-names></name> <name><surname>Loizou</surname> <given-names>P. C.</given-names></name> <name><surname>Dorman</surname> <given-names>M.</given-names></name> <name><surname>Spahr</surname> <given-names>T.</given-names></name></person-group> (<year>2002</year>). <article-title>The intelligibility of speech with &#x201C;holes&#x201D; in the spectrum.</article-title> <source><italic>J. Acoust. Soc. Am.</italic></source> <volume>112</volume>(<issue>3 Pt 1</issue>), <fpage>1102</fpage>&#x2013;<lpage>1111</lpage>. <pub-id pub-id-type="doi">10.1121/1.1498855</pub-id></citation></ref>
<ref id="B21"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kewley-Port</surname> <given-names>D.</given-names></name> <name><surname>Burkle</surname> <given-names>T. Z.</given-names></name> <name><surname>Lee</surname> <given-names>J. H.</given-names></name></person-group> (<year>2007</year>). <article-title>Contribution of consonant versus vowel information to sentence intelligibility for young normal-hearing and elderly hearing-impaired listeners.</article-title> <source><italic>J. Acoust. Soc. Am.</italic></source> <volume>122</volume> <fpage>2365</fpage>&#x2013;<lpage>2375</lpage>. <pub-id pub-id-type="doi">10.1121/1.2773986</pub-id></citation></ref>
<ref id="B22"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kong</surname> <given-names>Y. Y.</given-names></name> <name><surname>Zeng</surname> <given-names>F. G.</given-names></name></person-group> (<year>2006</year>). <article-title>Temporal and spectral cues in Mandarin tone recognition.</article-title> <source><italic>J. Acoust. Soc. Am.</italic></source> <volume>120</volume>(<issue>5 Pt 1</issue>), <fpage>2830</fpage>&#x2013;<lpage>2840</lpage>. <pub-id pub-id-type="doi">10.1121/1.2346009</pub-id></citation></ref>
<ref id="B23"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kuo</surname> <given-names>Y. C.</given-names></name> <name><surname>Rosen</surname> <given-names>S.</given-names></name> <name><surname>Faulkner</surname> <given-names>A.</given-names></name></person-group> (<year>2008</year>). <article-title>Acoustic cues to tonal contrasts in Mandarin: implications for cochlear implants.</article-title> <source><italic>J. Acoust. Soc. Am.</italic></source> <volume>123</volume>:<issue>2815</issue>. <pub-id pub-id-type="doi">10.1121/1.2896755</pub-id></citation></ref>
<ref id="B24"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>B.</given-names></name> <name><surname>Hou</surname> <given-names>L.</given-names></name> <name><surname>Xu</surname> <given-names>L.</given-names></name> <name><surname>Wang</surname> <given-names>H.</given-names></name> <name><surname>Yang</surname> <given-names>G.</given-names></name> <name><surname>Yin</surname> <given-names>S.</given-names></name><etal/></person-group> (<year>2015</year>). <article-title>Effects of steep high-frequency hearing loss on speech recognition using temporal fine structure in low-frequency region.</article-title> <source><italic>Hear. Res.</italic></source> <volume>326</volume> <fpage>66</fpage>&#x2013;<lpage>74</lpage>. <pub-id pub-id-type="doi">10.1016/j.heares.2015.04.004</pub-id> <pub-id pub-id-type="pmid">25916265</pub-id></citation></ref>
<ref id="B25"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>B.</given-names></name> <name><surname>Wang</surname> <given-names>H.</given-names></name> <name><surname>Yang</surname> <given-names>G.</given-names></name> <name><surname>Hou</surname> <given-names>L.</given-names></name> <name><surname>Su</surname> <given-names>K.</given-names></name> <name><surname>Feng</surname> <given-names>Y.</given-names></name><etal/></person-group> (<year>2016</year>). <article-title>The importance of acoustic temporal fine structure cues in different spectral regions for Mandarin sentence recognition.</article-title> <source><italic>Ear Hear.</italic></source> <volume>37</volume> <fpage>e52</fpage>&#x2013;<lpage>e56</lpage>. <pub-id pub-id-type="doi">10.1097/aud.0000000000000216</pub-id> <pub-id pub-id-type="pmid">26317161</pub-id></citation></ref>
<ref id="B26"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lopes</surname> <given-names>L. W.</given-names></name> <name><surname>Alves</surname> <given-names>J. D. N.</given-names></name> <name><surname>Evangelista</surname> <given-names>D. D. S.</given-names></name> <name><surname>Fran&#x00E7;a</surname> <given-names>F. P.</given-names></name> <name><surname>Vieira</surname> <given-names>V. J. D.</given-names></name> <name><surname>Lima-Silva</surname> <given-names>M. F. B.</given-names></name><etal/></person-group> (<year>2018</year>). <article-title>Accuracy of traditional and formant acoustic measurements in the evaluation of vocal quality.</article-title> <source><italic>Codas</italic></source> <volume>30</volume>:<issue>e20170282</issue>. <pub-id pub-id-type="doi">10.1590/2317-1782/20182017282</pub-id> <pub-id pub-id-type="pmid">30365651</pub-id></citation></ref>
<ref id="B27"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lunner</surname> <given-names>T.</given-names></name> <name><surname>Hietkamp</surname> <given-names>R. K.</given-names></name> <name><surname>Andersen</surname> <given-names>M. R.</given-names></name> <name><surname>Hopkins</surname> <given-names>K.</given-names></name> <name><surname>Moore</surname> <given-names>B. C.</given-names></name></person-group> (<year>2012</year>). <article-title>Effect of speech material on the benefit of temporal fine structure information in speech for young normal-hearing and older hearing-impaired participants.</article-title> <source><italic>Ear Hear.</italic></source> <volume>33</volume> <fpage>377</fpage>&#x2013;<lpage>388</lpage>. <pub-id pub-id-type="doi">10.1097/AUD.0b013e3182387a8c</pub-id> <pub-id pub-id-type="pmid">22246137</pub-id></citation></ref>
<ref id="B28"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Luo</surname> <given-names>X.</given-names></name> <name><surname>Fu</surname> <given-names>Q. J.</given-names></name></person-group> (<year>2004</year>). <article-title>Enhancing Chinese tone recognition by manipulating amplitude envelope: implications for cochlear implants.</article-title> <source><italic>J. Acoust. Soc. Am.</italic></source> <volume>116</volume> <fpage>3659</fpage>&#x2013;<lpage>3667</lpage>. <pub-id pub-id-type="doi">10.1121/1.1783352</pub-id></citation></ref>
<ref id="B29"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Macherey</surname> <given-names>O.</given-names></name> <name><surname>Carlyon</surname> <given-names>R. P.</given-names></name></person-group> (<year>2014</year>). <article-title>Cochlear implants.</article-title> <source><italic>Curr. Biol.</italic></source> <volume>24</volume> <fpage>R878</fpage>&#x2013;<lpage>R884</lpage>. <pub-id pub-id-type="doi">10.1016/j.cub.2014.06.053</pub-id> <pub-id pub-id-type="pmid">25247367</pub-id></citation></ref>
<ref id="B30"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Meng</surname> <given-names>Q.</given-names></name> <name><surname>Zheng</surname> <given-names>N.</given-names></name> <name><surname>Li</surname> <given-names>X.</given-names></name></person-group> (<year>2016</year>). <article-title>Mandarin speech-in-noise and tone recognition using vocoder simulations of the temporal limits encoder for cochlear implants.</article-title> <source><italic>J. Acoust. Soc. Am.</italic></source> <volume>139</volume> <fpage>301</fpage>&#x2013;<lpage>310</lpage>. <pub-id pub-id-type="doi">10.1121/1.4939707</pub-id></citation></ref>
<ref id="B31"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nissen</surname> <given-names>S. L.</given-names></name> <name><surname>Harris</surname> <given-names>R. W.</given-names></name> <name><surname>Jennings</surname> <given-names>L. J.</given-names></name> <name><surname>Eggett</surname> <given-names>D. L.</given-names></name> <name><surname>Buck</surname> <given-names>H.</given-names></name></person-group> (<year>2005</year>). <article-title>Psychometrically equivalent Mandarin bisyllabic speech discrimination materials spoken by male and female talkers.</article-title> <source><italic>Intern. J. Audiol.</italic></source> <volume>44</volume> <fpage>379</fpage>&#x2013;<lpage>390</lpage>. <pub-id pub-id-type="doi">10.1080/14992020500147615</pub-id> <pub-id pub-id-type="pmid">16136788</pub-id></citation></ref>
<ref id="B32"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Parikh</surname> <given-names>G.</given-names></name> <name><surname>Loizou</surname> <given-names>P. C.</given-names></name></person-group> (<year>2005</year>). <article-title>The influence of noise on vowel and consonant cues.</article-title> <source><italic>J. Acoust. Soc. Am.</italic></source> <volume>118</volume> <fpage>3874</fpage>&#x2013;<lpage>3888</lpage>. <pub-id pub-id-type="doi">10.1121/1.2118407</pub-id></citation></ref>
<ref id="B33"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pfingst</surname> <given-names>B. E.</given-names></name> <name><surname>Bowling</surname> <given-names>S. A.</given-names></name> <name><surname>Colesa</surname> <given-names>D. J.</given-names></name> <name><surname>Garadat</surname> <given-names>S. N.</given-names></name> <name><surname>Raphael</surname> <given-names>Y.</given-names></name> <name><surname>Shibata</surname> <given-names>S. B.</given-names></name><etal/></person-group> (<year>2011</year>). <article-title>Cochlear infrastructure for electrical hearing.</article-title> <source><italic>Hear. Res.</italic></source> <volume>281</volume> <fpage>65</fpage>&#x2013;<lpage>73</lpage>. <pub-id pub-id-type="doi">10.1016/j.heares.2011.05.002</pub-id> <pub-id pub-id-type="pmid">21605648</pub-id></citation></ref>
<ref id="B34"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Psarros</surname> <given-names>C. E.</given-names></name> <name><surname>Plant</surname> <given-names>K. L.</given-names></name> <name><surname>Lee</surname> <given-names>K.</given-names></name> <name><surname>Decker</surname> <given-names>J. A.</given-names></name> <name><surname>Whitford</surname> <given-names>L. A.</given-names></name> <name><surname>Cowan</surname> <given-names>R. S.</given-names></name></person-group> (<year>2002</year>). <article-title>Conversion from the SPEAK to the ACE strategy in children using the nucleus 24 cochlear implant system: speech perception and speech production outcomes.</article-title> <source><italic>Ear Hear.</italic></source> <volume>23</volume>(<issue>Suppl. 1</issue>), <fpage>18S</fpage>&#x2013;<lpage>27S</lpage>. <pub-id pub-id-type="doi">10.1097/00003446-200202001-00003</pub-id> <pub-id pub-id-type="pmid">11885571</pub-id></citation></ref>
<ref id="B35"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Qi</surname> <given-names>B.</given-names></name> <name><surname>Mao</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>J.</given-names></name> <name><surname>Liu</surname> <given-names>B.</given-names></name> <name><surname>Xu</surname> <given-names>L.</given-names></name></person-group> (<year>2017</year>). <article-title>Relative contributions of acoustic temporal fine structure and envelope cues for lexical tone perception in noise.</article-title> <source><italic>J. Acoust. Soc. Am.</italic></source> <volume>141</volume>:<issue>3022</issue>. <pub-id pub-id-type="doi">10.1121/1.4982247</pub-id></citation></ref>
<ref id="B36"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Roy</surname> <given-names>A. T.</given-names></name> <name><surname>Carver</surname> <given-names>C.</given-names></name> <name><surname>Jiradejvong</surname> <given-names>P.</given-names></name> <name><surname>Limb</surname> <given-names>C. J.</given-names></name></person-group> (<year>2015</year>). <article-title>Musical sound quality in cochlear implant users: a comparison in bass frequency perception between fine structure processing and high-definition continuous interleaved sampling strategies.</article-title> <source><italic>Ear Hear.</italic></source> <volume>36</volume> <fpage>582</fpage>&#x2013;<lpage>590</lpage>. <pub-id pub-id-type="doi">10.1097/aud.0000000000000170</pub-id> <pub-id pub-id-type="pmid">25906173</pub-id></citation></ref>
<ref id="B37"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Santurette</surname> <given-names>S.</given-names></name> <name><surname>Dau</surname> <given-names>T.</given-names></name></person-group> (<year>2011</year>). <article-title>The role of temporal fine structure information for the low pitch of high-frequency complex tones.</article-title> <source><italic>J. Acoust. Soc. Am.</italic></source> <volume>129</volume> <fpage>282</fpage>&#x2013;<lpage>292</lpage>. <pub-id pub-id-type="doi">10.1121/1.3518718</pub-id></citation></ref>
<ref id="B38"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schwartz</surname> <given-names>J.-L.</given-names></name> <name><surname>Bo&#x00EB;</surname> <given-names>L.-J.</given-names></name> <name><surname>Vall&#x00E9;e</surname> <given-names>N.</given-names></name> <name><surname>Abry</surname> <given-names>C.</given-names></name></person-group> (<year>1997</year>). <article-title>The dispersion-focalization theory of vowel systems.</article-title> <source><italic>J. Phonet.</italic></source> <volume>25</volume> <fpage>255</fpage>&#x2013;<lpage>286</lpage>. <pub-id pub-id-type="doi">10.1006/jpho.1997.0043</pub-id></citation></ref>
<ref id="B39"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shannon</surname> <given-names>R. V.</given-names></name> <name><surname>Galvin</surname> <given-names>J. J.</given-names> <suffix>III</suffix></name> <name><surname>Baskent</surname> <given-names>D.</given-names></name></person-group> (<year>2002</year>). <article-title>Holes in hearing.</article-title> <source><italic>J. Assoc. Res. Otolaryngol.</italic></source> <volume>3</volume> <fpage>185</fpage>&#x2013;<lpage>199</lpage>. <pub-id pub-id-type="doi">10.1007/s101620020021</pub-id> <pub-id pub-id-type="pmid">12162368</pub-id></citation></ref>
<ref id="B40"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shannon</surname> <given-names>R. V.</given-names></name> <name><surname>Zeng</surname> <given-names>F. G.</given-names></name> <name><surname>Kamath</surname> <given-names>V.</given-names></name> <name><surname>Wygonski</surname> <given-names>J.</given-names></name> <name><surname>Ekelid</surname> <given-names>M.</given-names></name></person-group> (<year>1995</year>). <article-title>Speech recognition with primarily temporal cues.</article-title> <source><italic>Science</italic></source> <volume>270</volume> <fpage>303</fpage>&#x2013;<lpage>304</lpage>. <pub-id pub-id-type="doi">10.1126/science.270.5234.303</pub-id> <pub-id pub-id-type="pmid">7569981</pub-id></citation></ref>
<ref id="B41"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Skinner</surname> <given-names>M. W.</given-names></name> <name><surname>Holden</surname> <given-names>L. K.</given-names></name> <name><surname>Whitford</surname> <given-names>L. A.</given-names></name> <name><surname>Plant</surname> <given-names>K. L.</given-names></name> <name><surname>Psarros</surname> <given-names>C.</given-names></name> <name><surname>Holden</surname> <given-names>T. A.</given-names></name></person-group> (<year>2002</year>). <article-title>Speech recognition with the nucleus 24 SPEAK, ACE, and CIS speech coding strategies in newly implanted adults.</article-title> <source><italic>Ear Hear.</italic></source> <volume>23</volume> <fpage>207</fpage>&#x2013;<lpage>223</lpage>. <pub-id pub-id-type="doi">10.1097/00003446-200206000-00005</pub-id> <pub-id pub-id-type="pmid">12072613</pub-id></citation></ref>
<ref id="B42"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Smith</surname> <given-names>Z. M.</given-names></name> <name><surname>Delgutte</surname> <given-names>B.</given-names></name> <name><surname>Oxenham</surname> <given-names>A. J.</given-names></name></person-group> (<year>2002</year>). <article-title>Chimaeric sounds reveal dichotomies in auditory perception.</article-title> <source><italic>Nature</italic></source> <volume>416</volume> <fpage>87</fpage>&#x2013;<lpage>90</lpage>. <pub-id pub-id-type="doi">10.1038/416087a</pub-id> <pub-id pub-id-type="pmid">11882898</pub-id></citation></ref>
<ref id="B43"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Stevens</surname> <given-names>K. N.</given-names></name></person-group> (<year>2002</year>). <article-title>Toward a model for lexical access based on acoustic landmarks and distinctive features.</article-title> <source><italic>J. Acoust. Soc. Am.</italic></source> <volume>111</volume> <fpage>1872</fpage>&#x2013;<lpage>1891</lpage>. <pub-id pub-id-type="doi">10.1121/1.1458026</pub-id></citation></ref>
<ref id="B44"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tabibi</surname> <given-names>S.</given-names></name> <name><surname>Kegel</surname> <given-names>A.</given-names></name> <name><surname>Lai</surname> <given-names>W. K.</given-names></name> <name><surname>Dillier</surname> <given-names>N.</given-names></name></person-group> (<year>2020</year>). <article-title>A bio-inspired coding (BIC) strategy for cochlear implants.</article-title> <source><italic>Hear. Res.</italic></source> <volume>388</volume>:<issue>107885</issue>. <pub-id pub-id-type="doi">10.1016/j.heares.2020.107885</pub-id> <pub-id pub-id-type="pmid">32035288</pub-id></citation></ref>
<ref id="B45"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Traunm&#x00FC;ller</surname> <given-names>H.</given-names></name></person-group> (<year>1981</year>). <article-title>Perceptual dimension of openness in vowels.</article-title> <source><italic>J. Acoust. Soc. Am.</italic></source> <volume>69</volume> <fpage>1465</fpage>&#x2013;<lpage>1475</lpage>. <pub-id pub-id-type="doi">10.1121/1.385780</pub-id></citation></ref>
<ref id="B46"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vandali</surname> <given-names>A.</given-names></name> <name><surname>Dawson</surname> <given-names>P.</given-names></name> <name><surname>Au</surname> <given-names>A.</given-names></name> <name><surname>Yu</surname> <given-names>Y.</given-names></name> <name><surname>Brown</surname> <given-names>M.</given-names></name> <name><surname>Goorevich</surname> <given-names>M.</given-names></name><etal/></person-group> (<year>2019</year>). <article-title>Evaluation of the optimized pitch and language strategy in cochlear implant recipients.</article-title> <source><italic>Ear Hear.</italic></source> <volume>40</volume> <fpage>555</fpage>&#x2013;<lpage>567</lpage>. <pub-id pub-id-type="doi">10.1097/aud.0000000000000627</pub-id> <pub-id pub-id-type="pmid">30067558</pub-id></citation></ref>
<ref id="B47"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vandali</surname> <given-names>A.</given-names></name> <name><surname>Sly</surname> <given-names>D.</given-names></name> <name><surname>Cowan</surname> <given-names>R.</given-names></name> <name><surname>van Hoesel</surname> <given-names>R.</given-names></name></person-group> (<year>2015</year>). <article-title>Training of cochlear implant users to improve pitch perception in the presence of competing place cues.</article-title> <source><italic>Ear Hear.</italic></source> <volume>36</volume> <fpage>e1</fpage>&#x2013;<lpage>e13</lpage>. <pub-id pub-id-type="doi">10.1097/aud.0000000000000109</pub-id> <pub-id pub-id-type="pmid">25329372</pub-id></citation></ref>
<ref id="B48"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>S.</given-names></name> <name><surname>Xu</surname> <given-names>L.</given-names></name> <name><surname>Mannell</surname> <given-names>R.</given-names></name></person-group> (<year>2011</year>). <article-title>Relative contributions of temporal envelope and fine structure cues to lexical tone recognition in hearing-impaired listeners.</article-title> <source><italic>J. Assoc. Res. Otolaryngol.</italic></source> <volume>12</volume> <fpage>783</fpage>&#x2013;<lpage>794</lpage>. <pub-id pub-id-type="doi">10.1007/s10162-011-0285-0</pub-id> <pub-id pub-id-type="pmid">21833816</pub-id></citation></ref>
<ref id="B49"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>W.</given-names></name> <name><surname>Zhou</surname> <given-names>N.</given-names></name> <name><surname>Xu</surname> <given-names>L.</given-names></name></person-group> (<year>2011</year>). <article-title>Musical pitch and lexical tone perception with cochlear implants.</article-title> <source><italic>Intern. J. Audiol.</italic></source> <volume>50</volume> <fpage>270</fpage>&#x2013;<lpage>278</lpage>. <pub-id pub-id-type="doi">10.3109/14992027.2010.542490</pub-id> <pub-id pub-id-type="pmid">21190394</pub-id></citation></ref>
<ref id="B50"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Warren</surname> <given-names>R. M.</given-names></name> <name><surname>Bashford</surname> <given-names>J. A.</given-names> <suffix>Jr.</suffix></name> <name><surname>Lenz</surname> <given-names>P. W.</given-names></name></person-group> (<year>2004</year>). <article-title>Intelligibility of bandpass filtered speech: steepness of slopes required to eliminate transition band contributions.</article-title> <source><italic>J. Acoust. Soc. Am.</italic></source> <volume>115</volume> <fpage>1292</fpage>&#x2013;<lpage>1295</lpage>. <pub-id pub-id-type="doi">10.1121/1.1646404</pub-id></citation></ref>
<ref id="B51"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wei</surname> <given-names>C. G.</given-names></name> <name><surname>Cao</surname> <given-names>K.</given-names></name> <name><surname>Zeng</surname> <given-names>F. G.</given-names></name></person-group> (<year>2004</year>). <article-title>Mandarin tone recognition in cochlear-implant subjects.</article-title> <source><italic>Hear. Res.</italic></source> <volume>197</volume> <fpage>87</fpage>&#x2013;<lpage>95</lpage>. <pub-id pub-id-type="doi">10.1016/j.heares.2004.06.002</pub-id> <pub-id pub-id-type="pmid">15504607</pub-id></citation></ref>
<ref id="B52"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Whalen</surname> <given-names>D. H.</given-names></name> <name><surname>Xu</surname> <given-names>Y.</given-names></name></person-group> (<year>1992</year>). <article-title>Information for Mandarin tones in the amplitude contour and in brief segments.</article-title> <source><italic>Phonetica</italic></source> <volume>49</volume> <fpage>25</fpage>&#x2013;<lpage>47</lpage>. <pub-id pub-id-type="doi">10.1159/000261901</pub-id> <pub-id pub-id-type="pmid">1603839</pub-id></citation></ref>
<ref id="B53"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wilson</surname> <given-names>B. S.</given-names></name> <name><surname>Finley</surname> <given-names>C. C.</given-names></name> <name><surname>Lawson</surname> <given-names>D. T.</given-names></name> <name><surname>Wolford</surname> <given-names>R. D.</given-names></name> <name><surname>Eddington</surname> <given-names>D. K.</given-names></name> <name><surname>Rabinowitz</surname> <given-names>W. M.</given-names></name></person-group> (<year>1991</year>). <article-title>Better speech recognition with cochlear implants.</article-title> <source><italic>Nature</italic></source> <volume>352</volume> <fpage>236</fpage>&#x2013;<lpage>238</lpage>. <pub-id pub-id-type="doi">10.1038/352236a0</pub-id> <pub-id pub-id-type="pmid">1857418</pub-id></citation></ref>
<ref id="B54"><citation citation-type="journal"><collab>World Health Organization [WHO]</collab> (<year>2020</year>). Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.who.int/news-room/factsheets/detail/deafness-and-hearing-loss">https://www.who.int/news-room/factsheets/detail/deafness-and-hearing-loss</ext-link> <comment>(accessed April 1, 2021)</comment>.</citation></ref>
<ref id="B55"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>J. L.</given-names></name> <name><surname>Yang</surname> <given-names>H. M.</given-names></name> <name><surname>Lin</surname> <given-names>Y. H.</given-names></name> <name><surname>Fu</surname> <given-names>Q. J.</given-names></name></person-group> (<year>2007</year>). <article-title>Effects of computer-assisted speech training on Mandarin-speaking hearing-impaired children.</article-title> <source><italic>Audiol. Neurootol.</italic></source> <volume>12</volume> <fpage>307</fpage>&#x2013;<lpage>312</lpage>. <pub-id pub-id-type="doi">10.1159/000103211</pub-id> <pub-id pub-id-type="pmid">17536199</pub-id></citation></ref>
<ref id="B56"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>L.</given-names></name> <name><surname>Pfingst</surname> <given-names>B. E.</given-names></name></person-group> (<year>2003</year>). <article-title>Relative importance of temporal envelope and fine structure in lexical-tone perception.</article-title> <source><italic>J. Acoust. Soc. Am.</italic></source> <volume>114</volume>(<issue>6 Pt 1</issue>), <fpage>3024</fpage>&#x2013;<lpage>3027</lpage>. <pub-id pub-id-type="doi">10.1121/1.1623786</pub-id></citation></ref>
<ref id="B57"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>L.</given-names></name> <name><surname>Thompson</surname> <given-names>C. S.</given-names></name> <name><surname>Pfingst</surname> <given-names>B. E.</given-names></name></person-group> (<year>2005</year>). <article-title>Relative contributions of spectral and temporal cues for phoneme recognition.</article-title> <source><italic>J. Acoust. Soc. Am.</italic></source> <volume>117</volume> <fpage>3255</fpage>&#x2013;<lpage>3267</lpage>. <pub-id pub-id-type="doi">10.1121/1.1886405</pub-id></citation></ref>
<ref id="B58"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zeng</surname> <given-names>F. G.</given-names></name></person-group> (<year>2004</year>). <article-title>Trends in cochlear implants.</article-title> <source><italic>Trends Amplif.</italic></source> <volume>8</volume> <fpage>1</fpage>&#x2013;<lpage>34</lpage>. <pub-id pub-id-type="doi">10.1177/108471380400800102</pub-id> <pub-id pub-id-type="pmid">15247993</pub-id></citation></ref>
<ref id="B59"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zeng</surname> <given-names>F. G.</given-names></name> <name><surname>Rebscher</surname> <given-names>S.</given-names></name> <name><surname>Harrison</surname> <given-names>W.</given-names></name> <name><surname>Sun</surname> <given-names>X.</given-names></name> <name><surname>Feng</surname> <given-names>H.</given-names></name></person-group> (<year>2008</year>). <article-title>Cochlear implants: system design, integration, and evaluation.</article-title> <source><italic>IEEE Rev. Biomed. Eng.</italic></source> <volume>1</volume> <fpage>115</fpage>&#x2013;<lpage>142</lpage>. <pub-id pub-id-type="doi">10.1109/rbme.2008.2008250</pub-id> <pub-id pub-id-type="pmid">19946565</pub-id></citation></ref>
<ref id="B60"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zheng</surname> <given-names>Z.</given-names></name> <name><surname>Li</surname> <given-names>K.</given-names></name> <name><surname>Guo</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Xiao</surname> <given-names>L.</given-names></name> <name><surname>Liu</surname> <given-names>C.</given-names></name><etal/></person-group> (<year>2021</year>). <article-title>The relative weight of temporal envelope cues in different frequency regions for Mandarin disyllabic word recognition.</article-title> <source><italic>Front. Neurosci.</italic></source> <volume>15</volume>:<issue>670192</issue>. <pub-id pub-id-type="doi">10.3389/fnins.2021.670192</pub-id> <pub-id pub-id-type="pmid">34335156</pub-id></citation></ref>
<ref id="B61"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ziese</surname> <given-names>M.</given-names></name> <name><surname>St&#x00FC;tzel</surname> <given-names>A.</given-names></name> <name><surname>von Specht</surname> <given-names>H.</given-names></name> <name><surname>Begall</surname> <given-names>K.</given-names></name> <name><surname>Freigang</surname> <given-names>B.</given-names></name> <name><surname>Sroka</surname> <given-names>S.</given-names></name><etal/></person-group> (<year>2000</year>). <article-title>Speech understanding with the CIS and the n-of-m strategy in the MED-EL COMBI 40+ system.</article-title> <source><italic>J. Otorhinolaryngol. Relat. Spec.</italic></source> <volume>62</volume> <fpage>321</fpage>&#x2013;<lpage>329</lpage>. <pub-id pub-id-type="doi">10.1159/000027763</pub-id> <pub-id pub-id-type="pmid">11054016</pub-id></citation></ref>
</ref-list>
<fn-group>
<fn id="footnote1">
<label>1</label>
<p><ext-link ext-link-type="uri" xlink:href="http://angelsound.tigerspeech.com/">http://angelsound.tigerspeech.com/</ext-link></p></fn>
</fn-group>
</back>
</article>