<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Psychol.</journal-id>
<journal-title>Frontiers in Psychology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Psychol.</abbrev-journal-title>
<issn pub-type="epub">1664-1078</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpsyg.2022.895063</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Psychology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Singing Mandarin? What Short-Term Memory Capacity, Basic Auditory Skills, and Musical and Singing Abilities Reveal About Learning Mandarin</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Christiner</surname>
<given-names>Markus</given-names>
</name>
<xref rid="aff1" ref-type="aff"><sup>1</sup></xref>
<xref rid="aff2" ref-type="aff"><sup>2</sup></xref>
<xref rid="c001" ref-type="corresp"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/94468/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Renner</surname>
<given-names>Julia</given-names>
</name>
<xref rid="aff3" ref-type="aff"><sup>3</sup></xref>
<xref rid="aff4" ref-type="aff"><sup>4</sup></xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Gro&#x00DF;</surname>
<given-names>Christine</given-names>
</name>
<xref rid="aff2" ref-type="aff"><sup>2</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/352168/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Seither-Preisler</surname>
<given-names>Annemarie</given-names>
</name>
<xref rid="aff1" ref-type="aff"><sup>1</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/343019/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Benner</surname>
<given-names>Jan</given-names>
</name>
<xref rid="aff5" ref-type="aff"><sup>5</sup></xref>
<xref rid="aff6" ref-type="aff"><sup>6</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/343167/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Schneider</surname>
<given-names>Peter</given-names>
</name>
<xref rid="aff1" ref-type="aff"><sup>1</sup></xref>
<xref rid="aff2" ref-type="aff"><sup>2</sup></xref>
<xref rid="aff5" ref-type="aff"><sup>5</sup></xref>
<xref rid="aff6" ref-type="aff"><sup>6</sup></xref>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Center for Systematic Musicology, Faculty of Arts and Humanities, University of Graz</institution>, <addr-line>Graz</addr-line>, <country>Austria</country></aff>
<aff id="aff2"><sup>2</sup><institution>Jazeps Vitols Latvian Academy of Music</institution>, <addr-line>Riga</addr-line>, <country>Latvia</country></aff>
<aff id="aff3"><sup>3</sup><institution>Department of East Asian Studies, University of Vienna</institution>, <addr-line>Vienna</addr-line>, <country>Austria</country></aff>
<aff id="aff4"><sup>4</sup><institution>Department of Linguistics, University of Vienna</institution>, <addr-line>Vienna</addr-line>, <country>Austria</country></aff>
<aff id="aff5"><sup>5</sup><institution>Department of Neuroradiology, Section Biomagnetism Heidelberg Medical School</institution>, <addr-line>Heidelberg</addr-line>, <country>Germany</country></aff>
<aff id="aff6"><sup>6</sup><institution>Department of Neurology, Section Biomagnetism Heidelberg Medical School</institution>, <addr-line>Heidelberg</addr-line>, <country>Germany</country></aff>
<author-notes>
<fn id="fn0001" fn-type="edited-by"><p>Edited by: Franco Delogu, Lawrence Technological University, United States</p></fn>
<fn id="fn0002" fn-type="edited-by"><p>Reviewed by: Matthew Cole, Lawrence Technological University, United States; Si Chen, Hong Kong Polytechnic University, Hong Kong SAR, China</p></fn>
<corresp id="c001">&#x002A;Correspondence: Markus Christiner, <email>markus.chrisitner@uni-graz.at</email></corresp>
<fn id="fn0003" fn-type="other"><p>This article was submitted to Cognitive Science, a section of the journal Frontiers in Psychology</p></fn>
</author-notes>
<pub-date pub-type="epub">
<day>16</day>
<month>06</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>13</volume>
<elocation-id>895063</elocation-id>
<history>
<date date-type="received">
<day>12</day>
<month>03</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>05</day>
<month>05</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2022 Christiner, Renner, Gro&#x00DF;, Seither-Preisler, Benner and Schneider.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Christiner, Renner, Gro&#x00DF;, Seither-Preisler, Benner and Schneider</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Learning Mandarin has become increasingly important in the Western world but is rather difficult to be learnt by speakers of non-tone languages. Since tone language learning requires very precise tonal ability, we set out to test whether musical skills, musical status, singing ability, singing behavior during childhood, basic auditory skills, and short-term memory ability contribute to individual differences in Mandarin performance. Therefore, we developed Mandarin tone discrimination and pronunciation tasks to assess individual differences in adult participants&#x2019; (<italic>N</italic>&#x2009;=&#x2009;109) tone language ability. Results revealed that short-term memory capacity, singing ability, pitch perception preferences, and tone frequency (high vs. low tones) were the most important predictors, which explained individual differences in the Mandarin performances of our participants. Therefore, it can be concluded that training of basic auditory skills, musical training including singing should be integrated in the educational setting for speakers of non-tone languages who learn tone languages such as Mandarin.</p>
</abstract>
<kwd-group>
<kwd>singing ability</kwd>
<kwd>fundamental and spectral listener</kwd>
<kwd>tone frequency</kwd>
<kwd>Mandarin</kwd>
<kwd>musical ability</kwd>
<kwd>short-term memory</kwd>
</kwd-group>
<counts>
<fig-count count="2"/>
<table-count count="3"/>
<equation-count count="0"/>
<ref-count count="75"/>
<page-count count="14"/>
<word-count count="11207"/>
</counts>
</article-meta>
</front>
<body>
<sec id="sec1" sec-type="intro">
<title>Introduction</title>
<p>Mandarin is not only one of the most widely spoken languages but also by far the most widely spoken tone language in the world. Mandarin appears to be rather difficult to be learnt by speakers of non-tone languages (<xref ref-type="bibr" rid="ref24">Christiner et al., 2018</xref>). This could be related to the fact that in tone languages, such as Mandarin, tones can denote semantic change at the lexical level. In Mandarin, every stressed/full syllable carries a syllable tone; in weak(&#x2212;stressed) syllables the tone becomes neutralized (<xref ref-type="bibr" rid="ref12">Chen, 1984</xref>). Mandarin has four different tones, which are, theoretically speaking, fixed/invariant syllable tones and one, so-called &#x201C;neutral tone,&#x201D; which varies according to the preceding syllable tone (<xref ref-type="bibr" rid="ref12">Chen, 1984</xref>). Mandarin tones are often illustrated with a pitch chart (<xref ref-type="bibr" rid="ref10">Chao, 1930</xref>), which locates the four tones within a five-level tone scale. Level 1 represents the lowest pitch and level 5 the highest as illustrated in <xref rid="fig1" ref-type="fig">Figure 1</xref>. According to the traditional tone chart, the first tone can be described as a &#x201C;high level tone&#x201D; (5-5), the second tone as a &#x201C;(high) rising tone&#x201D; (3-5), the third tone as a &#x201C;falling-rising tone&#x201D; (2-1-4), and the fourth tone as a &#x201C;falling tone&#x201D; (5-1; <xref ref-type="bibr" rid="ref11">Chao, 1965</xref>). To date, this presentation of Mandarin tones is still widely accepted. The analysis of speech data, however, showed that particularly the third tone is not accurately represented within the traditional tone chart. The rising part (1-4) is normally neglected in multisyllabic expressions, leading to the reconceptualization of the third tone as a &#x201C;low dipping&#x201D; tone (2-1; <xref ref-type="bibr" rid="ref48">Lin, 1985</xref>) or a low tone without the rising part. In comparison to the four full syllable tones, the neutral tone does not have a fixed tone contour. It therefore is not represented within the chart (see <xref rid="fig1" ref-type="fig">Figure 1</xref>). It is shorter in length and varies according to the preceding syllable. This phenomenon, also referred to as &#x201C;Tone-Sandhi&#x201D; (<xref ref-type="bibr" rid="ref67">Wang, 1967</xref>; <xref ref-type="bibr" rid="ref13">Chen, 2000</xref>) also applies to full tones, especially the third tone; however, its variation is less complex. Due to these circumstances, the neutral tone was excluded in this study. Since Mandarin requires precise ability to distinguish tones, the impact of musical ability on the acquisition of tone languages has gained increasing importance (<xref ref-type="bibr" rid="ref39">Han et al., 2019</xref>).</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption><p>The figure illustrates the tone chart of the five Mandarin tones. According to the traditional tone chart, the red line shows the first tone which is described as a &#x201C;high level tone&#x201D; (5-5). Furthermore, the green line represents the second tone which is defined as the &#x201C;(high) rising tone&#x201D; (3-5), while the orange line shows the third tone, the &#x201C;falling-rising tone&#x201D; (2-1-4). Finally, the fourth tone determined as a &#x201C;falling tone&#x201D; (5-1) is shown by the blue line (<xref ref-type="bibr" rid="ref11">Chao, 1965</xref>).</p></caption>
<graphic xlink:href="fpsyg-13-895063-g001.tif"/>
</fig>
</sec>
<sec id="sec2">
<title>Music and Language Acquisition</title>
<p>Music and language share a set of comparable features and are both based on hierarchical structures (<xref ref-type="bibr" rid="ref42">Jackendoff and Lerdahl, 2006</xref>). They consist of tonal properties and temporal features, which is a fundamental reason why overlapping features of both, language and music, have been intensively studied over the past 2&#x2009;decades. For acquiring foreign languages, for taking up musical instruments, as well as for learning to sing, individuals need to be responsive to perceive and to reproduce the input they receive. Consequently, individual differences require considering perceptual and productive domains in language and music which put emphasis on similar, but also on different abilities. For instance, research on individual differences in language pronunciation skills has shown that both vocalists and instrumentalists outperformed non-musicians (<xref ref-type="bibr" rid="ref21">Christiner and Reiterer, 2015</xref>). However, vocalists, who typically possess enhanced vocal flexibility and refined vocal motor skills, were still better in the pronunciation of unfamiliar languages than the instrumentalists (<xref ref-type="bibr" rid="ref23">Christiner and Reiterer, 2019</xref>). On the other hand, more elaborate music and speech perceptions skills have been noted for professional instrumentalists who outperformed vocalists in a follow-up study (<xref ref-type="bibr" rid="ref17">Christiner, 2020</xref>). Consequently, in order to give a more holistic impression of individual differences in music or language capacities, assessments should consider both dimensions production and perception. We therefore included language and music measures, which focus on perception and production tasks in order to relate these measures to Mandarin capacity.</p>
<sec id="sec3">
<title>Neurophysiology of Music and Language</title>
<p>Neurophysiological research suggests that neural processing of language and music is shared to some extent, since acoustic signals of speech and music show similarities in temporal and spectral complexity (<xref ref-type="bibr" rid="ref63">Sch&#x00F6;n et al., 2004</xref>; <xref ref-type="bibr" rid="ref28">Ding et al., 2017</xref>). Therefore, it has been suggested that musical training reorganizes common neural circuits which not only improves musical performance but also language functions (<xref ref-type="bibr" rid="ref41">Intartaglia et al., 2017</xref>). Positive transfer from music to language has particularly been found for aspects, which are related to phonetic ability. Musicians are generally said to be better in neural processing of non-native lexical tones (<xref ref-type="bibr" rid="ref1">Alexander et al., 2005</xref>). Musical training has been found to enhance the neural processing of speech (<xref ref-type="bibr" rid="ref4">Besson et al., 2011</xref>; <xref ref-type="bibr" rid="ref55">Parbery-Clark et al., 2012</xref>; <xref ref-type="bibr" rid="ref41">Intartaglia et al., 2017</xref>) while musical aptitude has been associated with enhanced duration speech perception (<xref ref-type="bibr" rid="ref14">Chobert et al., 2014</xref>) and speech segmentation ability (<xref ref-type="bibr" rid="ref33">Fran&#x00E7;ois et al., 2013</xref>). Neurophysiological research leaves no doubt that musical capacity and musical training have positive effect on language functions.</p>
</sec>
<sec id="sec4">
<title>Cognitive-Behavioral Components of Music and Language Acquisition</title>
<p>Pitch is one of the most prominent features of music and language (<xref ref-type="bibr" rid="ref49">Liu and Kager, 2017</xref>). Research has shown that musical training improves pitch discrimination ability in speech (<xref ref-type="bibr" rid="ref52">Moreno, 2009</xref>). The ability to discriminate high vs. low tones was related to speech perception aptitude and to the number of foreign languages participants mastered (<xref ref-type="bibr" rid="ref18">Christiner et al., 2022</xref>). Languages in which tones determine semantic information require high tonal ability. Research contrasting tone and non-tone language speakers has shown that tone language speakers possess enhanced pitch memory, show higher tonal skills and improved pitch processing ability compared to non-tone language speakers (<xref ref-type="bibr" rid="ref5">Bidelman et al., 2013</xref>; <xref ref-type="bibr" rid="ref23">Christiner and Reiterer, 2019</xref>). In another study, the ability to identify Mandarin tones was assessed in English speaking musicians and non-musicians. The findings revealed that musical training facilitated lexical tone identification (<xref ref-type="bibr" rid="ref46">Lee and Hung, 2008</xref>) and the learning of Mandarin in general (<xref ref-type="bibr" rid="ref39">Han et al., 2019</xref>). Other researchers presented similar findings and outlined that piano playing enhances processing of pitch which in turn improved word discrimination ability in Mandarin (<xref ref-type="bibr" rid="ref53">Nan et al., 2018</xref>). Since research has shown that musical training facilitates language functions, we also wanted to address musical tonal aptitude and musical status of our participants in the research design.</p>
<p>While there is no doubt that musical ability improves tone language learning, individual differences in how languages are perceived may also play an important role for how well languages are performed. Recent research has shown that individuals who perceive natural languages to be more melodic than others also retrieve and pronounce these languages more accurately (<xref ref-type="bibr" rid="ref19">Christiner et al., 2021</xref>). In addition, the researchers&#x2019; findings indicated that the high melodic language perceivers performed significantly better than the low melodic language perceivers in all typologically different languages (<xref ref-type="bibr" rid="ref19">Christiner et al., 2021</xref>). Individual listening types can also be found in the musical domain. Most acoustic signals including the voice and the sound of music are composed of one fundamental and multiple integer harmonics (<xref ref-type="bibr" rid="ref62">Schneider and Wengenroth, 2009</xref>). Therefore, two main dimensions have been recognized: First the fundamental pitch and second spectral pitches derived from the frequency components (<xref ref-type="bibr" rid="ref61">Schneider et al., 2005</xref>; <xref ref-type="bibr" rid="ref64">Seither-Preisler et al., 2007</xref>; <xref ref-type="bibr" rid="ref57">Preisler et al., 2011</xref>). In general, there are two complementary types of listeners: &#x201C;fundamental&#x201D; and &#x201C;spectral&#x201D; listeners. What influences whether individuals are &#x201C;fundamental&#x201D; or &#x201C;spectral&#x201D; listeners is not entirely understood. Potential explanations could be genetic dispositions, but also musical practice has been shown to induce a perceptual shift from spectral toward holistic listeners (<xref ref-type="bibr" rid="ref64">Seither-Preisler et al., 2007</xref>). Pitch perception preferences could be related to choices for musical instruments. For example, in a study about individual differences on preferences for musical instruments &#x201C;[f]undamental pitch listeners played predominantly percussive or high-pitched instruments, whereas spectral pitch listeners preferred lower-pitch melodic instruments and singing&#x201D; (<xref ref-type="bibr" rid="ref61">Schneider et al., 2005</xref>, 390). As the melodic perception of unfamiliar languages had also an impact on how well languages were imitated, we also wanted to assess whether pitch perception preferences, i.e., one of the two listening types (e.g., &#x201C;fundamental&#x201D; or &#x201C;spectral&#x201D; listeners) perform better in any of our language measures. Besides pitch, rhythm is the second prominent feature which plays a crucial role in language and music. The rhythmic components of music and speech are mainly involved in how both faculties are organized. Music is characterized by a regular timed beat to which one can synchronize with periodic movements (<xref ref-type="bibr" rid="ref56">Patel, 2007</xref>). In language, the rhythmic component facilitates that speech sounds are grouped into meaningful units. Foreign language learners often fail to understand languages as they do not master differentiating when words begin or end in a sequence of spoken language (<xref ref-type="bibr" rid="ref56">Patel, 2007</xref>). Thus the rhythm of language does not only provide crucial information about the language phonology, but also about the syntax and semantics of phrases and sentences (<xref ref-type="bibr" rid="ref42">Jackendoff and Lerdahl, 2006</xref>). It is therefore not surprising that musical studies have shown that rhythmic and duration abilities predict the ability to segment speech (<xref ref-type="bibr" rid="ref33">Fran&#x00E7;ois et al., 2013</xref>; <xref ref-type="bibr" rid="ref14">Chobert et al., 2014</xref>). Therefore, assessing musical parameters and its relationship to language functions should also include rhythmic musical measures.</p>
<p>Assessing musical performance is easily achieved by using familiar song singing tasks, since they can be targeted at both musicians and non-musicians (<xref ref-type="bibr" rid="ref17">Christiner, 2020</xref>). In particular, singing has been associated with pronunciation skills. Researchers have shown that the ability to memorize new vocabulary of adult language learners improves when words are sung (<xref ref-type="bibr" rid="ref50">Ludke et al., 2014</xref>). The same is also true when children sing new vocabulary (<xref ref-type="bibr" rid="ref65">Thiessen and Saffran, 2009</xref>) and a very recent study has shown that singing to infants has a positive impact on vocabulary building in later ages (<xref ref-type="bibr" rid="ref32">Franco et al., 2021</xref>). These studies assume that the singing of new words serves as mnemonic with which new utterances are better and more easily stored in the long-term memory (<xref ref-type="bibr" rid="ref37">Gordon et al., 2010</xref>). Beside singing as a tool which facilitates language acquisition processes, psychological and psycholinguistic research has focused on analyzing whether singing capacity facilitates language ability (<xref ref-type="bibr" rid="ref16">Christiner, 2018</xref>; <xref ref-type="bibr" rid="ref24">Christiner et al., 2018</xref>, <xref ref-type="bibr" rid="ref19">2021</xref>, <xref ref-type="bibr" rid="ref18">2022</xref>; <xref ref-type="bibr" rid="ref301">Coumel et al., 2019</xref>). These studies provided evidence that the relationship between singing ability and language pronunciation has its roots in enhanced vocal-motor skills. Singers outperformed non-musicians in language pronunciation tasks (<xref ref-type="bibr" rid="ref21">Christiner and Reiterer, 2015</xref>) as well as evidence has also been provided that the amount of singing during childhood influences both singing capacity and the ability to acquire foreign language pronunciation later in adulthood (<xref ref-type="bibr" rid="ref18">Christiner et al., 2022</xref>). The positive relationship between singing ability and language pronunciation has been replicated for all ages. Therefore, we also wanted to assess whether singing ability as well as singing behavior during childhood also contributes to Mandarin performances.</p>
<p>Short-term memory (STM) capacity as measured by digit spans or non-word spans is one of the most important predictors for individual differences in the ability to acquire new languages (<xref ref-type="bibr" rid="ref29">D&#x00F6;rnyei, 2005</xref>; <xref ref-type="bibr" rid="ref2">Baddeley, 2010</xref>). Primarily, language learning processes such as speech production, reading comprehension, and vocabulary learning have been related to STM capacity (<xref ref-type="bibr" rid="ref35">Gathercole and Baddeley, 1990</xref>, <xref ref-type="bibr" rid="ref36">1993</xref>; <xref ref-type="bibr" rid="ref16">Christiner, 2018</xref>; <xref ref-type="bibr" rid="ref24">Christiner et al., 2018</xref>). The phonological loop is most essential for language processes. It consists of two main aspects, the first being a kind of phonological store in which memory traces are held before they fade, and the second, the verbal or subvocal rehearsal mechanisms that allow decaying memory traces to be refreshed (<xref ref-type="bibr" rid="ref2">Baddeley, 2010</xref>). Therefore, the longer the words, the more slowly they are rehearsed (<xref ref-type="bibr" rid="ref66">Thorn and Gathercole, 2001</xref>). This increases the chance that words get lost in the phonological store (<xref ref-type="bibr" rid="ref3">Baddeley et al., 1984</xref>). Cross-linguistic comparisons of digit span testing have shown that the shorter the names for the digits are, the higher the number of items is that can be repeated (<xref ref-type="bibr" rid="ref66">Thorn and Gathercole, 2001</xref>). Typical tasks that measure the STM are forward span tasks where for instance numbers, dissimilar or simple words need to be recalled in a correct serial order by writing them down (<xref ref-type="bibr" rid="ref31">Engle et al., 1999</xref>), or non-word repetition which measures the (phonological) STM (<xref ref-type="bibr" rid="ref34">Gathercole, 2006</xref>). This may be one fundamental reason why one of our STM measures, the forward digit span, is also highly interrelated with one of the language measures as used in this study such as language pronunciation tasks of unfamiliar language stimuli. For the backward counterpart, participants are usually instructed to repeat digits or words in reversed order (<xref ref-type="bibr" rid="ref31">Engle et al., 1999</xref>). Presumably backward spans focus more on controlled attention which makes them more of a hybrid task, but still more evidence has been provided that they may be best categorized as STM tasks. A factor analysis revealed that both forward and backward tasks are components of the same factor (<xref ref-type="bibr" rid="ref31">Engle et al., 1999</xref>; <xref ref-type="bibr" rid="ref17">Christiner, 2020</xref>), which could lead to the interpretation that backward spans demand a &#x201C;mental transformation&#x201D; as there is not a new stimulus being imposed (<xref ref-type="bibr" rid="ref25">Conway et al., 2005</xref>). In previous research, we have noted that digit forward spans always yielded stronger relationships to language measures such as pronunciation tasks or non-word spans than backward spans (<xref ref-type="bibr" rid="ref17">Christiner, 2020</xref>; <xref ref-type="bibr" rid="ref19">Christiner et al., 2021</xref>). This may be the case since forward digit and forward language spans require many similar cognitive abilities which are why we also treat STM capacity as a covariate in the later parts.</p>
<p>Whether there is a &#x201C;tonal loop&#x201D; in music as an equivalent of the phonological loop for language capacity is not entirely understood. Early research speculated about the existence of a separate storage for tonal and speech material (<xref ref-type="bibr" rid="ref60">Salam&#x00E9; and Baddeley, 1989</xref>), while more recently shared processing and shared neuronal networks for musical and verbal sounds have been reported (<xref ref-type="bibr" rid="ref44">Koelsch et al., 2009</xref>; <xref ref-type="bibr" rid="ref72">Williamson et al., 2010</xref>). This may be one reason why STM capacity is associated with enhanced language and musical capacities.</p>
<p>The findings of the aforementioned studies suggest that beside STM ability, musical capacities improve language functions. This reflects what current research suggests: a strong relationship between musical and language abilities. Despite the many findings which outlined relationships between musical ability and language capacity, several aspects need to be explored and addressed in more detail. Language typology is a crucial factor, and it can be suggested that tone languages have a lot in common with music which is why musical abilities seem to facilitate tone language acquisition processes. Therefore, we used music measures of previous research and developed new Mandarin language tasks in order to examine their relationship. While research has already shown a relationship between musical status and pitch perception ability, fewer research has focused on whether rhythmic ability also predict tone language capacity. In addition, there is no research available which has focused on whether one of the two complementary listener types, &#x201C;fundamental&#x201D; and &#x201C;spectral&#x201D; listeners, are benefitted in tone language acquisition. Furthermore, while singing ability and singing behavior during childhood has been related to foreign language pronunciation tasks, we also wanted to examine whether we could detect associations between one of the singing variables and the Mandarin discrimination and/or the Mandarin syllable recognition task. To address our research questions, we used the newly developed Mandarin tasks, the measures of musical aptitude, singing ability, singing behavior during childhood, basic auditory skills, and STM capacity and tested musicians, amateurs, and non-musicians.</p>
</sec>
</sec>
<sec id="sec5" sec-type="materials|methods">
<title>Materials and Methods</title>
<sec id="sec6">
<title>Participants</title>
<p>In this investigation, a sample of 109 adult participants was selected who were all tested for their Mandarin performance, their musical ability (music aptitude and singing ability), their STM capacity, and basic auditory skills. The participants were all German native speakers and did not speak nor had been taught in Mandarin before they were tested. This was one of the most important parameters which should facilitate that the participants all had the same prior knowledge of Mandarin. The participants spoke foreign languages, such as English, French, Spanish, Italian, Croatian, and Dutch. The age range was <italic>M</italic>&#x2009;=&#x2009;24.26 <italic>SE</italic>&#x2009;=&#x2009;&#x00B1;1.06. In this study, 51 participants were female, whereas 58 participants were male.</p>
</sec>
<sec id="sec7">
<title>Educational Status</title>
<p>The participants&#x2019; educational status was specified according to the educational status which had been completed at testing time. The findings have shown that 41 participants had finished the main general secondary school, 16 the technical and vocational school, 33 secondary academic high school (general qualification for university entrance), one the post-secondary non-tertiary education, two &#x00B4;bachelor studies, 15 master studies, and one had a doctoral degree.</p>
</sec>
<sec id="sec8">
<title>Musical Background</title>
<p>Assessing the musical background was based on our previous research (<xref ref-type="bibr" rid="ref19">Christiner et al., 2021</xref>). Participants were instructed to label themselves to be either professional musicians, amateurs, or non-musicians. Therefore, the participants received further instructions. Being a non-musician meant that they were unable to play a musical instrument. In addition, we asked the participants whether they no longer trained or no longer played a musical instrument. The latter cases were not suitable for this research and excluded from the analyses. The participants were instructed to label themselves to be amateurs, if they could play one or more musical instruments and thus played them occasionally, but not professionally. The participants were asked to label themselves to be professional musicians if they played regularly in public as members of an orchestra at least for 2&#x2009;years, or studied music for three semesters, were music teachers, or showed equivalent qualifications. According to these definitions, 34 participants were non-musicians, 30 amateurs, and 45 professional musicians.</p>
</sec>
<sec id="sec9">
<title>Measuring Mandarin Ability</title>
<p>Individual differences in Mandarin ability were assessed by perception and production tasks. The perception tasks consisted of two different measures: a tone discrimination and a syllable tone recognition task. The production task consisted of pronunciation tasks. The collection of our sample included all four Mandarin syllables tones [the &#x201C;high level tone&#x201D; (5-5), the &#x201C;(high) rising tone&#x201D; (3-5), the &#x201C;falling-rising tone&#x201D; (2-1-4), and the &#x201C;falling tone&#x201D; (5-1)]. The &#x201C;Tone-Sandhi,&#x201D; the neutral tone, was excluded from the perception tasks. We did not coin separate scores for each of the four Mandarin syllable tones but used only composite scores of the three Mandarin measures which we developed.</p>
<sec id="sec10">
<title>Tone Discrimination Task (Mandarin D)</title>
<p>The tone discrimination task consisted of 18 paired samples, which were either identical or contained a change of a particular sound in the second statement (e.g., b&#x00F9;zh&#x00EC; vs. b&#x00F9;zh&#x012B;). The length of the words and phrases of the 18 examples varied between 2 and 11 syllables. The first statement of the paired samples was separated by a pause of 1&#x2009;s from the second statement played. All paired samples are introduced by a different speaker who indicates the paired sample by a number. Before the participants run the test, they receive four practicing items, where they were introduced to the task. They were instructed by the experimenter and allowed to practice as long as they understood the tasks correctly. After familiarization, they run the entire samples in a sequence.</p>
</sec>
<sec id="sec11">
<title>Syllable Tone Recognition Task (Mandarin S)</title>
<p>In this condition, the participants were again listening to paired samples but had to decide in which syllable of the second statement a tonal change occurred. The syllable tone recognition task consists of 16 paired samples where a particular syllable in the second statement contains a tonal change which has to be indicated. All syllables were separated from each other and visually presented to the participants. As the two statements differed by only one tonal change, the wording and syllable structure of both statements in all conditions were the same, and just the diacritics were removed. This aimed at avoiding tonal changes that were recognized based on the visual tone representation. The length of the words and phrases of the samples vary between two and seven syllables. Like for the tone discrimination task, the participants received practicing items before they ran the entire samples in a row.</p>
</sec>
<sec id="sec12">
<title>Mandarin Pronunciation Task (Mandarin P)</title>
<p>The speech production task consisted of three Mandarin phrases (spoken by native speakers) of 7, 9, and 11 syllables, which were repeated by the participants after they had listened to them for the third time. Assessing individual differences in language pronunciation has already been carried out in previous investigations and has shown high ecological validity, since it resembles a foreign language learning condition (<xref ref-type="bibr" rid="ref17">Christiner, 2020</xref>; <xref ref-type="bibr" rid="ref19">Christiner et al., 2021</xref>). Like in former investigations, the recordings of the participants were normalized for their loudness and rated by five Mandarin native speakers. They were introduced to evaluate how well the participants preserved the rhythmical structure, the melodic aspects of the original phrase, completeness of the sentence material, and the overall performance. Therefore, they had to give a score which ranged between 0 and 10. The four criteria were then collapsed into a single score. The interrater reliability was assessed by intra-class coefficient analysis, as provided in the supplement (see <xref ref-type="supplementary-material" rid="SM1">Supplementary Table S1</xref>).</p>
</sec>
</sec>
<sec id="sec13">
<title>Gordon&#x2019;s Musical Aptitude Test</title>
<p>The Advanced Measures of Musical Audiation (AMMA test; <xref ref-type="bibr" rid="ref302">Gordon, 1989</xref>) consists of rhythm and tonal discrimination tasks and measures the ability to internalize musical structures. The paired musical statements are embedded in one single test design where either rhythmic, tonal, or no changes may occur. For the tonal discrimination tasks (AMMA T), notes are modified in pitch in the condition when the melody is played for the second time. In the rhythm subtest (AMMA R), tempo, meter, or duration may be altered in the comparison condition. This aptitude test is usually targeted at university music and non-music majors and high school students. The test consists of 33 items, whereby the first 3 are familiarization tasks that were excluded from the final analysis.</p>
</sec>
<sec id="sec14">
<title>Tone Frequency and Duration</title>
<p>To test basic sound discrimination abilities, two subtests (tone frequency and duration) of the primary auditory threshold measure KLAWA (<italic>Klangwahrnehmug</italic>) were used. KLAWA is an inhouse computer-based threshold measurement. Difference limes are measured for tone frequency (&#x201C;low vs. high&#x201D;) and duration (&#x201C;short vs. long&#x201D;). Based on an &#x201C;alternative-forced choice&#x201D; (<xref ref-type="bibr" rid="ref303">Jepsen et al., 2008</xref>), this method to measure individual perceptual thresholds can be used for scientific investigations to study subjective auditory processing and language development. In this computer-aided test procedure exact scientifically measured quantities [cent&#x2009;=&#x2009;1/100 semitone for recording the pitch, and milliseconds (ms) for time measurements], the above-mentioned hearing performance was determined, which can largely vary from subject to subject (&#x003E;factor 100).</p>
<p>In an alternative forced-choice paradigm, reference, and test tones (sinusoids) separated by an interstimulus interval of 500&#x2009;ms are presented. Participants are asked to decide per mouse-click, which of the presented tones sounds higher or longer in the tone frequency subtests and which of the presented sounds are shorter or longer for the duration subtest. If the answers are correct, the differences become smaller in small steps; if the answers are incorrect, they become larger again. In this procedure, which automatically adapts to the performance of the tested subjects with increasing difficulty, the individual threshold values are finally calculated based on the convergence behavior.</p>
</sec>
<sec id="sec15">
<title>Pitch Perception Preference Test</title>
<p>The pitch perception preference test (Pitch PP) includes 144 different pairs of harmonic complex tones. Each tone pair consisted of two consecutive harmonic complex tones (duration 500&#x2009;ms, 10-ms rise-fall time, and interstimulus interval 250&#x2009;ms). Each test tone comprised two, three, or four adjacent harmonics, leaving out the fundamental frequency. Overall, the tone pairs were designed with six different upper component frequencies (293, 523, 932, 1,661, 2,960, and 5,274&#x2009;Hz) chosen to be equidistant on a logarithmical frequency scale corresponding to the musical interval of a major ninth, beginning with D3 (293&#x2009;Hz) up to C8 (5,274&#x2009;Hz). The upper component frequency of both tones in each tone pair was identical to minimize the perception of edge pitch. All stimuli were presented binaurally in pseudorandomized order using Hammerfall DSP Multiface System with a stimulus level of 50&#x2009;dB nSL to avoid the interfering superposition of combination tones. Each tone pair was repeated once and the next tone pair presented after a pause of 2&#x2009;s. Subjects were instructed to select the predominantly perceived pitch direction or to answer according to the first, spontaneous impression. They could also indicate, if either both directions were perceived at the same time or if tones lacked a clear pitch. Test duration was 22&#x2009;min. The experimental design has been described in detail in a previous study (<xref ref-type="bibr" rid="ref61">Schneider et al., 2005</xref>).</p>
</sec>
<sec id="sec16">
<title>Singing Ability and Singing Behavior During Childhood</title>
<p>Singing ability was tested by a familiar song singing task &#x201C;Happy Birthday,&#x201D; which is usually targeted at non-professionals (<xref ref-type="bibr" rid="ref27">Dalla Bella et al., 2007</xref>; <xref ref-type="bibr" rid="ref26">Dalla Bella and Berkowska, 2009</xref>). This approach has been used in several previous investigations (<xref ref-type="bibr" rid="ref20">Christiner and Reiterer, 2013</xref>; <xref ref-type="bibr" rid="ref24">Christiner et al., 2018</xref>, <xref ref-type="bibr" rid="ref19">2021</xref>; <xref ref-type="bibr" rid="ref17">Christiner, 2020</xref>). The participants were introduced to sing &#x201C;Happy Birthday&#x201D; as best as they could and to use a key which they found pleasurably for their own singing voice. The singing performances of the participants were rated and evaluated by singing experts (two male and two female raters), a test design which has successfully been used and tested in previous studies (<xref ref-type="bibr" rid="ref15">Christiner, 2013</xref>; <xref ref-type="bibr" rid="ref20">Christiner and Reiterer, 2013</xref>). The raters were introduced to their tasks and the rating criteria: melodic and rhythmic ability. The two criteria were collapsed into a single score (S total). For the interrater reliability, intraclass correlation coefficients were calculated and it was found that the ratings were reliable (see <xref ref-type="supplementary-material" rid="SM1">Supplementary Table S2</xref>).</p>
<p>We also used a multi-item scale concept, which asked for the participants&#x2019; singing behavior during childhood (S childhood) and reported how many hours the participants sang per week on average (S hours). To make sure that the participants referred to the same time period, they received further instructions. It was explained that their childhood meant before the age of 11&#x2009;years since it has been suggested that the singing voice reaches around two octaves at the age of 10 (<xref ref-type="bibr" rid="ref69">Welch et al., 2011</xref>), which is similar for adults without vocal training (<xref ref-type="bibr" rid="ref17">Christiner, 2020</xref>). The concept consists of eight questions and has already been used previously (<xref ref-type="bibr" rid="ref17">Christiner, 2020</xref>). The internal consistency of the concept was found to be reliable. Cronbach&#x2019;s <italic>&#x03B1;</italic>&#x2009;=&#x2009;0.79 for the eight questions. The single questions and the reliability analysis are contained in the supplement (see <xref ref-type="supplementary-material" rid="SM1">Supplementary Table S3</xref>).</p>
</sec>
<sec id="sec17">
<title>Short-Term Memory</title>
<p>In order to test the STM capacity of the participants, the Wechsler Digit Span (<xref ref-type="bibr" rid="ref68">Wechsler, 1939</xref>) was used. This measurement consists of a backward digit span (STMB) and a forward digit span (STMF) subtest. The test was programmed online, and the stimulus was presented acoustically. The participants had to repeat a steadily increasing sequence of digits in either a forward or a backward order. The sequences of digits varied between 3 and 9 digits for the forward span and between 2 and 8 digits for the backward span subtests. Participants received two scores, one for the forward task (STMF) and one for the backward task (STMB). The score corresponded to the number of items they were able to correctly repeat, the maximum being 14. Based on previous research, mean values for the forward span usually range between values of 7 and 8 for the forward span for adult participants, while the mean values for the backward span are usually around one point lower than that of the forward span (<xref ref-type="bibr" rid="ref17">Christiner, 2020</xref>; <xref ref-type="bibr" rid="ref19">Christiner et al., 2021</xref>).</p>
</sec>
</sec>
<sec id="sec18" sec-type="results">
<title>Results</title>
<sec id="sec19">
<title>Statistical Analysis</title>
<p>We ran correlational and regression analyses for the musical variables and the specific Mandarin measurements. In addition, we performed a MANOVA with musical status (non-musicians, amateurs, and professional musicians) as a fixed factor and the three measures of Mandarin as dependent variables. As a follow-up, we performed discriminant analyses to uncover whether the Mandarin measures differentiated between our groups. Since we were also looking at whether STM influence the relationship between Mandarin and our musical as well as basic auditory skills measures, we performed partial correlations for the variables which were correlated with Mandarin performance and STM. These are singing ability and Frequency (for examination, please consult the correlation&#x2019;s <xref rid="tab1" ref-type="table">Table 1</xref>). In addition, we performed an ANCOVA in which Mandarin total was the dependent variable, musical status the fixed factor, and STM (the forward span) the covariate. First, <xref rid="tab2" ref-type="table">Table 2</xref> below illustrates the descriptive statistics.</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption><p>Correlational analysis outlines the correlations between the variables under consideration.</p></caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th/>
<th align="center" valign="top">Mandarin S</th>
<th align="center" valign="top">Mandarin D</th>
<th align="center" valign="top">Mandarin P</th>
<th align="center" valign="top">S total</th>
<th align="center" valign="top">S hours</th>
<th align="center" valign="top">S childhood</th>
<th align="center" valign="top">Duration</th>
<th align="center" valign="top">Frequency</th>
<th align="center" valign="top">Pitch PP</th>
<th align="center" valign="top">STMF</th>
<th align="center" valign="top">STMB</th>
<th align="center" valign="top">AMMA T</th>
<th align="center" valign="top">AMMA R</th>
<th align="center" valign="top">Musical status</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Mandarin total</td>
<td align="left" valign="top">0.688<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
<td align="left" valign="top">0.708<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
<td align="left" valign="top">0.714<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
<td align="left" valign="top">0.387<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
<td align="left" valign="top">0.036</td>
<td align="left" valign="top">0.044</td>
<td align="left" valign="top">&#x2212;0.147</td>
<td align="left" valign="top">&#x2212;0.320<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
<td align="left" valign="top">&#x2212;0.352<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
<td align="left" valign="top">&#x2212;0.398<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
<td align="left" valign="top">0.167</td>
<td align="left" valign="top">0.178</td>
<td align="left" valign="top">0.053</td>
<td align="left" valign="top">0.363<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
</tr>
<tr>
<td align="left" valign="top">Mandarin S</td>
<td/>
<td align="left" valign="top">0.250<xref rid="tfn1" ref-type="table-fn"><sup>&#x002A;</sup></xref></td>
<td align="left" valign="top">0.412<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
<td align="left" valign="top">0.376<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
<td align="left" valign="top">0.060</td>
<td align="left" valign="top">0.061</td>
<td align="left" valign="top">0.000</td>
<td align="left" valign="top">&#x2212;0.277<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
<td align="left" valign="top">&#x2212;0.197<xref rid="tfn1" ref-type="table-fn"><sup>&#x002A;</sup></xref></td>
<td align="left" valign="top">0.209<xref rid="tfn1" ref-type="table-fn"><sup>&#x002A;</sup></xref></td>
<td align="left" valign="top">0.066</td>
<td align="left" valign="top">0.136</td>
<td align="left" valign="top">0.103</td>
<td align="left" valign="top">0.372<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
</tr>
<tr>
<td align="left" valign="top">Mandarin D</td>
<td/>
<td/>
<td align="left" valign="top">0.344<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
<td align="left" valign="top">0.234<xref rid="tfn1" ref-type="table-fn"><sup>&#x002A;</sup></xref></td>
<td align="left" valign="top">0.001</td>
<td align="left" valign="top">&#x2212;0.034</td>
<td align="left" valign="top">&#x2212;0.100</td>
<td align="left" valign="top">&#x2212;0.227<xref rid="tfn1" ref-type="table-fn"><sup>&#x002A;</sup></xref></td>
<td align="left" valign="top">&#x2212;0.283<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
<td align="left" valign="top">&#x2212;0.388<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
<td align="left" valign="top">0.273<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
<td align="left" valign="top">0.215<xref rid="tfn1" ref-type="table-fn"><sup>&#x002A;</sup></xref></td>
<td align="left" valign="top">0.090</td>
<td align="left" valign="top">0.313<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
</tr>
<tr>
<td align="left" valign="top">Mandarin P</td>
<td/>
<td/>
<td/>
<td align="left" valign="top">0.239<xref rid="tfn1" ref-type="table-fn"><sup>&#x002A;</sup></xref></td>
<td align="left" valign="top">0.001</td>
<td align="left" valign="top">0.269<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
<td align="left" valign="top">&#x2212;0.159</td>
<td align="left" valign="top">&#x2212;0.238<xref rid="tfn1" ref-type="table-fn"><sup>&#x002A;</sup></xref></td>
<td align="left" valign="top">&#x2212;0.378<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
<td align="left" valign="top">0.301<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
<td align="left" valign="top">0.028</td>
<td align="left" valign="top">0.051</td>
<td align="left" valign="top">0.003</td>
<td align="left" valign="top">0.307<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
</tr>
<tr>
<td align="left" valign="top">S total</td>
<td/>
<td/>
<td/>
<td/>
<td align="left" valign="top">0.277<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
<td align="left" valign="top">0.302<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
<td align="left" valign="top">&#x2212;0.074</td>
<td align="left" valign="top">&#x2212;0.320<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
<td align="left" valign="top">&#x2212;0.206<xref rid="tfn1" ref-type="table-fn"><sup>&#x002A;</sup></xref></td>
<td align="left" valign="top">0.310<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
<td align="left" valign="top">0.260<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
<td align="left" valign="top">0.250<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
<td align="left" valign="top">0.225<xref rid="tfn1" ref-type="table-fn"><sup>&#x002A;</sup></xref></td>
<td align="left" valign="top">0.429<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
</tr>
<tr>
<td align="left" valign="top">S hours</td>
<td/>
<td/>
<td/>
<td/>
<td/>
<td align="left" valign="top">0.384<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
<td align="left" valign="top">0.013</td>
<td align="left" valign="top">0.150</td>
<td align="left" valign="top">&#x2212;0.013</td>
<td align="left" valign="top">0.055</td>
<td align="left" valign="top">&#x2212;0.054</td>
<td align="left" valign="top">&#x2212;0.031</td>
<td align="left" valign="top">&#x2212;0.004</td>
<td align="left" valign="top">&#x2212;0.017</td>
</tr>
<tr>
<td align="left" valign="top">S childhood</td>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td align="left" valign="top">0.057</td>
<td align="left" valign="top">0.091</td>
<td align="left" valign="top">&#x2212;0.044</td>
<td align="left" valign="top">&#x2212;0.126</td>
<td align="left" valign="top">0.259<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
<td align="left" valign="top">&#x2212;0.060</td>
<td align="left" valign="top">&#x2212;0.100</td>
<td align="left" valign="top">0.141</td>
</tr>
<tr>
<td align="left" valign="top">Duration</td>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td align="left" valign="top">0.148</td>
<td align="left" valign="top">0.138</td>
<td align="left" valign="top">&#x2212;0.120</td>
<td align="left" valign="top">&#x2212;0.086</td>
<td align="left" valign="top">&#x2212;0.113</td>
<td align="left" valign="top">&#x2212;0.090</td>
<td align="left" valign="top">&#x2212;0.133</td>
</tr>
<tr>
<td align="left" valign="top">Frequency</td>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td align="left" valign="top">0.090</td>
<td align="left" valign="top">&#x2212;0.198<xref rid="tfn1" ref-type="table-fn"><sup>&#x002A;</sup></xref></td>
<td align="left" valign="top">&#x2212;0.072</td>
<td align="left" valign="top">&#x2212;0.185</td>
<td align="left" valign="top">&#x2212;0.180</td>
<td align="left" valign="top">0.311<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
</tr>
<tr>
<td align="left" valign="top">Pitch PP</td>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td align="left" valign="top">&#x2212;0.187</td>
<td align="left" valign="top">&#x2212;0.106</td>
<td align="left" valign="top">&#x2212;0.028</td>
<td align="left" valign="top">&#x2212;0.014</td>
<td align="left" valign="top">&#x2212;0.207<xref rid="tfn1" ref-type="table-fn"><sup>&#x002A;</sup></xref></td>
</tr>
<tr>
<td align="left" valign="top">STMF</td>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td align="left" valign="top">0.575<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
<td align="left" valign="top">0.268<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
<td align="left" valign="top">0.241<xref rid="tfn1" ref-type="table-fn"><sup>&#x002A;</sup></xref></td>
<td align="left" valign="top">0.223<xref rid="tfn1" ref-type="table-fn"><sup>&#x002A;</sup></xref></td>
</tr>
<tr>
<td align="left" valign="top">STMB</td>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td align="left" valign="top">0.328<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
<td align="left" valign="top">0.348<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
<td align="left" valign="top">0.106</td>
</tr>
<tr>
<td align="left" valign="top">AMMA T</td>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td align="left" valign="top">0.761<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
<td align="left" valign="top">0.291<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
</tr>
<tr>
<td align="left" valign="top">AMMA R</td>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td align="left" valign="top">0.266<xref rid="tfn2" ref-type="table-fn"><sup>&#x002A;&#x002A;</sup></xref></td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>The following description explains the acronym definitions of the variables. Mandarin S, syllable tone discrimination task; Mandarin D, tone discrimination task; Mandarin P, pronunciation task; S total, melodic and rhythmic singing ability total; S hours, singing hours per week; S childhood, singing behavior during childhood; Duration, primary auditory threshold test&#x2014;Subtest Duration; Frequency, primary auditory threshold test&#x2014;Subtest Frequency; Pitch PP, pitch perception preference test; STMF, short-term memory forwards; STMB, short-term memory backwards; AMMA R, Advanced Measures of Music Audiation&#x2014;rhythmic score; and AMMA T, Advanced Measures of Music Audiation&#x2014;tonal score.</p>
<fn id="tfn1">
<label>&#x002A;</label>
<p>Means that <italic>p</italic>&#x2009;&#x003C;&#x2009;0.05 (uncorrected, two-tailed).</p>
</fn>
<fn id="tfn2">
<label>&#x002A;&#x002A;</label>
<p>Indicates that <italic>p</italic>&#x2009;&#x003C;&#x2009;0.001 (uncorrected, two-tailed).</p>
</fn>
</table-wrap-foot>
</table-wrap>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption><p>Descriptive statistics provide the descriptives of the variables under consideration.</p></caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Variables</th>
<th align="center" valign="top">Mean (<italic>M</italic>)</th>
<th align="center" valign="top">Standard Error (<italic>SE</italic>)</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Mandarin total<xref rid="tfn3" ref-type="table-fn"><sup>&#x002A;</sup></xref></td>
<td align="left" valign="top">0.03</td>
<td align="left" valign="top">0.07</td>
</tr>
<tr>
<td align="left" valign="top">Mandarin S</td>
<td align="left" valign="top">4.65</td>
<td align="left" valign="top">0.19</td>
</tr>
<tr>
<td align="left" valign="top">Mandarin D</td>
<td align="left" valign="top">13.20</td>
<td align="left" valign="top">0.21</td>
</tr>
<tr>
<td align="left" valign="top">Mandarin P</td>
<td align="left" valign="top">3.23</td>
<td align="left" valign="top">0.13</td>
</tr>
<tr>
<td align="left" valign="top">S total</td>
<td align="left" valign="top">5.99</td>
<td align="left" valign="top">0.12</td>
</tr>
<tr>
<td align="left" valign="top">S&#x2009;hours</td>
<td align="left" valign="top">1.89</td>
<td align="left" valign="top">0.25</td>
</tr>
<tr>
<td align="left" valign="top">S childhood</td>
<td align="left" valign="top">37.71</td>
<td align="left" valign="top">1.51</td>
</tr>
<tr>
<td align="left" valign="top">Duration</td>
<td align="left" valign="top">44.03</td>
<td align="left" valign="top">2.09</td>
</tr>
<tr>
<td align="left" valign="top">Frequency</td>
<td align="left" valign="top">26.40</td>
<td align="left" valign="top">1.93</td>
</tr>
<tr>
<td align="left" valign="top">Pitch PP</td>
<td align="left" valign="top">35.24</td>
<td align="left" valign="top">2.62</td>
</tr>
<tr>
<td align="left" valign="top">STMF</td>
<td align="left" valign="top">7.19</td>
<td align="left" valign="top">0.19</td>
</tr>
<tr>
<td align="left" valign="top">STMB</td>
<td align="left" valign="top">6.58</td>
<td align="left" valign="top">0.21</td>
</tr>
<tr>
<td align="left" valign="top">AMMA T</td>
<td align="left" valign="top">24.25</td>
<td align="left" valign="top">0.39</td>
</tr>
<tr>
<td align="left" valign="top">AMMA R</td>
<td align="left" valign="top">26.57</td>
<td align="left" valign="top">0.41</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>The following description explains the acronym definitions of the variables. Mandarin S, syllable tone discrimination task; Mandarin D, tone discrimination task; Mandarin P, pronunciation task; S total, melodic and rhythmic singing ability total; S hours, singing hours per week; S childhood, singing behavior during childhood; Duration, primary auditory threshold test&#x2014;Subtest Duration; Frequency, primary auditory threshold test&#x2013;Subtest Frequency; Pitch PP, pitch perception preference test; STMF, short-term memory forwards; STMB, short-term memory backwards; AMMA R, Advanced Measures of Music Audiation&#x2014;rhythmic score; and AMMA T, Advanced Measures of Music Audiation&#x2014;tonal score.</p>
<fn id="tfn3">
<label>&#x002A;</label>
<p>Note that the Mandarin total is comprised of all three Mandarin tasks. Since the Mandarin subtests are based on different scorings, they were <italic>z</italic>-transformed before they were collapsed into a single score.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="sec20">
<title>Correlational Analysis</title>
<p>A Pearson correlational analysis was applied for the individual variables to illustrate the relationships between the Mandarin variables, STM, and the musicality measures. <xref rid="tab1" ref-type="table">Table 1</xref> illustrates the relationships between the variables.</p>
</sec>
<sec id="sec21">
<title>Regression Model for the Criterion Variable the &#x201C;Mandarin Total&#x201D;</title>
<p>Based on the findings of the correlations, we performed regression models. Before, we were assessing whether the dependent variable was normally distributed, and we therefore applied a Shapiro&#x2013;Wilk test. Results have shown the Mandarin total score was normally distributed <italic>p</italic>&#x2009;=&#x2009;0.86. The independent predictor variables were entered in the multiple linear regression models only if a probability of <italic>F</italic>-change&#x2009;&#x003C;&#x2009;0.05 was given. We used a stepwise method where the ordering of the variables is based on purely mathematical decisions. The findings revealed that four predictors, STMF, singing ability, pitch perception preference, and tone frequency could explain 33 % of the variances in the Mandarin total performance, which consists of all three tasks. <xref rid="tab3" ref-type="table">Table 3</xref> shows the results of the multiple regression. Note that we also performed multiple regressions for the three individual Mandarin tasks. The findings are provided in the <xref ref-type="supplementary-material" rid="SM1">Supplementary Table S4</xref>.</p>
<table-wrap position="float" id="tab3">
<label>Table 3</label>
<caption><p>Multiple regression models explaining the variance in Mandarin total.</p></caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Predictor</th>
<th align="center" valign="top">Partial correlation (pr)</th>
<th align="center" valign="top"><italic>p</italic>-value</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Step 1: <italic>R</italic>&#x2009;=&#x2009;0.41, <italic>F</italic>(1, 105)&#x2009;=&#x2009;21.48, <italic>p</italic>&#x2009;&#x003C;&#x2009;0.001</td>
<td/>
<td/>
</tr>
<tr>
<td align="left" valign="top">STMF</td>
<td align="left" valign="top">0.41</td>
<td align="left" valign="top">&#x003C; 0.001</td>
</tr>
<tr>
<td align="left" valign="top">Step 2: <italic>R</italic>&#x2009;=&#x2009;0.49, <italic>F</italic>(1, 104)&#x2009;=&#x2009;10.66, <italic>p</italic>&#x2009;&#x003C;&#x2009;0.001</td>
<td/>
<td/>
</tr>
<tr>
<td align="left" valign="top">STMF</td>
<td align="left" valign="top">0.33</td>
<td align="left" valign="top">&#x003C; 0.001</td>
</tr>
<tr>
<td align="left" valign="top">S total</td>
<td align="left" valign="top">0.31</td>
<td align="left" valign="top">&#x003C; 0.001</td>
</tr>
<tr>
<td align="left" valign="top">Step 3: <italic>R</italic>&#x2009;=&#x2009;0.55, <italic>F</italic>(1, 103)&#x2009;=&#x2009;8.19, <italic>p</italic>&#x2009;=&#x2009;0.005</td>
<td/>
<td/>
</tr>
<tr>
<td align="left" valign="top">STMF</td>
<td align="left" valign="top">0.31</td>
<td align="left" valign="top">&#x003C; 0.001</td>
</tr>
<tr>
<td align="left" valign="top">S total</td>
<td align="left" valign="top">0.27</td>
<td align="left" valign="top">0.005</td>
</tr>
<tr>
<td align="left" valign="top">Pitch PP</td>
<td align="left" valign="top">&#x2212;0.27</td>
<td align="left" valign="top">0.005</td>
</tr>
<tr>
<td align="left" valign="top">Step 4: <italic>R</italic>&#x2009;=&#x2009;0.58, <italic>F</italic>(1, 102)&#x2009;=&#x2009;4.43, <italic>p</italic>&#x2009;=&#x2009;0.038</td>
<td/>
<td/>
</tr>
<tr>
<td align="left" valign="top">STMF</td>
<td align="left" valign="top">0.29</td>
<td align="left" valign="top">&#x003C; 0.002</td>
</tr>
<tr>
<td align="left" valign="top">S total</td>
<td align="left" valign="top">0.22</td>
<td align="left" valign="top">0.028</td>
</tr>
<tr>
<td align="left" valign="top">Pitch PP</td>
<td align="left" valign="top">&#x2212;0.28</td>
<td align="left" valign="top">0.005</td>
</tr>
<tr>
<td align="left" valign="top">Frequency</td>
<td align="left" valign="top">&#x2212;0.20</td>
<td align="left" valign="top">0.038</td>
</tr>
<tr>
<td align="left" valign="top"><italic>Dependent variable: Mandarin total</italic></td>
<td/>
<td/>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>This table illustrates the stepwise multiple regression models. Mandarin total is the dependent variable. The ordering of the variables is based on mathematical decisions. The following description explains the acronym definitions of the variables. STMF, short-term memory forwards; S total, melodic and rhythmic singing ability total; Pitch PP, pitch perception preference test; and Frequency, primary auditory threshold test&#x2014;Subtest Frequency.</p>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="sec22">
<title>MANOVA for Variables &#x201C;Musical Status&#x201D; and &#x201C;Mandarin Performance&#x201D;</title>
<p>In order to find out, whether the Mandarin performances of our participants differed in their mean values according to their musical status, we performed a MANOVA. Using Pillai&#x2019;s trace, there was a significant effect of musical status and Mandarin performance <italic>V</italic>&#x2009;=&#x2009;0.257, <italic>F</italic>(6,192)&#x2009;=&#x2009;4.72, <italic>p</italic>&#x2009;&#x003C;&#x2009;0.001.</p>
</sec>
<sec id="sec23">
<title>Discriminant Analysis</title>
<p>The MANOVA was followed by a discriminant analysis, which revealed two discriminant functions. The first explained 85.4% of the variance, canonical <italic>R</italic><sup>2</sup>&#x2009;=&#x2009;0.31, whereas the second explained only 14.6%, canonical <italic>R</italic><sup>2</sup>&#x2009;=&#x2009;0.04. In combination, these discriminant functions significantly discriminated the groups, &#x039B;&#x2009;=&#x2009;0.75, <italic>&#x03C7;</italic><sup>2</sup>(6)&#x2009;=&#x2009;27.32, <italic>p</italic>&#x2009;&#x003C;&#x2009;0.001, but removing the first function indicated that the second function did not significantly differentiate the three groups &#x039B;&#x2009;=&#x2009;0.96, <italic>&#x03C7;</italic><sup>2</sup>(2)&#x2009;=&#x2009;2.31, <italic>p</italic>&#x2009;=&#x2009;0.51. The correlations between the considered variables and the discriminant functions revealed that the loads onto the first function were rather high for all three Mandarin variables, Mandarin S (<italic>r</italic>&#x2009;=&#x2009;0.86), Mandarin P (<italic>r</italic>&#x2009;=&#x2009;0.64), and Mandarin D (<italic>r</italic>&#x2009;=&#x2009;0.59). When a cutoff of 0.40 was used to decide which of the standardized discriminant coefficients were large, all three Mandarin variables separated the musicians from the amateurs and the non-musicians well. The musicians had higher scores in Mandarin S (<italic>M</italic>&#x2009;=&#x2009;5.69), Mandarin P (<italic>M</italic>&#x2009;=&#x2009;3.70), and Mandarin D (<italic>M</italic>&#x2009;=&#x2009;13.93), than the amateurs Mandarin S (<italic>M</italic>&#x2009;=&#x2009;3.83), Mandarin P (<italic>M</italic>&#x2009;=&#x2009;3.19), and Mandarin D (<italic>M</italic>&#x2009;=&#x2009;13.07), and non-musicians Mandarin S (<italic>M</italic>&#x2009;=&#x2009;4.00), Mandarin P (<italic>M</italic>&#x2009;=&#x2009;2.69), and Mandarin D (<italic>M</italic>&#x2009;=&#x2009;12.38). <xref rid="fig2" ref-type="fig">Figure 2</xref> below illustrates the discriminant plot.</p>
<fig position="float" id="fig2">
<label>Figure 2</label>
<caption><p>The figure illustrates the discriminant function of the three Mandarin measures. The horizontal axis represents the significant discriminant function, function 1. It discriminates the professional musicians from the amateurs and the non-musicians. The correlations between the outcome variables and the discriminant functions revealed that all three Mandarin variables load onto the first function illustrating that the professional musicians performed significantly better in Mandarin than both other groups. The vertical axis represents the second-discriminant function, function 2. Since this function was non-significant it will not be further discussed.</p></caption>
<graphic xlink:href="fpsyg-13-895063-g002.tif"/>
</fig>
</sec>
<sec id="sec24">
<title>Partial Correlations</title>
<p>Short-term memory, in particular the forward digit span, turned out to be an important predictor for Mandarin performance in the regression model. As introduced in section 1.3.4, forward digit spans and language measures require similar cognitive ability which is why it could be assumed that language measures and digit spans measure very similar concepts. Therefore, we ran also partial correlations for the variables which correlated with Mandarin and one of our STM ability measures. These variables were singing ability which correlated with Mandarin total and both the STMF and STMB measures as well as Frequency which correlated with Mandarin and STMF capacity.</p>
<p>The partial correlations reveal that when STM forward on the relationship between the singing ability and Mandarin total is controlled <italic>r</italic>&#x2009;=&#x2009;0.30, <italic>p</italic> (two-tailed)&#x2009;&#x003C;&#x2009;0.001, their relationship diminishes but remains significant. The partial correlations reveal that when STM backward on the relationship between the singing ability and Mandarin total is controlled <italic>r</italic>&#x2009;=&#x2009;0.36, <italic>p</italic> (two-tailed)&#x2009;&#x003C;&#x2009;0.001, their relationship diminishes but remains significant. The partial correlations reveal that when STM forward on the relationship between the Frequency and Mandarin total is controlled <italic>r</italic>&#x2009;=&#x2009;&#x2212;0.27, <italic>p</italic> (two-tailed)&#x2009;&#x003C;&#x2009;0.006, their relationship diminishes but is remains significant.</p>
</sec>
<sec id="sec25">
<title>ANCOVA</title>
<p>We also performed an ANCOVA in which Mandarin total was the dependent variable, musical status, the fixed factor, and STM (the forward span) as the covariate. The ANCOVA revealed that the covariate, STM, was significantly related to the Mandarin total performance <italic>F</italic>(1,105)&#x2009;=&#x2009;15.98, <italic>p</italic>&#x2009;&#x003C;&#x2009;0.001, <italic>r</italic>&#x2009;=&#x2009;0.36. There was also a significant effect which could be detected for the musical status and the Mandarin performances after controlling for STM capacity <italic>F</italic>(2,105)&#x2009;=&#x2009;3.72, <italic>p</italic>&#x2009;&#x003C;&#x2009;0.028, partial <italic>&#x03B7;<sup>2</sup></italic>&#x2009;=&#x2009;0.07.</p>
</sec>
</sec>
<sec id="sec26" sec-type="discussions">
<title>Discussion</title>
<p>We performed several different statistical analyses in order to outline the relationships between Mandarin ability and musical ability. Correlational analysis has shown that all three Mandarin subtests were related to STM ability, tone frequency, pitch perception preference, singing ability, and musical status. A regression analysis revealed that the variance of the Mandarin total performance could be explained by STM capacity, singing ability, pitch perception preference, and tone frequency. In addition, we performed a MANOVA followed by a discriminant analysis, which revealed that professional musicians were better than amateurs and non-musicians at all Mandarin conditions. Finally, we also assessed whether STM ability influences the relationship between the musical variables and the Mandarin performances and the basic auditory skills and STM capacity. Results revealed that both STM capacity and musical skills/status contribute to explaining individual differences in Mandarin performances.</p>
<sec id="sec27">
<title>Short-Term Memory</title>
<p>The STM capacity as measured <italic>via</italic> the forward digit span was correlated to all Mandarin measures and turned out to predict the Mandarin total performance as shown in the regression model. The backward digit span was only correlated with performance in the Mandarin tone discrimination task. The finding that the forward span was a better predictor for the language tasks than the backward span has also been found in previous research (<xref ref-type="bibr" rid="ref17">Christiner, 2020</xref>; <xref ref-type="bibr" rid="ref19">Christiner et al., 2021</xref>). In general, we expected STM capacity to be one of the most important predictors for explaining individual differences in Mandarin performance. STM capacity is one of the most important predictors which explains individual differences in speech production, reading ability, foreign language comprehension, or vocabulary learning (<xref ref-type="bibr" rid="ref35">Gathercole and Baddeley, 1990</xref>, <xref ref-type="bibr" rid="ref36">1993</xref>) and therefore has been associated with foreign language success (<xref ref-type="bibr" rid="ref29">D&#x00F6;rnyei, 2005</xref>; <xref ref-type="bibr" rid="ref71">Wen and Skehan, 2011</xref>; <xref ref-type="bibr" rid="ref70">Wen et al., 2017</xref>). In previous research, we used mainly pronunciation tasks which are rather similar to non-word spans (<xref ref-type="bibr" rid="ref17">Christiner, 2020</xref>; <xref ref-type="bibr" rid="ref19">Christiner et al., 2021</xref>, <xref ref-type="bibr" rid="ref18">2022</xref>) which are why we expected that a positive relationship between STM capacity and Mandarin P will be found. Mandarin S and Mandarin D are measures, which were newly developed and used the first time in our studies. Mandarin D is a discrimination task in which tonal changes of an equally worded second statement need to be detected&#x2014;a capacity which also requires STM capacity and is also very similar to measures of basic auditory skills (e.g., tone frequency) and musical aptitude (e.g., AMMA T). The latter both also correlated to the STM measures in this study. Mandarin S can also be seen as a task which focuses on phonetic coding ability. This has been associated with the ability to identify distinct sounds and to form associations between these sounds (<xref ref-type="bibr" rid="ref8">Carroll, 1981</xref>). The main difference of our Mandarin S measure to already established and approved measures of phonetic coding ability such as the, the <italic>Phonetic Script</italic> and the <italic>Spelling Cues</italic> of the <italic>Modern Language Aptitude Test</italic> by <xref ref-type="bibr" rid="ref9">Carroll and Sapon (1959)</xref> is that our participants had additionally to indicate a tonal change. Relationships between STM ability and coding ability have also been reported in multiple studies (e.g., <xref ref-type="bibr" rid="ref6">Brady, 1986</xref>; <xref ref-type="bibr" rid="ref40">Holtz, 1993</xref>).</p>
<p>However, STM capacity was also correlated with the musical aptitude measures and singing ability which suggests that STM capacity shows overlaps between tonal and verbal material. In early research, this notion was questioned, and it was speculated that there may be different storage components for tonal and speech material (<xref ref-type="bibr" rid="ref60">Salam&#x00E9; and Baddeley, 1989</xref>). In contrast, more recently findings from brain research suggest that verbal and tonal storage and rehearsal abilities largely rely on overlapping neuronal networks (<xref ref-type="bibr" rid="ref44">Koelsch et al., 2009</xref>). The findings of <xref ref-type="bibr" rid="ref44">Koelsch et al. (2009)</xref> suggest overlapping or shared storage components for verbal and tonal material, which suggests that STM capacity is crucial for explaining individual differences in the performance of both music and language tasks. Since some of music variables were also related to STM capacity, we assessed whether the relationship between musical and language variables was influenced by STM capacity. Therefore, we performed partial correlations and an ANCOVA where we treated the forward STM capacity as a covariate. Results have shown that the relationship between singing ability and Mandarin performance as well as Frequency and Mandarin total performance slightly diminish when controlled for STM capacity but still remain significant. Similar results provided the ANCOVA which has shown that a significant effect could be detected for the musical status and the Mandarin performances after controlling for STM capacity. This suggests that both STM capacity and musical ability contribute to explaining individual differences in Mandarin performances.</p>
</sec>
<sec id="sec28">
<title>Production: Singing Ability and Behavior During Childhood</title>
<p>Singing ability is also one of the most important predictors to explain the variability in the Mandarin performance in this investigation. Singing has been found to improve the ability to memorize new vocabulary (<xref ref-type="bibr" rid="ref50">Ludke et al., 2014</xref>; <xref ref-type="bibr" rid="ref32">Franco et al., 2021</xref>) and the ability to imitate unfamiliar languages (<xref ref-type="bibr" rid="ref22">Christiner and Reiterer, 2018</xref>; <xref ref-type="bibr" rid="ref17">Christiner, 2020</xref>; <xref ref-type="bibr" rid="ref19">Christiner et al., 2021</xref>) in children, adolescents, and adults. In this respect, two crucial elements should be discussed: melody as mnemonic and vocal-motor ability. Enhanced vocal-motor ability is required for an elaborate singing capacity, which has been identified as a predictor of individual differences in foreign language pronunciation for several times (<xref ref-type="bibr" rid="ref20">Christiner and Reiterer, 2013</xref>). Singing behavior during childhood was also correlated with Mandarin pronunciation&#x2014;a finding which has already been observed in our previous research (<xref ref-type="bibr" rid="ref17">Christiner, 2020</xref>; <xref ref-type="bibr" rid="ref18">Christiner et al., 2022</xref>). Singing ability as well as the amount of singing during childhood seems to enhance vocal-motor skills, sensorimotor ability, and vocal flexibility, which may be the link between singing ability and language pronunciation. Research has also shown that singing during childhood influences the ability to acquire foreign language pronunciation later in adulthood, while the same has not been found for language perception (<xref ref-type="bibr" rid="ref18">Christiner et al., 2022</xref>).</p>
<p>While it has been expected that Mandarin pronunciation will be related to singing ability, the finding that the two other Mandarin tasks also correlated with singing was more surprising. In this respect, the second parameter &#x201C;melody as mnemonic&#x201D; should be discussed. Melody has also been ascribed to play a key role in language acquisition processes. Infants and adults do acquire new utterances much faster when they are sung (<xref ref-type="bibr" rid="ref65">Thiessen and Saffran, 2009</xref>; <xref ref-type="bibr" rid="ref50">Ludke et al., 2014</xref>). This may be the case since melody seems to serve as mnemonic with which new utterances are probably stored in the long-term memory (<xref ref-type="bibr" rid="ref37">Gordon et al., 2010</xref>). Research has also noted that languages which appear to be more song-like or melodic are also better retrieved (<xref ref-type="bibr" rid="ref51">Margulis et al., 2015</xref>; <xref ref-type="bibr" rid="ref19">Christiner et al., 2021</xref>). Therefore, it may be assumed that individuals who perceive Mandarin to be more melodic or song-like may also perform better at all Mandarin tasks. We suggest that the singing benefit is based on enhanced sensorimotor ability and melody as mnemonic (<xref ref-type="bibr" rid="ref18">Christiner et al., 2022</xref>).</p>
</sec>
<sec id="sec29">
<title>Perception</title>
<p>To recap, as opposed to non-tone language speakers, tone language speakers possess enhanced pitch memory, show higher tonal ability, and possess enhanced pitch processing ability (<xref ref-type="bibr" rid="ref5">Bidelman et al., 2013</xref>; <xref ref-type="bibr" rid="ref53">Nan et al., 2018</xref>; <xref ref-type="bibr" rid="ref38">Gottfried, 2019</xref>). Tone frequency is one of the most important predictors to explain individual differences in the Mandarin ability of our tasks. Being able to differentiate the four Mandarin tones, the &#x201C;high level tone,&#x201D; the &#x201C;(high) rising tone,&#x201D; the &#x201C;falling-rising tone,&#x201D; and the &#x201C;falling tone&#x201D; requires the precise ability to discriminate low from high tones. Since our tasks consist of equally worded items, which in the different condition contained a single tonal change, we expected that tone frequency will turn out to be one of the most important predictors, while short vs. long tones (duration) played a minor role. Previous research has also outlined that individuals with better tonal ability performed better in Mandarin discrimination tasks (<xref ref-type="bibr" rid="ref38">Gottfried, 2019</xref>). Listeners were exposed to Mandarin words that had either different or the same tones&#x2014;tasks which were very similar to our Mandarin D measure. Results showed that individuals with higher musical ability and participants with professional musical background performed more accurately than non-musicians (<xref ref-type="bibr" rid="ref38">Gottfried, 2019</xref>). Other researchers presented similar findings and outlined that piano playing enhances neural processing of pitch which in turn improved Mandarin word discrimination ability (<xref ref-type="bibr" rid="ref53">Nan et al., 2018</xref>).</p>
<p>Studies have shown that musical ability was correlated with accuracy in the performance of tone-word perception and production ability (<xref ref-type="bibr" rid="ref47">Li and DeKeyser, 2017</xref>). Interestingly, in our study, musical aptitude was less related to the Mandarin performances as only the tonal parameter correlated with the Mandarin discrimination, while the rhythmic aptitude measure was not correlated with any of the Mandarin tasks. Even though musical aptitude is generally associated with language ability, similar findings have already been observed in our previous research where eight languages were investigated. Findings revealed that only the tonal aptitude factor contributed to individual differences in foreign language capacity while the rhythmic component always failed in regression models (<xref ref-type="bibr" rid="ref17">Christiner, 2020</xref>). On the other hand, the nature of the Mandarin tasks was designed to detect single tonal changes in sequences of language material which may be one reason why tone discrimination ability, the measurement frequency (high vs. low tones), was a better predictor than musical aptitude measures.</p>
<p>Interestingly, our findings have shown that Mandarin ability was enhanced in individuals who were fundamental listeners. There are two reasonable explanations. One is that musical practice may induce a shift from spectral toward fundamental pitch perception (<xref ref-type="bibr" rid="ref64">Seither-Preisler et al., 2007</xref>). This would also be supported in the present investigation since there was also a relationship between the tendency to classify complex tones according to their fundamental pitches and musical status (see the negative correlations). Another explanation could be found in differences between fundamental pitch and spectral listeners. Whereas the former perceive the sound predominantly according to its fundamental pitch, which is a holistic feature, spectral listeners may either decompose the sound into single harmonic constituents (<xref ref-type="bibr" rid="ref62">Schneider and Wengenroth, 2009</xref>) or perceive them in terms of global timbre (<xref ref-type="bibr" rid="ref64">Seither-Preisler et al., 2007</xref>). Since the Mandarin learning requires detecting precise single tone changes it could explain why fundamental pitch listeners seem to be better equipped for acquiring tone languages.</p>
<p>The MANOVA and discriminant analysis revealed that professional musicians outperformed the amateurs and non-musicians in all Mandarin tasks. In previous investigation, we noted that beside professional musicians also amateurs performed better at the acquisition of new non-tone languages. This suggests that little musical training facilitates the learning of new non-tone languages, but higher musical skills may be required for non-tone language speakers to have an advantage in the learning of tone languages such as Mandarin.</p>
<p>The study has also limitations and the singing and pronunciation tasks were rated subjectively. Even though research has shown that subjective and objective rating scales of music performance provide similar information when aspects are carefully introduced (<xref ref-type="bibr" rid="ref45">Larrouy-Maestri et al., 2013</xref>), there is no equivalent research for language pronunciation tasks available which should be subject to future studies.</p>
</sec>
<sec id="sec30">
<title>Implications for Pedagogies</title>
<p>The findings of this study have also pedagogical implications and suggest the following aspects. We suggest that the singing benefit is based on enhanced sensorimotor ability and melody as mnemonic (<xref ref-type="bibr" rid="ref18">Christiner et al., 2022</xref>). The first being crucial to mimic and imitate new language stimuli and the second to retain and memorize utterances. However, these two aspects need to be differentiated from precision as is required for the acquisition of particular language features such as syllable tones in Mandarin. Singing has often been employed as a tool to memorize utterances and to facilitate pronunciation. However, acquiring Mandarin syllable tones demands high tonal precision which becomes neutralized and altered when Mandarin is sung. In this respect, we want to stress that singing ability has to be differentiated from educational programs which make use of singing as a tool to learn new words. We propose that the ability to sing enhances language ability. However, our findings do not indicate that Mandarin syllable tones should be learnt by singing. Instead, the acquisition of syllable tones in Mandarin may be best supported by training of basic auditory skills such as discriminating tones (e.g., high vs. low tones as the Frequency measurement) or by acquiring musical instruments. Research has provided evidence that piano playing enhances neural processing and sound perception of pitch which in turn improved Mandarin word discrimination ability (<xref ref-type="bibr" rid="ref53">Nan et al., 2018</xref>). In our study, we also detected that the professional musicians performed better than the amateurs and non-musicians in all Mandarin measures&#x2014;a finding which has also been reported recently (<xref ref-type="bibr" rid="ref38">Gottfried, 2019</xref>). In addition, research has also shown that musical practice may induce a shift from spectral toward fundamental pitch perception (<xref ref-type="bibr" rid="ref64">Seither-Preisler et al., 2007</xref>). The latter being better at Mandarin performance according to our finding. Therefore, it can be suggested that learners of Mandarin will acquire syllable tones in Mandarin more easily when they train a musical instrument or basic auditory skills (e.g., tone discrimination tasks).</p>
<p>For STM capacity, it is more difficult to give suggestions. While research has shown that complex working memory paradigms can be improved by training (<xref ref-type="bibr" rid="ref43">Klingberg et al., 2002</xref>), there is hardly any evidence for capacity changes in verbal STM after extensive practice (<xref ref-type="bibr" rid="ref54">Norris et al., 2019</xref>). Training of digit spans has not been shown to substantially improve the capacity of verbal STM (<xref ref-type="bibr" rid="ref54">Norris et al., 2019</xref>). As STM capacity also defined as a language aptitude component (<xref ref-type="bibr" rid="ref7">Cain et al., 2004</xref>; <xref ref-type="bibr" rid="ref58">Robinson, 2005</xref>, <xref ref-type="bibr" rid="ref59">2019</xref>; <xref ref-type="bibr" rid="ref30">D&#x00F6;rnyei, 2006</xref>; <xref ref-type="bibr" rid="ref71">Wen and Skehan, 2011</xref>; <xref ref-type="bibr" rid="ref16">Christiner, 2018</xref>), one explanation why STM training has been reported to have little effect on language capacity could be that STM capacity may be more of an early acquired or inherent ability. In how far, musical training improves STM capacity and in turn language functions have largely been ignored as far as we know.</p>
</sec>
</sec>
<sec id="sec31" sec-type="conclusions">
<title>Conclusion</title>
<p>The findings of this investigation have revealed that STM ability, tone frequency, pitch perception preference, singing ability, and musical status were the best predictors for explaining individual differences in Mandarin ability. While STM capacity, tone frequency, and musical status as crucial aspects for Mandarin learning have already been discussed in detail in the current literature, fewer studies have discussed the role of singing in Mandarin. Singing ability was able to explain individual differences in the variability of Mandarin performance. In addition, participants who sang more often during childhood also performed better at Mandarin pronunciation. As far as known, fundamental pitch and spectral listening have not been investigated in the context of Mandarin ability. Results have shown that fundamental pitch listeners indeed seem to be advantaged in tone language learning. Thus, our findings also have implications on educating Mandarin to non-tone language speakers. Since singing predicts individual differences in Mandarin performances, singing tasks, or focusing on melodic aspects of Mandarin may be beneficial for Mandarin learning in initial learning settings. This may facilitate Mandarin pronunciation and retrieval. In addition, we suggest on the basis of our present findings that musical training, such as tone discrimination tasks, should be part of educational programs for non-tone language speakers learning tone languages.</p>
<p>Future research should also focus on the role of STM capacity and musical ability in more detail. While STM capacity and language functions have been studied in detail, STM capacity and musical ability are underrepresented. In addition, there are studies needed which treat STM capacity as mediator between musical ability and language capacity as well as studies should focus on whether musical training improves STM capacity and in turn language ability.</p>
</sec>
<sec id="sec32" sec-type="data-availability">
<title>Data Availability Statement</title>
<p>The original contributions presented in the study are included in the article, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="sec33">
<title>Ethics Statement</title>
<p>The studies involving human participants were reviewed and approved by Medical Faculty of Heidelberg S-778/2018. The patients/participants provided their written informed consent to participate in this study.</p>
</sec>
<sec id="sec34">
<title>Author Contributions</title>
<p>MC and JR developed the Mandarin measures, contributed to the conception and design of the work, and drafted the work. MC, CG, JB, and PS were involved in the acquisition of data. AS-P and MC performed the statistical analysis. MC was responsible for finalizing the work. CG and AS-P performed a critical revision of the manuscript. All authors contributed to the article and approved the submitted version.</p>
</sec>
<sec id="sec35" sec-type="funding-information">
<title>Funding</title>
<p>MC is funded within the Post-DocTrack Program of the OeAW. Open Access Funding by the University of Graz.</p>
</sec>
<sec id="conf1" sec-type="COI-statement">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="sec38" sec-type="disclaimer">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
</body>
<back>
<ack>
<p>The authors thank L. Chan for creating <xref rid="fig1" ref-type="fig">Figure 1</xref>.</p>
</ack>
<sec id="sec37" sec-type="supplementary-material">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link xlink:href="https://www.frontiersin.org/articles/10.3389/fpsyg.2022.895063/full#supplementary-material" ext-link-type="uri">https://www.frontiersin.org/articles/10.3389/fpsyg.2022.895063/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Table_1.DOCX" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="ref1"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Alexander</surname> <given-names>J. A.</given-names></name> <name><surname>Wong</surname> <given-names>P. C. M.</given-names></name> <name><surname>Bradlow</surname> <given-names>A. R.</given-names></name></person-group> (<year>2005</year>). &#x201C;Lexical tone perception in musicians and non-musicians.&#x201D; in <italic>Interspeech 2005</italic>. ISCA; September 4&#x2013;8, 2005, 397&#x2013;400</citation></ref>
<ref id="ref2"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Baddeley</surname> <given-names>A. D.</given-names></name></person-group> (<year>2010</year>). <article-title>Working memory</article-title>. <source>Curr. Biol.</source> <volume>20</volume>, <fpage>R136</fpage>&#x2013;<lpage>R140</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.cub.2009.12.014</pub-id></citation></ref>
<ref id="ref3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Baddeley</surname> <given-names>A. D.</given-names></name> <name><surname>Lewis</surname> <given-names>V.</given-names></name> <name><surname>Vallar</surname> <given-names>G.</given-names></name></person-group> (<year>1984</year>). <article-title>Exploring the articulatory loop</article-title>. <source>Q. J. Exp. Psychol. A</source> <volume>36</volume>, <fpage>233</fpage>&#x2013;<lpage>252</lpage>. doi: <pub-id pub-id-type="doi">10.1080/14640748408402157</pub-id></citation></ref>
<ref id="ref4"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Besson</surname> <given-names>M.</given-names></name> <name><surname>Chobert</surname> <given-names>J.</given-names></name> <name><surname>Marie</surname> <given-names>C.</given-names></name></person-group> (<year>2011</year>). <article-title>Transfer of training between music and speech: common processing, attention, and memory</article-title>. <source>Front. Psychol.</source> <volume>2</volume>:<fpage>94</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fpsyg.2011.00094</pub-id>, PMID: <pub-id pub-id-type="pmid">21738519</pub-id></citation></ref>
<ref id="ref5"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bidelman</surname> <given-names>G. M.</given-names></name> <name><surname>Hutka</surname> <given-names>S.</given-names></name> <name><surname>Moreno</surname> <given-names>S.</given-names></name></person-group> (<year>2013</year>). <article-title>Tone language speakers and musicians share enhanced perceptual and cognitive abilities for musical pitch: evidence for bidirectionality between the domains of language and music</article-title>. <source>PLoS One</source> <volume>8</volume>:<fpage>e60676</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pone.0060676</pub-id>, PMID: <pub-id pub-id-type="pmid">23565267</pub-id></citation></ref>
<ref id="ref6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Brady</surname> <given-names>S.</given-names></name></person-group> (<year>1986</year>). <article-title>Short-term memory, phonological processing, and reading ability</article-title>. <source>Ann. Dyslexia</source> <volume>36</volume>, <fpage>138</fpage>&#x2013;<lpage>153</lpage>. doi: <pub-id pub-id-type="doi">10.1007/BF02648026</pub-id></citation></ref>
<ref id="ref7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cain</surname> <given-names>K.</given-names></name> <name><surname>Oakhill</surname> <given-names>J.</given-names></name> <name><surname>Bryant</surname> <given-names>P.</given-names></name></person-group> (<year>2004</year>). <article-title>Children&#x2019;s reading comprehension ability: concurrent prediction by working memory, verbal ability, and component skills</article-title>. <source>J. Educ. Psychol.</source> <volume>96</volume>, <fpage>31</fpage>&#x2013;<lpage>42</lpage>. doi: <pub-id pub-id-type="doi">10.1037/0022-0663.96.1.31</pub-id></citation></ref>
<ref id="ref8"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Carroll</surname> <given-names>J. B.</given-names></name></person-group> (<year>1981</year>). &#x201C;<article-title>Twenty-five years of research on foreign language aptitude</article-title>,&#x201D; in <source>Individual Differences and Universals in Language Learning Aptitude.</source> ed. <person-group person-group-type="editor"><name><surname>Diller</surname> <given-names>K. C.</given-names></name></person-group> (<publisher-loc>Rowley, MA</publisher-loc>: <publisher-name>Newbury House</publisher-name>), <fpage>83</fpage>&#x2013;<lpage>118</lpage>.</citation></ref>
<ref id="ref9"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Carroll</surname> <given-names>J. B.</given-names></name> <name><surname>Sapon</surname> <given-names>S.</given-names></name></person-group> (<year>1959</year>). <source>Modern Language Aptitude Test (MLAT).</source> <publisher-loc>New York, NY</publisher-loc>: <publisher-name>The Psychological Corporation.</publisher-name></citation></ref>
<ref id="ref10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chao</surname> <given-names>Y.-R.</given-names></name></person-group> (<year>1930</year>). <article-title>A system of tone letters</article-title>. <source>Le Ma&#x00EE;tre Phon&#x00E9;tique</source> <volume>45</volume>, <fpage>24</fpage>&#x2013;<lpage>27</lpage>.</citation></ref>
<ref id="ref11"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Chao</surname> <given-names>Y. R.</given-names></name></person-group> (<year>1965</year>). <source>A Grammar of Spoken Chinese.</source> <publisher-loc>Berkeley, CA</publisher-loc>: <publisher-name>University of California Press.</publisher-name></citation></ref>
<ref id="ref12"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>C.-Y.</given-names></name></person-group> (<year>1984</year>). <article-title>Neutral tone in mandarin: phonotactic description and the issue of the norm</article-title>. <source>J. Chin. Linguist.</source> <volume>2</volume>, <fpage>299</fpage>&#x2013;<lpage>333</lpage>.</citation></ref>
<ref id="ref13"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>M. Y.</given-names></name></person-group> (<year>2000</year>). <source>Tone Sandhi: Patterns Across Chinese Dialects.</source> <publisher-name>New York, UK: Cambridge</publisher-name>.</citation></ref>
<ref id="ref14"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chobert</surname> <given-names>J.</given-names></name> <name><surname>Fran&#x00E7;ois</surname> <given-names>C.</given-names></name> <name><surname>Velay</surname> <given-names>J.-L.</given-names></name> <name><surname>Besson</surname> <given-names>M.</given-names></name></person-group> (<year>2014</year>). <article-title>Twelve months of active musical training in 8- to 10-year-old children enhances the preattentive processing of syllabic duration and voice onset time</article-title>. <source>Cereb. Cortex</source> <volume>24</volume>, <fpage>956</fpage>&#x2013;<lpage>967</lpage>. doi: <pub-id pub-id-type="doi">10.1093/cercor/bhs377</pub-id>, PMID: <pub-id pub-id-type="pmid">23236208</pub-id></citation></ref>
<ref id="ref15"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Christiner</surname> <given-names>M.</given-names></name></person-group> (<year>2013</year>). Singing Performance and Language Aptitude: Behavioural Study on Singing Performance and its Relation to the Pronunciation of a second Language. Master Thesis. University of Vienna, Vienna.</citation></ref>
<ref id="ref16"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Christiner</surname> <given-names>M.</given-names></name></person-group> (<year>2018</year>). &#x201C;<article-title>Let the music speak: examining the relationship between music and language aptitude in pre-school children</article-title>,&#x201D; in <source>Exploring Language Aptitude: Views From Psychology, the Language Sciences, and Cognitive Neuroscience.</source> ed. <person-group person-group-type="editor"><name><surname>Reiterer</surname> <given-names>S. M.</given-names></name></person-group> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer Nature</publisher-name>), <fpage>149</fpage>&#x2013;<lpage>166</lpage>.</citation></ref>
<ref id="ref17"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Christiner</surname> <given-names>M.</given-names></name></person-group> (<year>2020</year>). Musicality and Second Language Acquisition: Singing and Phonetic Language Aptitude Phonetic Language Aptitude. Dissertation. University of Vienna.</citation></ref>
<ref id="ref18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Christiner</surname> <given-names>M.</given-names></name> <name><surname>Bernhofs</surname> <given-names>V.</given-names></name> <name><surname>Gro&#x00DF;</surname> <given-names>C.</given-names></name></person-group> (<year>2022</year>). <article-title>Individual differences in singing behavior during childhood predicts language performance during adulthood</article-title>. <source>Language</source> <volume>7</volume>:<fpage>72</fpage>. doi: <pub-id pub-id-type="doi">10.3390/languages7020072</pub-id></citation></ref>
<ref id="ref19"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Christiner</surname> <given-names>M.</given-names></name> <name><surname>Gross</surname> <given-names>C.</given-names></name> <name><surname>Seither-Preisler</surname> <given-names>A.</given-names></name> <name><surname>Schneider</surname> <given-names>P.</given-names></name></person-group> (<year>2021</year>). <article-title>The melody of speech: what the melodic perception of speech reveals about language performance and musical abilities</article-title>. <source>Language</source> <volume>6</volume>:<fpage>132</fpage>. doi: <pub-id pub-id-type="doi">10.3390/languages6030132</pub-id></citation></ref>
<ref id="ref20"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Christiner</surname> <given-names>M.</given-names></name> <name><surname>Reiterer</surname> <given-names>S. M.</given-names></name></person-group> (<year>2013</year>). <article-title>Song and speech: examining the link between singing talent and speech imitation ability</article-title>. <source>Front. Psychol.</source> <volume>4</volume>:<fpage>874</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fpsyg.2013.00874</pub-id>, PMID: <pub-id pub-id-type="pmid">24319438</pub-id></citation></ref>
<ref id="ref21"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Christiner</surname> <given-names>M.</given-names></name> <name><surname>Reiterer</surname> <given-names>S. M.</given-names></name></person-group> (<year>2015</year>). <article-title>A Mozart is not a Pavarotti: singers outperform instrumentalists on foreign accent imitation</article-title>. <source>Front. Hum. Neurosci.</source> <volume>9</volume>:<fpage>482</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fnhum.2015.00482</pub-id>, PMID: <pub-id pub-id-type="pmid">26379537</pub-id></citation></ref>
<ref id="ref22"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Christiner</surname> <given-names>M.</given-names></name> <name><surname>Reiterer</surname> <given-names>S. M.</given-names></name></person-group> (<year>2018</year>). <article-title>Early influence of musical abilities and working memory on speech imitation abilities: study with pre-school children</article-title>. <source>Brain Sci.</source> <volume>8</volume>:<fpage>169</fpage>. doi: <pub-id pub-id-type="doi">10.3390/brainsci8090169</pub-id>, PMID: <pub-id pub-id-type="pmid">30200479</pub-id></citation></ref>
<ref id="ref23"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Christiner</surname> <given-names>M.</given-names></name> <name><surname>Reiterer</surname> <given-names>S.</given-names></name></person-group> (<year>2019</year>). &#x201C;<article-title>Music, song and speech</article-title>,&#x201D; in <source>The Internal Context of Bilingual Processing.</source> eds. <person-group person-group-type="editor"><name><surname>Truscott</surname> <given-names>J.</given-names></name> <name><surname>Smith</surname> <given-names>M. S.</given-names></name></person-group> (<publisher-loc>Amsterdam</publisher-loc>: <publisher-name>John Benjamins</publisher-name>), <fpage>131</fpage>&#x2013;<lpage>156</lpage>.</citation></ref>
<ref id="ref24"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Christiner</surname> <given-names>M.</given-names></name> <name><surname>R&#x00FC;degger</surname> <given-names>S.</given-names></name> <name><surname>Reiterer</surname> <given-names>S. M.</given-names></name></person-group> (<year>2018</year>). <article-title>Sing Chinese and tap Tagalog? Predicting individual differences in musical and phonetic aptitude using language families differing by sound-typology</article-title>. <source>Int. J. Multiling.</source> <volume>15</volume>, <fpage>455</fpage>&#x2013;<lpage>471</lpage>. doi: <pub-id pub-id-type="doi">10.1080/14790718.2018.1424171</pub-id></citation></ref>
<ref id="ref25"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Conway</surname> <given-names>A. R. A.</given-names></name> <name><surname>Kane</surname> <given-names>M. J.</given-names></name> <name><surname>Bunting</surname> <given-names>M. F.</given-names></name> <name><surname>Hambrick</surname> <given-names>D. Z.</given-names></name> <name><surname>Wilhelm</surname> <given-names>O.</given-names></name> <name><surname>Engle</surname> <given-names>R. W.</given-names></name></person-group> (<year>2005</year>). <article-title>Working memory span tasks: a methodological review and user&#x2019;s guide</article-title>. <source>Psychon. Bull. Rev.</source> <volume>12</volume>, <fpage>769</fpage>&#x2013;<lpage>786</lpage>. doi: <pub-id pub-id-type="doi">10.3758/bf03196772</pub-id>, PMID: <pub-id pub-id-type="pmid">16523997</pub-id></citation></ref>
<ref id="ref301"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Coumel</surname> <given-names>M.</given-names></name> <name><surname>Christiner</surname> <given-names>M.</given-names></name> <name><surname>Reiterer</surname> <given-names>S. M.</given-names></name></person-group> (<year>2019</year>). <article-title>Second language accent faking ability depends on musical abilities, not on working memory</article-title>. <source>Front. Psychol.</source> <volume>10</volume>:<fpage>257</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fpsyg.2019.00257</pub-id>, PMID: <pub-id pub-id-type="pmid">17348539</pub-id></citation></ref>
<ref id="ref26"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dalla Bella</surname> <given-names>S.</given-names></name> <name><surname>Berkowska</surname> <given-names>M.</given-names></name></person-group> (<year>2009</year>). <article-title>Singing proficiency in the majority: normality and &#x201C;phenotypes&#x201D; of poor singing</article-title>. <source>Ann. N. Y. Acad. Sci.</source> <volume>1169</volume>, <fpage>99</fpage>&#x2013;<lpage>107</lpage>. doi: <pub-id pub-id-type="doi">10.1111/j.1749-6632.2009.04558.x</pub-id>, PMID: <pub-id pub-id-type="pmid">19673762</pub-id></citation></ref>
<ref id="ref27"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dalla Bella</surname> <given-names>S.</given-names></name> <name><surname>Gigu&#x00E8;re</surname> <given-names>J.-F.</given-names></name> <name><surname>Peretz</surname> <given-names>I.</given-names></name></person-group> (<year>2007</year>). <article-title>Singing proficiency in the general population</article-title>. <source>J. Acoust. Soc. Am.</source> <volume>121</volume>, <fpage>1182</fpage>&#x2013;<lpage>1189</lpage>. doi: <pub-id pub-id-type="doi">10.1121/1.2427111</pub-id>, PMID: <pub-id pub-id-type="pmid">17348539</pub-id></citation></ref>
<ref id="ref28"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ding</surname> <given-names>N. i.</given-names></name> <name><surname>Patel</surname> <given-names>A. D.</given-names></name> <name><surname>Chen</surname> <given-names>L.</given-names></name> <name><surname>Butler</surname> <given-names>H.</given-names></name> <name><surname>Luo</surname> <given-names>C.</given-names></name> <name><surname>Poeppel</surname> <given-names>D.</given-names></name></person-group> (<year>2017</year>). <article-title>Temporal modulations in speech and music</article-title>. <source>Neurosci. Biobehav. Rev.</source> <volume>81</volume>, <fpage>181</fpage>&#x2013;<lpage>187</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neubiorev.2017.02.011</pub-id>, PMID: <pub-id pub-id-type="pmid">28212857</pub-id></citation></ref>
<ref id="ref29"><citation citation-type="book"><person-group person-group-type="author"><name><surname>D&#x00F6;rnyei</surname> <given-names>Z.</given-names></name></person-group> (<year>2005</year>). &#x201C;<article-title>The psychology of the language learner</article-title>,&#x201D; in <source>Individual Differences in Second Language Acquisition.</source> <publisher-loc>Mahwah, NJ</publisher-loc>: <publisher-name>L. Erlbaum</publisher-name>.</citation></ref>
<ref id="ref30"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>D&#x00F6;rnyei</surname> <given-names>Z.</given-names></name></person-group> (<year>2006</year>). <article-title>Themes in SLA research</article-title>. <source>AILA Rev.</source> <volume>19</volume>, <fpage>42</fpage>&#x2013;<lpage>68</lpage>. doi: <pub-id pub-id-type="doi">10.1075/aila.19.05dor</pub-id></citation></ref>
<ref id="ref31"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Engle</surname> <given-names>R. W.</given-names></name> <name><surname>Tuholski</surname> <given-names>S. W.</given-names></name> <name><surname>Laughlin</surname> <given-names>J. E.</given-names></name> <name><surname>Conway</surname> <given-names>A. R. A.</given-names></name></person-group> (<year>1999</year>). <article-title>Working memory, short-term memory, and general fluid intelligence: a latent-variable approach</article-title>. <source>J. Exp. Psychol.</source> <volume>128</volume>, <fpage>309</fpage>&#x2013;<lpage>331</lpage>. doi: <pub-id pub-id-type="doi">10.1037//0096-3445.128.3.309</pub-id>, PMID: <pub-id pub-id-type="pmid">10513398</pub-id></citation></ref>
<ref id="ref32"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Franco</surname> <given-names>F.</given-names></name> <name><surname>Suttora</surname> <given-names>C.</given-names></name> <name><surname>Spinelli</surname> <given-names>M.</given-names></name> <name><surname>Kozar</surname> <given-names>I.</given-names></name> <name><surname>Fasolo</surname> <given-names>M.</given-names></name></person-group> (<year>2021</year>). <article-title>Singing to infants matters: early singing interactions affect musical preferences and facilitate vocabulary building</article-title>. <source>J. Child Lang.</source> <volume>49</volume>, <fpage>552</fpage>&#x2013;<lpage>577</lpage>. doi: <pub-id pub-id-type="doi">10.1017/S0305000921000167</pub-id>, PMID: <pub-id pub-id-type="pmid">33908341</pub-id></citation></ref>
<ref id="ref33"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fran&#x00E7;ois</surname> <given-names>C.</given-names></name> <name><surname>Chobert</surname> <given-names>J.</given-names></name> <name><surname>Besson</surname> <given-names>M.</given-names></name> <name><surname>Sch&#x00F6;n</surname> <given-names>D.</given-names></name></person-group> (<year>2013</year>). <article-title>Music training for the development of speech segmentation</article-title>. <source>Cereb. Cortex</source> <volume>23</volume>, <fpage>2038</fpage>&#x2013;<lpage>2043</lpage>. doi: <pub-id pub-id-type="doi">10.1093/cercor/bhs180</pub-id>, PMID: <pub-id pub-id-type="pmid">22784606</pub-id></citation></ref>
<ref id="ref34"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gathercole</surname> <given-names>S. E.</given-names></name></person-group> (<year>2006</year>). <article-title>Nonword repetition and word learning: the nature of the relationship</article-title>. <source>Appl. Psycholinguist.</source> <volume>27</volume>, <fpage>513</fpage>&#x2013;<lpage>543</lpage>. doi: <pub-id pub-id-type="doi">10.1017/S0142716406060383</pub-id></citation></ref>
<ref id="ref35"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gathercole</surname> <given-names>S. E.</given-names></name> <name><surname>Baddeley</surname> <given-names>A. D.</given-names></name></person-group> (<year>1990</year>). <article-title>The role of phonological memory in vocabulary acquisition: a study of young children learning new names</article-title>. <source>Br. J. Psychol.</source> <volume>81</volume>, <fpage>439</fpage>&#x2013;<lpage>454</lpage>. doi: <pub-id pub-id-type="doi">10.1111/j.2044-8295.1990.tb02371.x</pub-id></citation></ref>
<ref id="ref36"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Gathercole</surname> <given-names>S. E.</given-names></name> <name><surname>Baddeley</surname> <given-names>A. D.</given-names></name></person-group> (<year>1993</year>). <source>Working Memory and Language.</source> <publisher-loc>Hove</publisher-loc>: <publisher-name>Erlbaum</publisher-name>.</citation></ref>
<ref id="ref302"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Gordon</surname> <given-names>E.</given-names></name></person-group> (<year>1989</year>). <source>Advanced Measures of Music Audiation.</source> <publisher-name>Chicago, IL: GIA.</publisher-name></citation></ref>
<ref id="ref37"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gordon</surname> <given-names>R. L.</given-names></name> <name><surname>Sch&#x00F6;n</surname> <given-names>D.</given-names></name> <name><surname>Magne</surname> <given-names>C.</given-names></name> <name><surname>Ast&#x00E9;sano</surname> <given-names>C.</given-names></name> <name><surname>Besson</surname> <given-names>M.</given-names></name></person-group> (<year>2010</year>). <article-title>Words and melody are intertwined in perception of sung words: EEG and behavioral evidence</article-title>. <source>PLoS One</source> <volume>5</volume>:<fpage>e9889</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pone.0009889</pub-id>, PMID: <pub-id pub-id-type="pmid">20360991</pub-id></citation></ref>
<ref id="ref38"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Gottfried</surname> <given-names>T. L.</given-names></name></person-group> (<year>2019</year>). &#x201C;<article-title>Music and language learning</article-title>,&#x201D; in <source>Doing SLA Research With Implications for the Classroom.</source> eds. <person-group person-group-type="editor"><name><surname>Robert</surname> <given-names>M. D.</given-names></name> <name><surname>Goretti</surname> <given-names>P. B.</given-names></name></person-group> (<publisher-loc>Amsterdam</publisher-loc>: <publisher-name>John Benjamins</publisher-name>), <fpage>221</fpage>&#x2013;<lpage>237</lpage>.</citation></ref>
<ref id="ref39"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Han</surname> <given-names>Y.</given-names></name> <name><surname>Goudbeek</surname> <given-names>M.</given-names></name> <name><surname>Mos</surname> <given-names>M.</given-names></name> <name><surname>Swerts</surname> <given-names>M.</given-names></name></person-group> (<year>2019</year>). <article-title>Mandarin tone identification by tone-na&#x00EF;ve musicians and non-musicians in auditory-visual and auditory-only conditions</article-title>. <source>Front. Commun.</source> <volume>4</volume>:<fpage>70</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fcomm.2019.00070</pub-id></citation></ref>
<ref id="ref40"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Holtz</surname> <given-names>K. L.</given-names></name></person-group> (<year>1993</year>). &#x201C;<article-title>Information integration and reading disabilities</article-title>,&#x201D; in <source>Facets of Dyslexia and its Remediation.</source> eds. <person-group person-group-type="editor"><name><surname>Wright</surname> <given-names>S. F.</given-names></name> <name><surname>Groner</surname> <given-names>R.</given-names></name></person-group> (<publisher-name>Amsterdam: Elsevier Book Series</publisher-name>), <fpage>305</fpage>&#x2013;<lpage>320</lpage>.</citation></ref>
<ref id="ref41"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Intartaglia</surname> <given-names>B.</given-names></name> <name><surname>White-Schwoch</surname> <given-names>T.</given-names></name> <name><surname>Kraus</surname> <given-names>N.</given-names></name> <name><surname>Sch&#x00F6;n</surname> <given-names>D.</given-names></name></person-group> (<year>2017</year>). <article-title>Music training enhances the automatic neural processing of foreign speech sounds</article-title>. <source>Sci. Rep.</source> <volume>7</volume>:<fpage>12631</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s41598-017-12575-1</pub-id>, PMID: <pub-id pub-id-type="pmid">28974695</pub-id></citation></ref>
<ref id="ref42"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jackendoff</surname> <given-names>R.</given-names></name> <name><surname>Lerdahl</surname> <given-names>F.</given-names></name></person-group> (<year>2006</year>). <article-title>The capacity for music: what is it, and what&#x2019;s special about it?</article-title> <source>Cognition</source> <volume>100</volume>, <fpage>33</fpage>&#x2013;<lpage>72</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.cognition.2005.11.005</pub-id>, PMID: <pub-id pub-id-type="pmid">16384553</pub-id></citation></ref>
<ref id="ref303"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jepsen</surname> <given-names>M. L.</given-names></name> <name><surname>Ewert</surname> <given-names>S. D.</given-names></name> <name><surname>Dau</surname> <given-names>T.</given-names></name></person-group> (<year>2008</year>). <article-title>A computational model of human auditory signal processing and perception</article-title>. <source>J. Acoust. Soc. Am.</source> <volume>124</volume>, <fpage>422</fpage>&#x2013;<lpage>438</lpage>. doi: <pub-id pub-id-type="doi">10.1121/1.2924135</pub-id></citation></ref>
<ref id="ref43"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Klingberg</surname> <given-names>T.</given-names></name> <name><surname>Forssberg</surname> <given-names>H.</given-names></name> <name><surname>Westerberg</surname> <given-names>H.</given-names></name></person-group> (<year>2002</year>). <article-title>Training of working memory in children with ADHD</article-title>. <source>J. Clin. Exp. Neuropsychol.</source> <volume>24</volume>, <fpage>781</fpage>&#x2013;<lpage>791</lpage>. doi: <pub-id pub-id-type="doi">10.1076/jcen.24.6.781.8395</pub-id></citation></ref>
<ref id="ref44"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Koelsch</surname> <given-names>S.</given-names></name> <name><surname>Schulze</surname> <given-names>K.</given-names></name> <name><surname>Sammler</surname> <given-names>D.</given-names></name> <name><surname>Fritz</surname> <given-names>T.</given-names></name> <name><surname>M&#x00FC;ller</surname> <given-names>K.</given-names></name> <name><surname>Gruber</surname> <given-names>O.</given-names></name></person-group> (<year>2009</year>). <article-title>Functional architecture of verbal and tonal working memory: an FMRI study</article-title>. <source>Hum. Brain Mapp.</source> <volume>30</volume>, <fpage>859</fpage>&#x2013;<lpage>873</lpage>. doi: <pub-id pub-id-type="doi">10.1002/hbm.20550</pub-id>, PMID: <pub-id pub-id-type="pmid">18330870</pub-id></citation></ref>
<ref id="ref45"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Larrouy-Maestri</surname> <given-names>P.</given-names></name> <name><surname>L&#x00E9;v&#x00EA;que</surname> <given-names>Y.</given-names></name> <name><surname>Sch&#x00F6;n</surname> <given-names>D.</given-names></name> <name><surname>Giovanni</surname> <given-names>A.</given-names></name> <name><surname>Morsomme</surname> <given-names>D.</given-names></name></person-group> (<year>2013</year>). <article-title>The evaluation of singing voice accuracy: a comparison between subjective and objective methods</article-title>. <source>J. Voice</source> <volume>27</volume>, <fpage>259.e1</fpage>&#x2013;<lpage>259.e5</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jvoice.2012.11.003</pub-id>, PMID: <pub-id pub-id-type="pmid">23280380</pub-id></citation></ref>
<ref id="ref46"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lee</surname> <given-names>C.-Y.</given-names></name> <name><surname>Hung</surname> <given-names>T.-H.</given-names></name></person-group> (<year>2008</year>). <article-title>Identification of mandarin tones by English-speaking musicians and nonmusicians</article-title>. <source>J. Acoust. Soc. Am.</source> <volume>124</volume>, <fpage>3235</fpage>&#x2013;<lpage>3248</lpage>. doi: <pub-id pub-id-type="doi">10.1121/1.2990713</pub-id>, PMID: <pub-id pub-id-type="pmid">19045807</pub-id></citation></ref>
<ref id="ref47"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>M.</given-names></name> <name><surname>DeKeyser</surname> <given-names>R.</given-names></name></person-group> (<year>2017</year>). <article-title>Perception practice, production practice, and musical ability in L2 mandarin tone-word learning</article-title>. <source>Stud. Second. Lang. Acquis.</source> <volume>39</volume>, <fpage>593</fpage>&#x2013;<lpage>620</lpage>. doi: <pub-id pub-id-type="doi">10.1017/S0272263116000358</pub-id></citation></ref>
<ref id="ref48"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Lin</surname> <given-names>T.</given-names></name></person-group> (<year>1985</year>). &#x201C;<article-title>Preliminary experiments on the nature of mandarin neutral tone</article-title>,&#x201D; in <source>Working Papers in Experimental Phonetics.</source> eds. <person-group person-group-type="editor"><name><surname>Lin</surname> <given-names>T.</given-names></name> <name><surname>Wand</surname> <given-names>L.</given-names></name></person-group> (<publisher-loc>Beijing</publisher-loc>: <publisher-name>Beijing University Press</publisher-name>), <fpage>1</fpage>&#x2013;<lpage>26</lpage>.</citation></ref>
<ref id="ref49"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>L.</given-names></name> <name><surname>Kager</surname> <given-names>R.</given-names></name></person-group> (<year>2017</year>). <article-title>Enhanced music sensitivity in 9-month-old bilingual infants</article-title>. <source>Cogn. Process.</source> <volume>18</volume>, <fpage>55</fpage>&#x2013;<lpage>65</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s10339-016-0780-7</pub-id>, PMID: <pub-id pub-id-type="pmid">27817073</pub-id></citation></ref>
<ref id="ref50"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ludke</surname> <given-names>K. M.</given-names></name> <name><surname>Ferreira</surname> <given-names>F.</given-names></name> <name><surname>Overy</surname> <given-names>K.</given-names></name></person-group> (<year>2014</year>). <article-title>Singing can facilitate foreign language learning</article-title>. <source>Mem. Cogn.</source> <volume>42</volume>, <fpage>41</fpage>&#x2013;<lpage>52</lpage>. doi: <pub-id pub-id-type="doi">10.3758/s13421-013-0342-5</pub-id>, PMID: <pub-id pub-id-type="pmid">23860945</pub-id></citation></ref>
<ref id="ref51"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Margulis</surname> <given-names>E. H.</given-names></name> <name><surname>Simchy-Gross</surname> <given-names>R.</given-names></name> <name><surname>Black</surname> <given-names>J. L.</given-names></name></person-group> (<year>2015</year>). <article-title>Pronunciation difficulty, temporal regularity, and the speech-to-song illusion</article-title>. <source>Front. Psychol.</source> <volume>6</volume>:<fpage>48</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fpsyg.2015.00048</pub-id>, PMID: <pub-id pub-id-type="pmid">25688225</pub-id></citation></ref>
<ref id="ref52"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Moreno</surname> <given-names>S.</given-names></name></person-group> (<year>2009</year>). <article-title>Can music influence language and cognition?</article-title> <source>Contemp. Music. Rev.</source> <volume>28</volume>, <fpage>329</fpage>&#x2013;<lpage>345</lpage>. doi: <pub-id pub-id-type="doi">10.1080/07494460903404410</pub-id></citation></ref>
<ref id="ref53"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nan</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>L.</given-names></name> <name><surname>Geiser</surname> <given-names>E.</given-names></name> <name><surname>Shu</surname> <given-names>H.</given-names></name> <name><surname>Gong</surname> <given-names>C. C.</given-names></name> <name><surname>Dong</surname> <given-names>Q.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Piano training enhances the neural processing of pitch and improves speech perception in mandarin-speaking children</article-title>. <source>Proc. Natl. Acad. Sci.</source> <volume>115</volume>, <fpage>E6630</fpage>&#x2013;<lpage>E6639</lpage>. doi: <pub-id pub-id-type="doi">10.1073/pnas.1808412115</pub-id>, PMID: <pub-id pub-id-type="pmid">29941577</pub-id></citation></ref>
<ref id="ref54"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Norris</surname> <given-names>D. G.</given-names></name> <name><surname>Hall</surname> <given-names>J.</given-names></name> <name><surname>Gathercole</surname> <given-names>S. E.</given-names></name></person-group> (<year>2019</year>). <article-title>Can short-term memory be trained?</article-title> <source>Mem. Cogn.</source> <volume>47</volume>, <fpage>1012</fpage>&#x2013;<lpage>1023</lpage>. doi: <pub-id pub-id-type="doi">10.3758/s13421-019-00901-z</pub-id>, PMID: <pub-id pub-id-type="pmid">30815843</pub-id></citation></ref>
<ref id="ref55"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Parbery-Clark</surname> <given-names>A.</given-names></name> <name><surname>Tierney</surname> <given-names>A.</given-names></name> <name><surname>Strait</surname> <given-names>D. L.</given-names></name> <name><surname>Kraus</surname> <given-names>N.</given-names></name></person-group> (<year>2012</year>). <article-title>Musicians have fine-tuned neural distinction of speech syllables</article-title>. <source>Neuroscience</source> <volume>219</volume>, <fpage>111</fpage>&#x2013;<lpage>119</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neuroscience.2012.05.042</pub-id>, PMID: <pub-id pub-id-type="pmid">22634507</pub-id></citation></ref>
<ref id="ref56"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Patel</surname> <given-names>A. D.</given-names></name></person-group> (<year>2007</year>). <source>Music, Language, and the Brain.</source> <publisher-loc>Oxford</publisher-loc>: <publisher-name>Oxford University Press</publisher-name></citation></ref>
<ref id="ref57"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Preisler</surname> <given-names>A.</given-names></name> <name><surname>Johnson</surname> <given-names>L.</given-names></name> <name><surname>Preisler</surname> <given-names>E.</given-names></name> <name><surname>Seither</surname> <given-names>S.</given-names></name> <name><surname>L&#x00FC;tkenh&#x00F6;ner</surname> <given-names>B.</given-names></name></person-group> (<year>2011</year>). <article-title>The perception of dual-aspect tone sequences changes with stimulus exposure</article-title>. <source>Brain Res. Dev.</source> <fpage>49</fpage>&#x2013;<lpage>72</lpage>.</citation></ref>
<ref id="ref58"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Robinson</surname> <given-names>P.</given-names></name></person-group> (<year>2005</year>). <article-title>Aptitude and second language acquisition</article-title>. <source>Annu. Rev. Appl. Linguist.</source> <volume>25</volume>, <fpage>46</fpage>&#x2013;<lpage>73</lpage>. doi: <pub-id pub-id-type="doi">10.1017/S0267190505000036</pub-id></citation></ref>
<ref id="ref59"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Robinson</surname> <given-names>P.</given-names></name></person-group> (<year>2019</year>). &#x201C;<article-title>Learning conditions, aptitude complexes, and SLA</article-title>,&#x201D; in <source>Doing SLA Research With Implications for the Classroom.</source> eds. <person-group person-group-type="editor"><name><surname>Robert</surname> <given-names>M. D.</given-names></name> <name><surname>Goretti</surname> <given-names>P. B.</given-names></name></person-group> (<publisher-loc>Amsterdam</publisher-loc>: <publisher-name>John Benjamins</publisher-name>), <fpage>113</fpage>&#x2013;<lpage>133</lpage>.</citation></ref>
<ref id="ref60"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Salam&#x00E9;</surname> <given-names>P.</given-names></name> <name><surname>Baddeley</surname> <given-names>A. D.</given-names></name></person-group> (<year>1989</year>). <article-title>Effects of background music on phonological short-term memory</article-title>. <source>Q. J. Exp. Psychol. A</source> <volume>41</volume>, <fpage>107</fpage>&#x2013;<lpage>122</lpage>. doi: <pub-id pub-id-type="doi">10.1080/14640748908402355</pub-id></citation></ref>
<ref id="ref61"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schneider</surname> <given-names>P.</given-names></name> <name><surname>Sluming</surname> <given-names>V.</given-names></name> <name><surname>Roberts</surname> <given-names>N.</given-names></name> <name><surname>Bleeck</surname> <given-names>S.</given-names></name> <name><surname>Rupp</surname> <given-names>A.</given-names></name></person-group> (<year>2005</year>). <article-title>Structural, functional, and perceptual differences in Heschl&#x2019;s gyrus and musical instrument preference</article-title>. <source>Ann. N. Y. Acad. Sci.</source> <volume>1060</volume>, <fpage>387</fpage>&#x2013;<lpage>394</lpage>. doi: <pub-id pub-id-type="doi">10.1196/annals.1360.033</pub-id>, PMID: <pub-id pub-id-type="pmid">16597790</pub-id></citation></ref>
<ref id="ref62"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schneider</surname> <given-names>P.</given-names></name> <name><surname>Wengenroth</surname> <given-names>M.</given-names></name></person-group> (<year>2009</year>). <article-title>The neural basis of individual holistic and spectral sound perception</article-title>. <source>Contemp. Music. Rev.</source> <volume>28</volume>, <fpage>315</fpage>&#x2013;<lpage>328</lpage>. doi: <pub-id pub-id-type="doi">10.1080/07494460903404402</pub-id></citation></ref>
<ref id="ref63"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sch&#x00F6;n</surname> <given-names>D.</given-names></name> <name><surname>Magne</surname> <given-names>C.</given-names></name> <name><surname>Besson</surname> <given-names>M.</given-names></name></person-group> (<year>2004</year>). <article-title>The music of speech: music training facilitates pitch processing in both music and language</article-title>. <source>Psychophysiology</source> <volume>41</volume>, <fpage>341</fpage>&#x2013;<lpage>349</lpage>. doi: <pub-id pub-id-type="doi">10.1111/1469-8986.00172.x</pub-id>, PMID: <pub-id pub-id-type="pmid">15102118</pub-id></citation></ref>
<ref id="ref64"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Seither-Preisler</surname> <given-names>A.</given-names></name> <name><surname>Johnson</surname> <given-names>L.</given-names></name> <name><surname>Krumbholz</surname> <given-names>K.</given-names></name> <name><surname>Nobbe</surname> <given-names>A.</given-names></name> <name><surname>Patterson</surname> <given-names>R.</given-names></name> <name><surname>Seither</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2007</year>). <article-title>Tone sequences with conflicting fundamental pitch and timbre changes are heard differently by musicians and nonmusicians. Journal of experimental psychology</article-title>. <source>Hum. Percept. Perfor.</source> <volume>33</volume>, <fpage>743</fpage>&#x2013;<lpage>751</lpage>. doi: <pub-id pub-id-type="doi">10.1037/0096-1523.33.3.743</pub-id>, PMID: <pub-id pub-id-type="pmid">17563235</pub-id></citation></ref>
<ref id="ref65"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Thiessen</surname> <given-names>E. D.</given-names></name> <name><surname>Saffran</surname> <given-names>J. R.</given-names></name></person-group> (<year>2009</year>). <article-title>How the melody facilitates the message and vice versa in infant learning and memory</article-title>. <source>Ann. N. Y. Acad. Sci.</source> <volume>1169</volume>, <fpage>225</fpage>&#x2013;<lpage>233</lpage>. doi: <pub-id pub-id-type="doi">10.1111/j.1749-6632.2009.04547.x</pub-id>, PMID: <pub-id pub-id-type="pmid">19673786</pub-id></citation></ref>
<ref id="ref66"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Thorn</surname> <given-names>A. S. C.</given-names></name> <name><surname>Gathercole</surname> <given-names>S. E.</given-names></name></person-group> (<year>2001</year>). <article-title>Language differences in verbal short-term memory do not exclusively originate in the process of subvocal rehearsal</article-title>. <source>Psychon. Bull. Rev.</source> <volume>8</volume>, <fpage>357</fpage>&#x2013;<lpage>364</lpage>. doi: <pub-id pub-id-type="doi">10.3758/BF03196173</pub-id>, PMID: <pub-id pub-id-type="pmid">11495126</pub-id></citation></ref>
<ref id="ref67"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>W. S.-Y.</given-names></name></person-group> (<year>1967</year>). <article-title>Phonological features of tone</article-title>. <source>Int. J. Am. Linguist.</source> <volume>33</volume>, <fpage>93</fpage>&#x2013;<lpage>105</lpage>.</citation></ref>
<ref id="ref68"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Wechsler</surname> <given-names>D.</given-names></name></person-group> (<year>1939</year>). <source>The Measurement of Adult Intelligence.</source> <publisher-loc>Baltimore, MD</publisher-loc>: <publisher-name>Williams &#x0026; Wilkins.</publisher-name></citation></ref>
<ref id="ref69"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Welch</surname> <given-names>G. F.</given-names></name> <name><surname>Himonides</surname> <given-names>E.</given-names></name> <name><surname>Saunders</surname> <given-names>J.</given-names></name> <name><surname>Papageorgi</surname> <given-names>I.</given-names></name> <name><surname>Rinta</surname> <given-names>T.</given-names></name> <name><surname>Preti</surname> <given-names>C.</given-names></name> <etal/></person-group>. (<year>2011</year>). <article-title>Researching the first year of the national singing programme sing up in England: an initial impact evaluation</article-title>. <source>Psychomusicology</source> <volume>21</volume>, <fpage>83</fpage>&#x2013;<lpage>97</lpage>. doi: <pub-id pub-id-type="doi">10.1037/h0094006</pub-id></citation></ref>
<ref id="ref70"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wen</surname> <given-names>Z.</given-names></name> <name><surname>Biedro&#x0144;</surname> <given-names>A.</given-names></name> <name><surname>Skehan</surname> <given-names>P.</given-names></name></person-group> (<year>2017</year>). <article-title>Foreign language aptitude theory: yesterday, today and tomorrow</article-title>. <source>Lang. Teach.</source> <volume>50</volume>, <fpage>1</fpage>&#x2013;<lpage>31</lpage>. doi: <pub-id pub-id-type="doi">10.1017/S0261444816000276</pub-id></citation></ref>
<ref id="ref71"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wen</surname> <given-names>Z.</given-names></name> <name><surname>Skehan</surname> <given-names>P.</given-names></name></person-group> (<year>2011</year>). <article-title>A new perspective on foreign language aptitude research: building and supporting a case for &#x201C;working memory as language aptitude&#x201D;</article-title>. <source>Ilha do Desterro</source> <fpage>15</fpage>&#x2013;<lpage>44</lpage>. doi: <pub-id pub-id-type="doi">10.5007/2175-8026.2011n60p015</pub-id></citation></ref>
<ref id="ref72"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Williamson</surname> <given-names>V. J.</given-names></name> <name><surname>Baddeley</surname> <given-names>A. D.</given-names></name> <name><surname>Hitch</surname> <given-names>G. J.</given-names></name></person-group> (<year>2010</year>). <article-title>Musicians&#x2019; and nonmusicians&#x2019; short-term memory for verbal and musical sequences: comparing phonological similarity and pitch proximity</article-title>. <source>Mem. Cogn.</source> <volume>38</volume>, <fpage>163</fpage>&#x2013;<lpage>175</lpage>. doi: <pub-id pub-id-type="doi">10.3758/MC.38.2.163</pub-id>, PMID: <pub-id pub-id-type="pmid">20173189</pub-id></citation></ref>
</ref-list>
</back>
</article>