<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Rehabil. Sci.</journal-id>
<journal-title>Frontiers in Rehabilitation Sciences</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Rehabil. Sci.</abbrev-journal-title>
<issn pub-type="epub">2673-6861</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fresc.2023.1121034</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Rehabilitation Sciences</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Estimation of subjective quality of life in schizophrenic patients using speech features</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes"><name><surname>Shibata</surname><given-names>Yuko</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="cor1">&#x002A;</xref>
<xref ref-type="author-notes" rid="an1"><sup>&#x2020;</sup></xref><uri xlink:href="https://loop.frontiersin.org/people/2120284/overview"/></contrib>
<contrib contrib-type="author"><name><surname>Victorino</surname><given-names>John Noel</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="an1"><sup>&#x2020;</sup></xref><uri xlink:href="https://loop.frontiersin.org/people/2160354/overview"/></contrib>
<contrib contrib-type="author"><name><surname>Natsuyama</surname><given-names>Tomoya</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="author-notes" rid="an1"><sup>&#x2020;</sup></xref><uri xlink:href="https://loop.frontiersin.org/people/2017950/overview" /></contrib>
<contrib contrib-type="author"><name><surname>Okamoto</surname><given-names>Naomichi</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="author-notes" rid="an1"><sup>&#x2020;</sup></xref><uri xlink:href="https://loop.frontiersin.org/people/1196480/overview" /></contrib>
<contrib contrib-type="author"><name><surname>Yoshimura</surname><given-names>Reiji</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="author-notes" rid="an1"><sup>&#x2020;</sup></xref><uri xlink:href="https://loop.frontiersin.org/people/505537/overview" /></contrib>
<contrib contrib-type="author"><name><surname>Shibata</surname><given-names>Tomohiro</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="an1"><sup>&#x2020;</sup></xref><uri xlink:href="https://loop.frontiersin.org/people/2424/overview" /></contrib>
</contrib-group>
<aff id="aff1"><label><sup>1</sup></label><addr-line>Department of Life Science and System Engineering, Graduate School of Life Science and Systems Engineering</addr-line>, <institution>Kyushu Institute of Technology</institution>, <addr-line>Kitakyushu</addr-line>, <country>Japan</country></aff>
<aff id="aff2"><label><sup>2</sup></label><addr-line>Department of Psychiatry</addr-line>, <institution>University of Occupational and Environmental Health</institution>, <addr-line>Kitakyushu</addr-line>, <country>Japan</country></aff>
<author-notes>
<fn fn-type="edited-by"><p><bold>Edited by:</bold> Corneliu Bolbocean, University of Oxford, United Kingdom</p></fn>
<fn fn-type="edited-by"><p><bold>Reviewed by:</bold> Salim Heddam, University of Skikda, Algeria Ismail Mohd Khairuddin, Universiti Malaysia Pahang, Malaysia</p></fn>
<corresp id="cor1"><label>&#x002A;</label><bold>Correspondence:</bold> Yuko Shibata <email>shibata.yuko975@gmail.com</email></corresp><fn><p><bold>Citation:</bold> Shibata Y, Victorino JN, Natsuyama T, Okamoto N, Yoshimura R and Shibata T (2023) Estimation of subjective quality of life in schizophrenic patients using speech features. <italic>Front. Rehabil. Sci.</italic> 4:1121034. doi: 10.3389/fresc.2023.1121034</p></fn>
<fn id="an1"><label><sup>&#x2020;</sup></label><p>These authors have contributed equally to this work and share first authorship</p></fn>
<fn fn-type="other" id="fn001"><p><bold>Specialty Section:</bold> This article was submitted to Disability, Rehabilitation, and Inclusion, a section of the journal Frontiers in Rehabilitation Sciences</p></fn>
</author-notes>
<pub-date pub-type="epub"><day>10</day><month>03</month><year>2023</year></pub-date>
<pub-date pub-type="collection"><year>2023</year></pub-date>
<volume>4</volume><elocation-id>1121034</elocation-id>
<history>
<date date-type="received"><day>11</day><month>12</month><year>2022</year></date>
<date date-type="accepted"><day>13</day><month>02</month><year>2023</year></date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2023 Shibata, Victorino, Natsuyama, Okamoto, Yoshimura and Shibata.</copyright-statement>
<copyright-year>2023</copyright-year><copyright-holder>Shibata, Victorino, Natsuyama, Okamoto, Yoshimura and Shibata</copyright-holder><license license-type="open-access" xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract><sec><title>Introduction</title>
<p>Patients with schizophrenia experience the most prolonged hospital stay in Japan. Also, the high re-hospitalization rate affects their quality of life (QoL). Despite being an effective predictor of treatment, QoL has not been widely utilized due to time constraints and lack of interest. As such, this study aimed to estimate the schizophrenic patients&#x0027; subjective quality of life using speech features. Specifically, this study uses speech from patients with schizophrenia to estimate the subscale scores, which measure the subjective QoL of the patients. The objectives were to (1) estimate the subscale scores from different patients or cross-sectional measurements, and 2) estimate the subscale scores from the same patient in different periods or longitudinal measurements.</p>
</sec><sec><title>Methods</title>
<p>A conversational agent was built to record the responses of 18 schizophrenic patients on the Japanese Schizophrenia Quality of Life Scale (JSQLS) with three subscales: &#x201C;Psychosocial,&#x201D; &#x201C;Motivation and Energy,&#x201D; and &#x201C;Symptoms and Side-effects.&#x201D; These three subscales were used as objective variables. On the other hand, the speech features during measurement (Chromagram, Mel spectrogram, Mel-Frequency Cepstrum Coefficient) were used as explanatory variables. For the first objective, a trained model estimated the subscale scores for the 18 subjects using the Nested Cross-validation (CV) method. For the second objective, six of the 18 subjects were measured twice. Then, another trained model estimated the subscale scores for the second time using the 18 subjects&#x0027; data as training data. Ten different machine learning algorithms were used in this study, and the errors of the learned models were compared.</p>
</sec><sec><title>Results and Discussion</title>
<p>The results showed that the mean RMSE of the cross-sectional measurement was 13.433, with k-Nearest Neighbors as the best model. Meanwhile, the mean RMSE of the longitudinal measurement was 13.301, using Random Forest as the best. RMSE of less than 10 suggests that the estimated subscale scores using speech features were close to the actual JSQLS subscale scores. Ten out of 18 subjects were estimated with an RMSE of less than 10 for cross-sectional measurement. Meanwhile, five out of six had the same observation for longitudinal measurement. Future studies using a larger number of subjects and the development of more personalized models based on longitudinal measurements are needed to apply the results to telemedicine for continuous monitoring of QoL.</p>
</sec>
</abstract>
<kwd-group>
<kwd>quality of life</kwd>
<kwd>schizophrenia</kwd>
<kwd>speech analysis</kwd>
<kwd>machine learning</kwd>
<kwd>model development</kwd>
</kwd-group><counts>
<fig-count count="5"/>
<table-count count="11"/><equation-count count="18"/><ref-count count="41"/><page-count count="0"/><word-count count="0"/></counts>
</article-meta>
</front>
<body><sec id="s1" sec-type="intro"><label>1.</label><title>Introduction</title>
<p>The number of psychiatric beds in Japan is much larger than in other countries, and the length of hospital stay is as long as 285 days (Italy: 13.9 days, U.K.: 42.3 days) (<xref ref-type="bibr" rid="B1">1</xref>). The Ministry of Health, Labour, and Welfare (MHLW) has announced a vision for reforming mental health and medical welfare in response to prolonged hospitalization. The MHLW vision fundamentally shifts its policy from inpatient care to community-based care. The vision clearly states the improvement of inpatient treatment, the improvement of patients&#x0027; Quality of Life (QoL), and the development of support for early discharge from the hospital (<xref ref-type="bibr" rid="B2">2</xref>). Furthermore, a survey of readmission rates for 24,781 patients discharged in 2014 showed that 23&#x0025; were re-admitted three months after discharge, 30&#x0025; six months later, and 37&#x0025; one year later (<xref ref-type="bibr" rid="B3">3</xref>). Prolonged hospitalization and high readmission rates are issues for psychiatric care in Japan.</p>
<p>QoL is an effective predictor of symptom remission and functional recovery among schizophrenic patients. As such, QoL is an essential measure of outcome in treatment (<xref ref-type="bibr" rid="B4">4</xref>, <xref ref-type="bibr" rid="B5">5</xref>). It is crucial to understand and assess fluctuations in QoL scores and routine tests such as blood sampling; then use this information in interventions (<xref ref-type="bibr" rid="B6">6</xref>). However, QoL assessment is not routinely performed in clinical practice due to time constraints and lack of training and interest (<xref ref-type="bibr" rid="B7">7</xref>&#x2013;<xref ref-type="bibr" rid="B9">9</xref>).</p>
<p>QoL can be divided into objective assessment and subjective assessment. This study focuses on subjective QoL because patients are the main actors in their lives during hospitalization and after discharge. The subjective assessment is possible because schizophrenic patients can feel and report social impairment (<xref ref-type="bibr" rid="B10">10</xref>). In addition, this study examined the use of voice input to estimate QoL status instead of the conventional self-administered and semi-constructed interview methods. Multi-lingual speech recognition and emotional speech recognition have been actively studied in recent years (<xref ref-type="bibr" rid="B11">11</xref>&#x2013;<xref ref-type="bibr" rid="B13">13</xref>). Many voice-based applications have also been developed to remotely monitor the status and characteristics of speakers, such as health status (<xref ref-type="bibr" rid="B14">14</xref>&#x2013;<xref ref-type="bibr" rid="B16">16</xref>). These latest developments motivate this study to consider speech recognition as a fast and efficient means of human-machine interaction (<xref ref-type="bibr" rid="B17">17</xref>).</p>
<p>Therefore, this study examined the estimation of subscale scores that measure the subjective QoL of schizophrenic patients using speech features as a simple method to measure QoL. The objectives were to (<xref ref-type="bibr" rid="B1">1</xref>) estimate the subscale scores from different patients or cross-sectional measurements, and (<xref ref-type="bibr" rid="B2">2</xref>) estimate the subscale scores from the same patient in different periods or longitudinal measurements. The proposed method allows schizophrenic patients to measure their subjective QoL by themselves in the future. Furthermore, the proposed method provides opportunities to monitor QoL continuously and regularly during hospitalization and after discharge. Patients and medical care providers can share data and analysis.</p>
</sec>
<sec id="s2"><label>2.</label><title>Material and methods</title>
<p>We examined the feasibility of estimating the subjective QoL of schizophrenic patients using speech features. The three subscale scores of the Japanese Schizophrenia Quality of Life Scale (JSQLS) were collected using a conversational agent. The conversational agent recorded the JSQLS responses and the audio of the conversation. Then, models were developed and compared to estimate the subscale scores. There were two kinds of models developed in this study.
<list list-type="simple">
<list-item><label>1)</label>
<p>Model development to estimate subscale scores from among different patients or cross-sectional measurements</p></list-item>
<list-item><label>2)</label>
<p>Model development to estimate subscale scores from the same patient in different periods or longitudinal measurements</p></list-item>
</list></p>
<sec id="s2a"><label>2.1.</label><title>Target population demographics</title>
<p>Eighteen schizophrenic patients who agreed to participate in the study were included (<xref ref-type="table" rid="T1">Table&#x00A0;1</xref>). The mean age was 47.17 years, with seven males and 11 females. Global Assessment of Functioning (GAF) is a scale used to assess an overview of a subject&#x0027;s functioning. Psychological, social, and occupational functioning is rated as a single variable on an integer scale of 1&#x2013;100 points (<xref ref-type="bibr" rid="B18">18</xref>). The rater evaluates the subject&#x0027;s condition according to the scale&#x0027;s rating criteria. For example, a 91&#x2013;100 score indicates &#x201C;very good functioning and no psychiatric symptoms.&#x201D; Higher scores mean better symptoms and functioning. In this study, the psychiatrist or nurse in charge of the patient performed the evaluation (<xref ref-type="table" rid="T2">Table&#x00A0;2</xref>).</p>
<table-wrap id="T1" position="float"><label>Table 1</label>
<caption><p>Subjects&#x2019; demographics</p></caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th align="left">Characteristics</th>
<th align="center">Subjects</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left">Sample size, <italic>n</italic></td>
<td>18.000</td>
</tr>
<tr>
<td align="left">Age, mean (Std. Dev.)</td>
<td>45.170 (16.576)</td>
</tr>
<tr>
<td align="left">Male sex, <italic>n</italic> (&#x0025;)</td>
<td>7.000 (38.890)</td>
</tr>
<tr>
<td align="left">GAF, Range</td>
<td>32.000&#x2013;70.000</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T2" position="float"><label>Table 2</label>
<caption><p>Demographics of subjects measured twice.</p></caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th align="left">Characteristics</th>
<th align="center">Subjects</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left">Sample size, <italic>n</italic></td>
<td align="center">6.000</td>
</tr>
<tr>
<td align="left">Age, mean (Std. Dev.)</td>
<td align="center">32.160 (6.150)</td>
</tr>
<tr>
<td align="left">Male sex, <italic>n</italic> (&#x0025;)</td>
<td align="center">1.000 (16.670)</td>
</tr>
<tr>
<td align="left">GAF, Range</td>
<td align="center">50.000&#x2013;70.000</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2b"><label>2.2.</label><title>SQLS as a measure of subjective QoL</title>
<p>JSQLS was used to measure subjective QoL. The JSQLS provides a subjective assessment of the impact of the disease on the subject&#x0027;s life. The JSQLS consists of three scales: Psychosocial (15 items), Motivation and Energy (7 items), and Symptoms and Side-effects (8 items). During the scale development in previous studies, the questions were selected based on in-depth patient interviews. Then, the JSQLS questions were examined for reliability and validity (<xref ref-type="bibr" rid="B18">18</xref>, <xref ref-type="bibr" rid="B19">19</xref>).</p>
<sec id="s2b1"><label>2.2.1.</label><title>JSQLS calculation method</title>
<p>This section describes how the three subscale scores are calculated and evaluated based on the subject&#x0027;s answers. There are five options for each question, and the score for each question ranges from 0 to 4 points.
<list list-type="simple">
<list-item>
<p>&#x201C;Always&#x201D; (4 points)</p></list-item>
<list-item>
<p>&#x201C;Often&#x201D; (3 points)</p></list-item>
<list-item>
<p>&#x201C;Sometimes&#x201D; (2 points)</p></list-item>
<list-item>
<p>&#x201C;Rarely&#x201D; (1 point)</p></list-item>
<list-item>
<p>&#x201C;Never&#x201D; (0 points)</p></list-item>
</list>Each subscale was calculated to take values between 0 and 100, with higher scores indicating worse QoL while lower scores indicating better QoL.<disp-formula id="disp-formula1"><label>(1)</label><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="DM1"><mml:mtable columnalign="right left" rowspacing=".5em" columnspacing="thickmathspace" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mrow><mml:mi mathvariant="normal">Score</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /><mml:mrow><mml:mi mathvariant="normal">of</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /><mml:mrow><mml:mi mathvariant="normal">each</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /><mml:mrow><mml:mi mathvariant="normal">subscale</mml:mi></mml:mrow><mml:mspace width="thinmathspace" /></mml:mtd></mml:mtr><mml:mtr><mml:mtd /><mml:mtd><mml:mo>=</mml:mo><mml:mspace width="thinmathspace" /><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mfrac><mml:mrow><mml:mrow><mml:mi mathvariant="normal">Sum</mml:mi><mml:mspace width="thickmathspace" /></mml:mrow><mml:mspace width="thickmathspace" /><mml:mrow><mml:mi mathvariant="normal">of</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /><mml:mrow><mml:mi mathvariant="normal">the</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /><mml:mrow><mml:mi mathvariant="normal">crude</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /><mml:mrow><mml:mi mathvariant="normal">scores</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /><mml:mrow><mml:mi mathvariant="normal">for</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /><mml:mrow><mml:mi mathvariant="normal">each</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /><mml:mrow><mml:mi mathvariant="normal">scale</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mn>4</mml:mn><mml:mspace width="thickmathspace" /><mml:mo>&#x00D7;</mml:mo><mml:mspace width="thickmathspace" /><mml:mrow><mml:mi mathvariant="normal">Number</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /><mml:mrow><mml:mi mathvariant="normal">of</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /><mml:mrow><mml:mi mathvariant="normal">questions</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /><mml:mrow><mml:mi mathvariant="normal">for</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /><mml:mrow><mml:mi mathvariant="normal">each</mml:mi></mml:mrow><mml:mspace width="thickmathspace" /><mml:mrow><mml:mi mathvariant="normal">scale</mml:mi></mml:mrow></mml:mrow></mml:mfrac></mml:mrow><mml:mspace width="thinmathspace" /><mml:mo>&#x00D7;</mml:mo><mml:mspace width="thinmathspace" /><mml:mn>100</mml:mn></mml:mstyle></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>On the one hand, the numerator is the total score based on each subject&#x0027;s choices. The denominator is calculated as 4&#x2009;&#x00D7;&#x2009;15 questions for &#x201C;Psychosocial,&#x201D; 4&#x2009;&#x00D7;&#x2009;7 questions for &#x201C;Motivation and Energy,&#x201D; and 4&#x2009;&#x00D7;&#x2009;8 questions for &#x201C;Symptoms and Side-effects.&#x201D; Note that four questions under &#x201C;Motivation and Energy&#x201D; are scored inversely, i.e., &#x201C;Always&#x201D; (0 points), &#x201C;Often&#x201D; (1 point), &#x201C;Sometimes&#x201D; (2 points), &#x201C;Rarely&#x201D; (3 points), and &#x201C;never&#x201D; (4 points).</p>
</sec>
</sec>
<sec id="s2c"><label>2.3.</label><title>Development of a conversational agent to measure subjective QoL</title>
<p>In this study, the conversational agent asked the patient 30 JSQLS questions and recorded the subject&#x0027;s voice as he or she answered each question (<xref ref-type="fig" rid="F1">Figure&#x00A0;1</xref>).</p>
<fig id="F1" position="float"><label>Figure 1</label>
<caption><p>Conversational agent system architecture.</p></caption>
<graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="fresc-04-1121034-g001.tif"/>
</fig>
<p>The Web Speech API converts the participant&#x0027;s speech into text. Then, the conversation agent uses natural language understanding to classify the answer choices (&#x201C;Always,&#x201D; &#x201C;Often,&#x201D; &#x201C;Sometimes,&#x201D; &#x201C;Rarely,&#x201D; and &#x201C;Never&#x201D;). Next, Rasa Core (<xref ref-type="bibr" rid="B20">20</xref>) manages the interaction, including the flow of conversation and context processing. Rasa Core&#x0027;s natural language generator selects appropriate text responses based on the context and flow of the conversation. Finally, ResponsiveVoice.js generates the spoken response from the text response.</p>
</sec>
<sec id="s2d"><label>2.4.</label><title>Measurement method</title>
<p>Measurements were taken at the subject&#x0027;s hospital, a continuous employment support facility, and the subject&#x0027;s home. A quiet environment was ensured during the measurement for voice interaction and recording. First, the subjects were asked to read the 30 JSQLS questions before the measurement with the conversation agent. This step was implemented to prepare the subjects with the subsequent questions and clarify any questions. Then, the conversation agent spoke and displayed a question for the subject to listen and to see, respectively (<xref ref-type="fig" rid="F2">Figure&#x00A0;2</xref>). Next, the subject answered back to the conversation agent. The conversation agent recorded the subject&#x0027;s response and audio using a microphone array. From this point, the conversation agent either (A) repeats the question if the subject&#x0027;s response is not understood or (B) proceeds to the next question until all 30 questions are finished (<xref ref-type="fig" rid="F2">Figure&#x00A0;2</xref>).</p>
<fig id="F2" position="float"><label>Figure 2</label>
<caption><p>Preliminary experiment setup.</p></caption>
<graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="fresc-04-1121034-g002.tif"/>
</fig>
</sec>
<sec id="s2e"><label>2.5.</label><title>Audio processing and speech feature extraction</title>
<p>The following describes the acquired speech data. Subjects&#x0027; speech was recorded at a sampling frequency of 48&#x2005;kHz; the total unedited recording time, including conversational agent announcements, for the 18 subjects&#x2019; speech data was 98.200&#x2005;min, with an average recording time of 5.456&#x2005;min. The shortest recording time was 3.600&#x2005;min and the longest was 10.917&#x2005;min. The speech for the analysis was stripped of the conversational agent&#x0027;s announcements, silences, false responses, and noises. When the subject&#x0027;s speech was unclear, the conversation agent would listen back to the subject&#x0027;s speech, resulting in individual differences in recording time. The total recording time after removing the conversational agent&#x0027;s voice and the noise was 18.200&#x2005;min, with an average duration of 1.011&#x2005;min.</p>
<p>Using the Librosa audio library, speech data from 18 subjects were input, and Mel-spectrogram (128 dimensions), Mel-Frequency Cepstrum Coefficient (MFCC) (40 dimensions), and Chromagram (12 dimensions) speech features were Chromagram (12 dimensions) were extracted (<xref ref-type="fig" rid="F3">Figure&#x00A0;3</xref>). A total of 3,240 speech features with 180 dimensions per subject and 18 subjects were used as the objective variables. Mel spectrogram and MFCC mimicked, to an extent, the natural sound frequency reception pattern of humans (<xref ref-type="bibr" rid="B13">13</xref>) and are often used for voice separation and classification (<xref ref-type="bibr" rid="B21">21</xref>). Chromagram can infer the vocal tract&#x0027;s resonance characteristics as the signal&#x0027;s energy distribution concerning saturation and time (<xref ref-type="bibr" rid="B22">22</xref>).</p>
</sec>
<sec id="s2f"><label>2.6.</label><title>Model development</title>
<p>A model was developed using the extracted speech features to estimate the three JSQLS subscale scores. The features used to develop the model are the following.</p>
<fig id="F3" position="float"><label>Figure 3</label>
<caption><p>Data processing.</p></caption>
<graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="fresc-04-1121034-g003.tif"/>
</fig>
<p>Let <italic>x</italic> be the explanatory variable and <italic>y</italic>. be the objective variable. Specifically,
<list list-type="simple">
<list-item>
<p><inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM1"><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x223C;</mml:mo><mml:mn>12</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>: Chromagram</p></list-item>
<list-item>
<p><inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM2"><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x223C;</mml:mo><mml:mn>128</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>: Mel-spectrogram</p></list-item>
<list-item>
<p><inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM3"><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x223C;</mml:mo><mml:mn>40</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>: MFCC</p></list-item>
<list-item>
<p><inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM4"><mml:msub><mml:mi>y</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:math></inline-formula>: &#x201C;Psychosocial&#x201D; subscale</p></list-item>
<list-item>
<p><inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM5"><mml:msub><mml:mi>y</mml:mi><mml:mn>2</mml:mn></mml:msub></mml:math></inline-formula>: &#x201C;Motivation and Energy&#x201D; subscale</p></list-item>
<list-item>
<p><inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM6"><mml:msub><mml:mi>y</mml:mi><mml:mn>3</mml:mn></mml:msub></mml:math></inline-formula>: &#x201C;Symptoms and Side-effects&#x201D; subscale</p></list-item>
</list>Python libraries like Pandas and Sklearn were used for data processing and model development. In this study, ten machine learning algorithms were utilized, and the errors of the trained models were compared. Ridge Regression, Lasso Regression, Elastic-Net Regression, k-Nearest Neighbors (k-NN), Decision Tree (DT), Support Vector Regression (SVR), Linear SVR (L.SVR), Random Forest (RF), Gradient Boosting (GB), and AdaBoost algorithms, were considered for developina model in estimating the subjective QoL of the subjects. Each algorithm is described below.</p>
<sec id="s2f1"><label>2.6.1.</label><title>Ridge regression</title>
<p>Ridge Regression is a parameter estimation method used to address that addresses the collinearity problem frequently arising in multiple linear regression (<xref ref-type="bibr" rid="B23">23</xref>). Ridge Regression&#x0027;s coefficients minimize the sum of squared penalized residuals (<xref ref-type="bibr" rid="B24">24</xref>). L2 regularization is used in Ridge Regression.<disp-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="UDM1"><mml:munder><mml:mrow><mml:mrow><mml:mi mathvariant="normal">min</mml:mi></mml:mrow></mml:mrow><mml:mi>&#x03C9;</mml:mi></mml:munder><mml:mo>&#x2061;</mml:mo><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mrow><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mrow><mml:mi>X</mml:mi><mml:mi>&#x03C9;</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo fence="false" stretchy="false">|</mml:mo><mml:msubsup><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mn>2</mml:mn><mml:mn>2</mml:mn></mml:msubsup><mml:mo>+</mml:mo><mml:mi>&#x03B1;</mml:mi></mml:mrow><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mi>&#x03C9;</mml:mi><mml:mo fence="false" stretchy="false">|</mml:mo><mml:msubsup><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mn>2</mml:mn><mml:mn>2</mml:mn></mml:msubsup></mml:math></disp-formula></p>
</sec>
<sec id="s2f2"><label>2.6.2.</label><title>Lasso regression</title>
<p>Lasso Regression minimizes the residual sum of squares subject to the sum of the absolute value of the coefficients being less than a constant. Because of the nature of this constraint, it tends to produce some coefficients that are exactly 0 and hence gives interpretable models (<xref ref-type="bibr" rid="B25">25</xref>). L1 regularization is used in Lasso Regression. The objective function to minimize is (<xref ref-type="bibr" rid="B26">26</xref>):<disp-formula><label>(2)</label><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="DM2"><mml:munder><mml:mrow><mml:mrow><mml:mi mathvariant="normal">min</mml:mi></mml:mrow></mml:mrow><mml:mi>w</mml:mi></mml:munder><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mn>2</mml:mn><mml:msub><mml:mi>n</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">samples</mml:mi></mml:mrow></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mrow><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mrow><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mrow><mml:mi>X</mml:mi><mml:mi>w</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo fence="false" stretchy="false">|</mml:mo><mml:msubsup><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mn>2</mml:mn><mml:mn>2</mml:mn></mml:msubsup><mml:mo>+</mml:mo><mml:mi>&#x03B1;</mml:mi></mml:mrow><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mi>w</mml:mi><mml:mo fence="false" stretchy="false">|</mml:mo><mml:msub><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mn>1</mml:mn></mml:msub></mml:mstyle></mml:math></disp-formula></p>
</sec>
<sec id="s2f3"><label>2.6.3.</label><title>Elastic-Net regression</title>
<p>The Elastic-Net is particularly useful when the number of predictors (<italic>p</italic>) is much bigger than the number of observations (<italic>n</italic>). In contrast, the Lasso Regression does not have a satisfactory variable selection method to handle the <italic>p&#x2009;&#x003E;</italic>&#x2009;<italic>n</italic> case. Therefore, Elastic-Net was proposed as an improved version of Lasso Regression to analyze high-dimensional data. The L1 part of the Elastic-Net performs automatic variable selection, while the L2 part stabilizes the solution paths. Hence, this method improves the prediction (<xref ref-type="bibr" rid="B27">27</xref>). The objective function to be minimized is (<xref ref-type="bibr" rid="B28">28</xref>):<disp-formula><label>(3)</label><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="DM3"><mml:munder><mml:mrow><mml:mrow><mml:mi mathvariant="normal">min</mml:mi></mml:mrow></mml:mrow><mml:mi>w</mml:mi></mml:munder><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mn>2</mml:mn><mml:msub><mml:mi>n</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">samples</mml:mi></mml:mrow></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mrow><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mrow><mml:mi>X</mml:mi><mml:mi>w</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo fence="false" stretchy="false">|</mml:mo><mml:msubsup><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mn>2</mml:mn><mml:mn>2</mml:mn></mml:msubsup><mml:mo>+</mml:mo><mml:mi>&#x03B1;</mml:mi><mml:mi>&#x03C1;</mml:mi><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mrow><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mi>w</mml:mi><mml:mo fence="false" stretchy="false">|</mml:mo><mml:msub><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mn>1</mml:mn></mml:msub><mml:mo>+</mml:mo><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mfrac><mml:mrow><mml:mi>&#x03B1;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:mi>&#x03C1;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:mfrac></mml:mrow></mml:mstyle></mml:mrow><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mi>w</mml:mi><mml:mo fence="false" stretchy="false">|</mml:mo><mml:msubsup><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mn>2</mml:mn><mml:mn>2</mml:mn></mml:msubsup></mml:mstyle></mml:math></disp-formula></p>
</sec>
<sec id="s2f4"><label>2.6.4.</label><title>k-Nearest neighbors (k-NN)</title>
<p>k-NN algorithm for regression is a supervised learning approach. It predicts the target based on the similarity with other available cases. The similarity is calculated using the distance measure, with Euclidian distance being the most common approach. Predictions are made by finding the <italic>k</italic> most similar instances, i.e., the neighbors, of the testing point, from the entire dataset (<xref ref-type="bibr" rid="B29">29</xref>).</p>
</sec>
<sec id="s2f5"><label>2.6.5.</label><title>Decision tree (DT)</title>
<p>The Decision Trees algorithm is a non-parametric supervised learning method used for classification and regression.</p>
<p>In Decision Trees, a hierarchical tree structure consisting of Yes-No questions is learned. The disadvantage of decision trees is that they are prone to over-fitting and tend to be less versatile (<xref ref-type="bibr" rid="B30">30</xref>).</p>
</sec>
<sec id="s2f6"><label>2.6.6.</label><title>Support vector regression (SVR)</title>
<p>Instead of minimizing the observed training error, Support Vector Regression (SVR) attempts to minimize the generalization error bound to achieve generalized performance. SVR&#x0027;s concept is based on the computation of a linear regression function in a high-dimensional feature space where the input data are mapped <italic>via</italic> a nonlinear function (<xref ref-type="bibr" rid="B31">31</xref>).</p>
</sec>
<sec id="s2f7"><label>2.6.7.</label><title>Linear SVR (L.SVR)</title>
<p>Support Vector Regression (SVR) and Support Vector Classification (SVC) are time-consuming when using kernels. It has been demonstrated that Linear SVC and L. SVR generate models equivalent to kernel-SVR efficiently (<xref ref-type="bibr" rid="B32">32</xref>).</p>
</sec>
<sec id="s2f8"><label>2.6.8.</label><title>Random forest (RF)</title>
<p>Random Forest is one of the methods to deal with the problem of over-fitting to the training data in the DT algorithm. RFs consist of tree-structured classifiers &#x007B;<italic>h</italic> (<inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM7"><mml:mrow><mml:mi mathvariant="normal">x</mml:mi></mml:mrow></mml:math></inline-formula>, <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM8"><mml:mi>k</mml:mi></mml:math></inline-formula>), <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM9"><mml:mi>k</mml:mi></mml:math></inline-formula>&#x2009;&#x003D;&#x2009;1, &#x2026;&#x007D; where the &#x007B;<inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM10"><mml:mi>k</mml:mi></mml:math></inline-formula>&#x007D; are identically independent distributed random vectors. Each tree cats a vote for the most popular class at input <italic>x</italic> (<xref ref-type="bibr" rid="B33">33</xref>).</p>
</sec>
<sec id="s2f9"><label>2.6.9.</label><title>Gradient boosting (GB)</title>
<p>The Gradient Boosting algorithm constructs additive regression models by sequentially fitting a simple parameterized function (base learner) to current &#x201C;pseudo&#x201D;-residuals by least squares at each iteration. The execution speed and approximation accuracy of GB can be greatly improved by incorporating randomization into the procedure (<xref ref-type="bibr" rid="B34">34</xref>).</p>
</sec>
<sec id="s2f10"><label>2.6.10.</label><title>Adaboost</title>
<p>Boosting is an approach to machine learning based on combining many relatively weak and i inaccurate rules to create a highly accurate prediction rule (<xref ref-type="bibr" rid="B35">35</xref>). The core principle of the AdaBoost regressor is to learn a sequence of weak regressors with high bias error but with low variance error. Repeatedly reweighted training instances are done based on the prediction error of each boosting iteration (<xref ref-type="bibr" rid="B36">36</xref>).</p>
</sec>
<sec id="s2f11"><label>2.6.11.</label><title>Nested cross-validation approach</title>
<p>The Nested Cross-validation (CV) approach was used to compare machine learning algorithms on smaller subsets of the dataset (<xref ref-type="bibr" rid="B37">37</xref>) (<xref ref-type="fig" rid="F4">Figure&#x00A0;4</xref>). Conventional CV uses the same data to compare different algorithms and evaluate the model&#x0027;s performance. The conventional CV approach leads to data leakage and over-fitting. On the other hand, the nested CV splits the data into training (&#x2460;), validation (&#x2461;), and test sets (&#x2462;) multiple times. First, the outer loop divides the entire dataset into train and test sets. The outer loop is in charge of evaluating the model performance using the test set (&#x2462;). Then, the inner loop divides the train set further into smaller training (&#x2460;) and validation sets (&#x2461;). In the inner loop, the best model was selected among the ten algorithms by comparing the average RMSE and MAE values. The Leave-One-Out method was used for partitioning the dataset. The default hyperparameters for each algorithm were kept during the entire model development. These default hyperparameters were provided in the Scikit learn library (see <xref ref-type="app" rid="app1">Appendix</xref>).</p>
<fig id="F4" position="float"><label>Figure 4</label>
<caption><p>Model development.</p></caption>
<graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="fresc-04-1121034-g004.tif"/>
</fig>
<p>For each of the ten algorithms, the error between ground truth and validation data was calculated using RMSE (Root Mean Squared Error) and MAE (Mean Absolute Error). RMSE is characterized by a strict evaluation of the error between the ground truth and the estimate using the squared form. The lower the RMSE is, the better the estimates of the model. On the other hand, MAE is the mean of the absolute difference between the ground truth and the estimated values. The lower the MAE is, the better the estimates of the model.</p>
<p>RMSE is computed as follows.<disp-formula id="disp-formula2"><label>(4)</label><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="DM4"><mml:mrow><mml:mi>R</mml:mi><mml:mi>M</mml:mi><mml:mi>S</mml:mi><mml:mi>E</mml:mi></mml:mrow><mml:mspace width="thinmathspace" /><mml:mo>=</mml:mo><mml:mspace width="thinmathspace" /><mml:msqrt><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mi>n</mml:mi></mml:mfrac></mml:mrow><mml:msubsup><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msubsup><mml:mrow><mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:mover><mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>&#x005E;</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:mstyle></mml:msqrt></mml:math></disp-formula></p>
<p>On the other hand, MAE is calculated as follows.<disp-formula id="disp-formula3"><label>(5)</label><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="DM5"><mml:mrow><mml:mi>M</mml:mi><mml:mi>A</mml:mi><mml:mi>E</mml:mi></mml:mrow><mml:mspace width="thinmathspace" /><mml:mo>=</mml:mo><mml:mspace width="thinmathspace" /><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mi>n</mml:mi></mml:mfrac></mml:mrow><mml:msubsup><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>n</mml:mi></mml:msubsup><mml:mrow><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:mrow><mml:mover><mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>&#x005E;</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mo fence="false" stretchy="false">|</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></disp-formula>where <italic>n</italic> is the total number of data, <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM11"><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> is the actual value, <inline-formula><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM12"><mml:msub><mml:mrow><mml:mover><mml:mi>y</mml:mi><mml:mo stretchy="false">&#x005E;</mml:mo></mml:mover></mml:mrow><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> denotes the predicted value. Since this study estimates the subjective QoL thru three subscale scores, the average RMSE and MAE over these three subscale scores were also calculated. The average RMSE and MAE over three subscales were used during the model comparison.</p>
</sec>
</sec>
<sec id="s2g"><label>2.7.</label><title>Speech feature importance by SHAP value</title>
<p>In recent years, the interpretability of models has become more important than their accuracy. SHApley Additive exPlanations (SHAP) is a unified framework for interpreting predictions, allowing us to understand each feature&#x0027;s importance for the prediction (<xref ref-type="bibr" rid="B38">38</xref>).</p>
<p>Therefore, in this study, the SHAP value helps identify which of the three speech features contributes to the model.</p>
</sec>
<sec id="s2h"><label>2.8.</label><title>Model development and evaluation to estimate scale scores from longitudinal measurements</title>
<p>QoL scores are inferred to change over time depending on the subject&#x0027;s condition. Therefore, we selected six subjects out of 18 subjects and conducted the second measurement after an average of 54.333 days (S.D.&#x2009;&#x003D;&#x2009;24.426). The data from the first 18 subjects were used as training data. For each of the ten models (using the same machine learning algorithm as in 2.6), the mean RMSE values for the three scales were compared using the validation data. The best model was used. Next, we evaluated the model using the scale scores of the six participants as unseen test data.</p>
</sec>
</sec>
<sec id="s3" sec-type="results"><label>3.</label><title>Results</title>
<p>This section describes the best models and evaluations of the ten algorithms selected for the cross-sectional and longitudinal measurements. Finally, we discuss the speech features that contributed to the model development.</p>
<sec id="s3a"><label>3.1.</label><title>Model comparison on validation set for cross-sectional measurement</title>
<p>First, the mean subscale scores among the 18 subjects (<xref ref-type="table" rid="T3">Table&#x00A0;3</xref>) were 45.000 for the &#x201C;Psychosocial&#x201D; subscale, 49.389 for the &#x201C;Motivation and Energy&#x201D; subscale, and 27.944 for the &#x201C;Symptoms and Side-effects&#x201D; subscale. These subscale scores were obtained from the subjects&#x0027; JSQLS responses. The last subscale had the lowest score among the three subscales, which suggests that the subjects of this study had a good QoL concerning their symptoms and subsequent side effects. However, the minimum and maximum scores for the &#x201C;Symptoms and Side Effects&#x201D; scale were 0 and 78, respectively, indicating significant individual differences.</p>
<table-wrap id="T3" position="float"><label>Table 3</label>
<caption><p>RMSE and MAE scores for each scale on cross-sectional measurements.</p></caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th align="left" valign="top">Scale</th>
<th align="center" valign="top">Mean</th>
<th align="center">Std. Dev.</th>
<th align="center" valign="top">Variance</th>
<th align="center" valign="top">Median</th>
<th align="center" valign="top">Range</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left">Psychosocial</td>
<td align="center">45.000</td>
<td align="center">17.057</td>
<td align="center">274.778</td>
<td align="center">46.000</td>
<td align="center">15.000&#x2013;80.000</td>
</tr>
<tr>
<td align="left">Motivation and Energy</td>
<td align="center">49.389</td>
<td align="center">16.288</td>
<td align="center">250.571</td>
<td align="center">50.000</td>
<td align="center">21.000&#x2013;82.000</td>
</tr>
<tr>
<td align="left">Symptoms and Side-effects</td>
<td align="center">27.944</td>
<td align="center">18.031</td>
<td align="center">307.053</td>
<td align="center">23.500</td>
<td align="center">0.000&#x2013;78.000</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Then, the ten algorithms were compared using the validation set produced in the inner loop. The mean RMSE and MAE values were calculated and ranked. With this method, the trained k-NN algorithm produced the lowest mean RMSE and MAE of 13.433 (SD&#x2009;&#x003D;&#x2009;10.206, <italic>n</italic>&#x2009;&#x003D;&#x2009;18). The average RMSE and MAE values for the validation data of the other models were in the following order (the mean values of RMSE and MAE are equal, and therefore one value is shown): SVR: 13.697, RF: 13.697, GB: 14.263, AdaBoost: 14.663, DT: 15.973, L.SVR: 21.466, Ridge: 21.496, ElasticNet: 25.976, Lasso: 28.975 (<xref ref-type="table" rid="T4">Table&#x00A0;4</xref>).</p>
<table-wrap id="T4" position="float"><label>Table 4</label>
<caption><p>The average RMSE and MAE values for the validation data.</p></caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="center"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th align="left" valign="top">Models</th>
<th align="center" valign="top">Average RMSE and MAE scores</th>
<th align="center" valign="top">S.D.</th>
</tr>
<tr>
<td align="left">K-NN</td>
<td align="center">13.443</td>
<td align="center">10.206</td>
</tr>
</thead>
<tbody>
<tr>
<td align="left">SVR</td>
<td align="center">13.697</td>
<td align="center">9.580</td>
</tr>
<tr>
<td align="left">RF</td>
<td align="center">13.697</td>
<td align="center">9.346</td>
</tr>
<tr>
<td align="left">GB</td>
<td align="center">14.263</td>
<td align="center">7.715</td>
</tr>
<tr>
<td align="left">AdaBoost</td>
<td align="center">14.663</td>
<td align="center">9.391</td>
</tr>
<tr>
<td align="left">DT</td>
<td align="center">15.973</td>
<td align="center">8.156</td>
</tr>
<tr>
<td align="left">L. SVR</td>
<td align="center">21.466</td>
<td align="center">12.217</td>
</tr>
<tr>
<td align="left">Ridge</td>
<td align="center">21.496</td>
<td align="center">12.242</td>
</tr>
<tr>
<td align="left">Elastic-Net</td>
<td align="center">25.976</td>
<td align="center">14.460</td>
</tr>
<tr>
<td align="left">Lasso</td>
<td align="center">28.975</td>
<td align="center">18.129</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3b"><label>3.2.</label><title>Model comparison on test set for cross-sectional measurement</title>
<p>The selected best model (k-NN) was evaluated using the test set produced in the outer loop. The training resulted in a mean RMSE of 14.361 (SD&#x2009;&#x003D;&#x2009;0.674, <italic>n</italic>&#x2009;&#x003D;&#x2009;18) and a mean MAE of 10.9510 (SD&#x2009;&#x003D;&#x2009;0.6347, <italic>n</italic>&#x2009;&#x003D;&#x2009;18). The trained k-NN model produced a mean RMSE and MAE of 13.304 (SD&#x2009;&#x003D;&#x2009;10.392, <italic>n</italic>&#x2009;&#x003D;&#x2009;18) on the test set. The mean test RMSE was better than during training, and this observation can also be seen in 12 out of the 18 folds (<xref ref-type="table" rid="T5">Table&#x00A0;5</xref>).</p>
<table-wrap id="T5" position="float"><label>Table 5</label>
<caption><p>Evaluation of training and test data with k-NN.</p></caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th align="left">Fold</th>
<th align="center">1</th>
<th align="center">2</th>
<th align="center">3</th>
<th align="center">4</th>
<th align="center">5</th>
<th align="center">6</th>
<th align="center">7</th>
<th align="center">8</th>
<th align="center">9</th>
<th align="center">10</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left">Training</td>
<td align="center">14.755</td>
<td align="center">14.148</td>
<td align="center">14.671</td>
<td align="center">14.218</td>
<td align="center">14.836</td>
<td align="center">12.697</td>
<td align="center">13.886</td>
<td align="center">14.196</td>
<td align="center">12.648</td>
<td align="center">14.648</td>
</tr>
<tr>
<td align="left">Test</td>
<td align="center">3.267</td>
<td align="center">21.133</td>
<td align="center">5.933</td>
<td align="center">11.800</td>
<td align="center">8.467</td>
<td align="center">38.200</td>
<td align="center">28.333</td>
<td align="center">17.000</td>
<td align="center">32.867</td>
<td align="center">7.133</td>
</tr>
<tr>
<th align="left">Fold</th>
<th align="center">11</th>
<th align="center">12</th>
<th align="center">13</th>
<th align="center">14</th>
<th align="center">15</th>
<th align="center">16</th>
<th align="center">17</th>
<th align="center">18</th>
<th align="center">Mean</th>
<th align="center">Std. Dev.</th>
</tr>
<tr>
<td align="left">Training</td>
<td align="center">14.628</td>
<td align="center">14.906</td>
<td align="center">14.857</td>
<td align="center">14.641</td>
<td align="center">14.660</td>
<td align="center">14.623</td>
<td align="center">14.776</td>
<td align="center">14.711</td>
<td align="center">14.361</td>
<td align="center">0.674</td>
</tr>
<tr>
<td align="left">Test</td>
<td align="center">14.867</td>
<td align="center">3.400</td>
<td align="center">6.200</td>
<td align="center">12.467</td>
<td align="center">9.067</td>
<td align="center">7.600</td>
<td align="center">6.600</td>
<td align="center">5.133</td>
<td align="center">13.304</td>
<td align="center">10.392</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>&#x002A;Values are shown for RMSE and MAE.</p></fn>
</table-wrap-foot>
</table-wrap>
<p>The RMSE and MAE for each subscale were 14.644 for &#x201C;Psychosocial&#x201D;, 13.633 for &#x201C;Motivation and Energy&#x201D;, and 11.633 for &#x201C;Symptoms and Side-effects&#x201D; (<xref ref-type="table" rid="T6">Table&#x00A0;6</xref>). The RMSE and MAE for &#x201C;Symptoms and Side-effects&#x201D; had the lowest values, but the minimum and maximum values were 1.200 and 51.800, respectively, which were larger than the other scales.</p>
<table-wrap id="T6" position="float"><label>Table 6</label>
<caption><p>RMSE and MAE scores for each subscale.</p></caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th align="left" valign="top">Scale</th>
<th align="center" valign="top">Mean</th>
<th align="center" valign="top">Std. Dev.</th>
<th align="center" valign="top">Variance</th>
<th align="center" valign="top">Median</th>
<th align="center" valign="top">Range</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left">Psychosocial</td>
<td align="center">14.644</td>
<td align="center">10.568</td>
<td align="center">111.691</td>
<td align="center">11.100</td>
<td align="center">0.200&#x2212;36.600</td>
</tr>
<tr>
<td align="left">Motivation and Energy</td>
<td align="center">13.633</td>
<td align="center">10.323</td>
<td align="center">106.570</td>
<td align="center">12.300</td>
<td align="center">1.200&#x2212;34.200</td>
</tr>
<tr>
<td align="left">Symptoms and Side-effects</td>
<td align="center">11.633</td>
<td align="center">13.380</td>
<td align="center">179.041</td>
<td align="center">5.600</td>
<td align="center">1.200&#x2212;51.800</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In each fold, RMSE and MAE were above 10 in 8 folds. Among them, fold 6 had the highest RMSE and MAE (38.200) and the highest scale scores (Psychosocial: 80, Motivation and Energy: 82, Symptoms and Side-effects: 78). Similarly, fold 9 had RMSE and MAE of 32.867 and lowest scale scores (Psychosocial: 15, Motivation and Energy: 21, Symptoms and side- effects: 0) (<xref ref-type="table" rid="T7">Table&#x00A0;7</xref>). There were 10 folds that could be estimated with RMSE and MAE less than 10. Among them, fold 1 had the lowest value of 3.267 (<xref ref-type="table" rid="T8">Table&#x00A0;8</xref>). The RMSE and MAE for each fold were then divided into two groups, above and less than 10, to determine whether there was a significant difference between the ground truth and the estimates for each group; for the group with RMSE and MAE above 10 (<xref ref-type="fig" rid="F5">Figure&#x00A0;5A</xref>), &#x201C;Psychosocial&#x201D; <italic>p</italic>&#x2009;&#x003D;&#x2009;0.724, &#x201C;Motivation and Energy&#x201D; <italic>p</italic>&#x2009;&#x003D;&#x2009;0.724, and &#x201C;Symptoms and Side-effects&#x201D; <italic>p</italic>&#x2009;&#x003D;&#x2009;0.535. In the group with RMSE and MAE less than ten (<xref ref-type="fig" rid="F5">Figure&#x00A0;5B</xref>), &#x201C;Psychosocial&#x201D; <italic>p</italic>&#x2009;&#x003D;&#x2009;0.724, &#x201C;Motivation and Energy&#x201D; <italic>p</italic>&#x2009;&#x003D;&#x2009;0.477, &#x201C;Symptoms and Side-effects&#x201D; <italic>p</italic>&#x2009;&#x003D;&#x2009;0.929. There were no statistically significant differences between the ground truth and the estimates.</p>
<fig id="F5" position="float"><label>Figure 5</label>
<caption><p>(<bold>A</bold>) Examination of significant differences between ground truth and estimates of RMSE and MAE above 10. &#x002A;P, psychosocial; M, motivation and energy; S, symptoms and side&#x2013;effects. (<bold>B</bold>) Examination of significant differences between ground truth and estimates of RMSE and MAE less than 10. &#x002A;P, psychosocial; M, motivation and Energy; S, symptoms and side&#x2013;effects.</p></caption>
<graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="fresc-04-1121034-g005.tif"/>
</fig>
<table-wrap id="T7" position="float"><label>Table 7</label>
<caption><p>Ground truth and estimated values for each subject with RMSE above 10.</p></caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="right"/>
<col align="right"/>
<col align="right"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th align="left" rowspan="2" valign="top">Fold</th>
<th align="center" colspan="3">Ground truth</th>
<th align="center" colspan="3">Estimated value</th>
<th align="center" rowspan="2" valign="top">RMSE</th>
<th align="center" rowspan="2" valign="top">MAE</th>
</tr>
<tr>
<th align="left" valign="top">Psychosocial</th>
<th align="center">Motivation and Energy</th>
<th align="center">Symptoms and Side-effects</th>
<th align="center" valign="top">Psychosocial</th>
<th align="center">Motivation and Energy</th>
<th align="center">Symptoms and Side-effects</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left">2</td>
<td align="center">27.000</td>
<td align="center">25.000</td>
<td align="center">38.000</td>
<td align="center">49.200</td>
<td align="center">51.400</td>
<td align="center">23.200</td>
<td align="center">21.133</td>
<td align="center">21.133</td>
</tr>
<tr>
<td align="left">4</td>
<td align="center">28.000</td>
<td align="center">39.000</td>
<td align="center">25.000</td>
<td align="center">49.200</td>
<td align="center">51.400</td>
<td align="center">23.200</td>
<td align="center">11.800</td>
<td align="center">11.800</td>
</tr>
<tr>
<td align="left">6</td>
<td align="center">80.000</td>
<td align="center">82.000</td>
<td align="center">78.000</td>
<td align="center">47.200</td>
<td align="center">52.000</td>
<td align="center">26.200</td>
<td align="center">38.200</td>
<td align="center">38.200</td>
</tr>
<tr>
<td align="left">7</td>
<td align="center">62.000</td>
<td align="center">79.000</td>
<td align="center">53.000</td>
<td align="center">44.800</td>
<td align="center">44.800</td>
<td align="center">19.400</td>
<td align="center">28.333</td>
<td align="center">28.333</td>
</tr>
<tr>
<td align="left">8</td>
<td align="center">72.000</td>
<td align="center">64.000</td>
<td align="center">22.000</td>
<td align="center">40.400</td>
<td align="center">46.400</td>
<td align="center">23.800</td>
<td align="center">17.000</td>
<td align="center">17.000</td>
</tr>
<tr>
<td align="left">9</td>
<td align="center">15.000</td>
<td align="center">21.000</td>
<td align="center">0.000</td>
<td align="center">51.600</td>
<td align="center">52.200</td>
<td align="center">30.800</td>
<td align="center">32.867</td>
<td align="center">32.867</td>
</tr>
<tr>
<td align="left">11</td>
<td align="center">62.000</td>
<td align="center">61.000</td>
<td align="center">44.000</td>
<td align="center">43.200</td>
<td align="center">48.600</td>
<td align="center">30.600</td>
<td align="center">14.867</td>
<td align="center">14.867</td>
</tr>
<tr>
<td align="left">14</td>
<td align="center">55.000</td>
<td align="center">61.000</td>
<td align="center">22.000</td>
<td align="center">38.400</td>
<td align="center">41.400</td>
<td align="center">23.200</td>
<td align="center">12.467</td>
<td align="center">12.467</td>
</tr>
<tr>
<td align="left">Median of all folds<sup>a</sup></td>
<td align="center">46.000</td>
<td align="center">50.000</td>
<td align="center">24.000</td>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p><sup>a</sup>Median of all folds in all scales.</p></fn>
</table-wrap-foot>
</table-wrap>
<table-wrap id="T8" position="float"><label>Table 8</label>
<caption><p>Ground truth and estimated values for each subject with RMSE less than 10.</p></caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th align="left" rowspan="2" valign="top">Fold</th>
<th align="center" colspan="3" valign="top">Ground truth</th>
<th align="center" colspan="3" valign="top">Estimated value</th>
<th align="center" rowspan="2" valign="top">RMSE</th>
<th align="center" rowspan="2" valign="top">MAE</th>
</tr>
<tr>
<th align="left" valign="top">Psychosocial</th>
<th align="center" valign="top">Motivation and Energy</th>
<th align="center" valign="top">Symptoms and Side-effects</th>
<th align="center" valign="top">Psychosocial</th>
<th align="center" valign="top">Motivation and Energy</th>
<th align="center" valign="top">Symptoms and Side-effects</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left">1</td>
<td align="center">47.000</td>
<td align="center">50.000</td>
<td align="center">22.000</td>
<td align="center">45.200</td>
<td align="center">46.400</td>
<td align="center">26.400</td>
<td align="center">3.267</td>
<td align="center">3.267</td>
</tr>
<tr>
<td align="left">3</td>
<td align="center">50.000</td>
<td align="center">43.000</td>
<td align="center">19.000</td>
<td align="center">37.800</td>
<td align="center">39.800</td>
<td align="center">21.400</td>
<td align="center">5.933</td>
<td align="center">5.933</td>
</tr>
<tr>
<td align="left">5</td>
<td align="center">40.000</td>
<td align="center">46.000</td>
<td align="center">34.000</td>
<td align="center">49.200</td>
<td align="center">51.400</td>
<td align="center">23.200</td>
<td align="center">8.467</td>
<td align="center">8.467</td>
</tr>
<tr>
<td align="left">10</td>
<td align="center">30.000</td>
<td align="center">50.000</td>
<td align="center">13.000</td>
<td align="center">38.000</td>
<td align="center">37.800</td>
<td align="center">11.800</td>
<td align="center">7.133</td>
<td align="center">7.133</td>
</tr>
<tr>
<td align="left">12</td>
<td align="center">47.000</td>
<td align="center">57.000</td>
<td align="center">34.000</td>
<td align="center">46.800</td>
<td align="center">51.600</td>
<td align="center">29.400</td>
<td align="center">3.400</td>
<td align="center">3.400</td>
</tr>
<tr>
<td align="left">13</td>
<td align="center">52.000</td>
<td align="center">50.000</td>
<td align="center">31.000</td>
<td align="center">42.800</td>
<td align="center">45.800</td>
<td align="center">25.800</td>
<td align="center">6.200</td>
<td align="center">6.200</td>
</tr>
<tr>
<td align="left">15</td>
<td align="center">30.000</td>
<td align="center">36.000</td>
<td align="center">22.000</td>
<td align="center">40.000</td>
<td align="center">50.000</td>
<td align="center">18.800</td>
<td align="center">9.067</td>
<td align="center">9.067</td>
</tr>
<tr>
<td align="left">16</td>
<td align="center">45.000</td>
<td align="center">36.000</td>
<td align="center">9.000</td>
<td align="center">38.200</td>
<td align="center">42.000</td>
<td align="center">19.000</td>
<td align="center">7.600</td>
<td align="center">7.600</td>
</tr>
<tr>
<td align="left">17</td>
<td align="center">35.000</td>
<td align="center">50.000</td>
<td align="center">28.000</td>
<td align="center">42.400</td>
<td align="center">43.600</td>
<td align="center">22.000</td>
<td align="center">6.600</td>
<td align="center">6.600</td>
</tr>
<tr>
<td align="left">18</td>
<td align="center">33.000</td>
<td align="center">39.000</td>
<td align="center">9.000</td>
<td align="center">34.800</td>
<td align="center">37.800</td>
<td align="center">21.400</td>
<td align="center">5.133</td>
<td align="center">5.133</td>
</tr>
<tr>
<td align="left">Median of all folds<sup>a</sup></td>
<td align="center">46.000</td>
<td align="center">50.000</td>
<td align="center">24.000</td>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
<td align="center"/>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p><sup>a</sup>Median of all scale scores.</p></fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3c"><label>3.3.</label><title>Model comparison on validation set for longitudinal (model to estimate scores for six longitudinal measurements)</title>
<p>First, the mean subscale scores among the six subjects were 46.500 for the &#x201C;Psychosocial&#x201D; subscale, 47.667 for the &#x201C;Motivation and Energy&#x201D; subscale, and 25.000 for the &#x201C;Symptoms and Side-effects&#x201D; subscale. Similar to the results of the first measurement, the scores on the &#x201C;Symptoms and Side-effects&#x201D; subscale were the lowest.</p>
<p>The models were developed using data from 18 subjects in order to estimate the scale scores for the 6 subjects. Ten algorithms were then compared using the validation set created in the inner loop. Mean RMSE and MAE values were computed and ranked. Thus, the trained RF algorithm produced the lowest mean RMSE and MAE of 13.301 (SD&#x2009;&#x003D;&#x2009;8.870, <italic>n</italic>&#x2009;&#x003D;&#x2009;18). The mean RMSE and MAE for the validation data of the other models were in the following order (mean RMSE and MAE are equal and represent a single value): k-NN: 13.304, SVR: 13.537, GB: 13.832, AdaBoost: 15.090, DT: 15.648, L. SVR: 20. 630, Ridge: 20.637, Elastic-Net: 26.105, Lasso: 30.787.</p>
</sec>
<sec id="s3d"><label>3.4.</label><title>Model comparison on test set for longitudinal measurement</title>
<p>The RMSE and MAE for each subscale were 9.607 for the &#x201C;Psychosocial&#x201D; subscale, 4.767 for the &#x201C;Motivation and Energy&#x201D; subscale, and 9.508 for the &#x201C;Symptoms and Side-effects&#x201D; subscale (<xref ref-type="table" rid="T9">Table 9</xref>). The minimum and maximum values of RMSE and MAE for the &#x201C;Psychosocial&#x201D; subscale were 2.030 and 17.000, respectively, which were larger than the other scales.</p>
<table-wrap id="T9" position="float"><label>Table 9</label>
<caption><p>RMSE and MAE scores for each scale on longitudinal measurements.</p></caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th align="left" valign="top">Scale</th>
<th align="center" valign="top">Mean</th>
<th align="center" valign="top">Std. Dev.</th>
<th align="center" valign="top">Variance</th>
<th align="center" valign="top">Median</th>
<th align="center" valign="top">Range</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left">Psychosocial</td>
<td align="center">9.607</td>
<td align="center">6.226</td>
<td align="center">38.757</td>
<td align="center">10.085</td>
<td align="center">2.030&#x2013;17.000</td>
</tr>
<tr>
<td align="left">Motivation and Energy</td>
<td align="center">4.767</td>
<td align="center">3.990</td>
<td align="center">15.924</td>
<td align="center">5.020</td>
<td align="center">0.140&#x2013;9.060</td>
</tr>
<tr>
<td align="left">Symptoms and Side-effects</td>
<td align="center">9.508</td>
<td align="center">5.734</td>
<td align="center">32.883</td>
<td align="center">8.465</td>
<td align="center">2.770&#x2013;17.050</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The results of the first and second measurements for the six subjects showed that the scores for each scale varied between &#x2212;18 and &#x002B;14 from the first measurement (<xref ref-type="table" rid="T10">Table&#x00A0;10</xref>). The RMSE and MAE values for each fold, five out of 6 folds, were less than 10. Fold 3 had the lowest RMSE and MAE at 4.557 and fold 2 had the highest at 13.540. The three subscales of Fold 2 remained high in the two measurements.</p>
<table-wrap id="T10" position="float"><label>Table 10</label>
<caption><p>First and second measurements (ground truth), and estimation scores<sup>a</sup> from longitudinal measurements.</p></caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th align="left" rowspan="2" valign="top">Fold</th>
<th align="center" valign="top" colspan="3">First measurement</th>
<th align="center" valign="top" colspan="3">Second measurement</th>
<th align="center" rowspan="2" valign="top">RMSE</th>
<th align="center" rowspan="2" valign="top">MAE</th>
</tr>
<tr>
<th align="left" valign="top">Psychosocial</th>
<th align="center" valign="top">Motivation and Energy</th>
<th align="center" valign="top">Symptoms and Side-effects</th>
<th align="center" valign="top">Psychosocial</th>
<th align="center" valign="top">Motivation and Energy</th>
<th align="center" valign="top">Symptoms and Side-effects</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left">1</td>
<td>30.000</td>
<td>50.000</td>
<td>13.000</td>
<td>37.000 (&#x002B;7)</td>
<td>36.000 (&#x002B;14)</td>
<td>22.000 (&#x002B;9)</td>
<td align="center">7.380</td>
<td align="center">7.380</td>
</tr>
<tr>
<td align="left">2</td>
<td>62.000</td>
<td>61.000</td>
<td>44.000</td>
<td>60.000 (&#x2212;2)</td>
<td>54.000 (&#x2212;7)</td>
<td>47.000 (&#x002B;3)</td>
<td align="center">13.540</td>
<td align="center">13.540</td>
</tr>
<tr>
<td align="left">3</td>
<td>47.000</td>
<td>57.000</td>
<td>34.000</td>
<td>40.000 (&#x2212;7)</td>
<td>54.000 (&#x2212;3)</td>
<td>25.000 (&#x2212;9)</td>
<td align="center">4.577</td>
<td align="center">4.577</td>
</tr>
<tr>
<td align="left">4</td>
<td>52.000</td>
<td>50.000</td>
<td>31.000</td>
<td>35.000 (&#x2212;17)</td>
<td>46.000 (&#x2212;4)</td>
<td>28.000 (&#x2212;3)</td>
<td align="center">6.513</td>
<td align="center">6.513</td>
</tr>
<tr>
<td align="left">5</td>
<td>55.000</td>
<td>61.000</td>
<td>22.000</td>
<td>58.000 (&#x002B;3)</td>
<td>43.000 (&#x2212;18)</td>
<td>16.000 (&#x2212;6)</td>
<td align="center">8.360</td>
<td align="center">8.360</td>
</tr>
<tr>
<td align="left">6</td>
<td>45.000</td>
<td>36.000</td>
<td>9.000</td>
<td>47.000 (&#x002B;2)</td>
<td>43.000 (&#x002B;7)</td>
<td>3.000 (&#x2212;6)</td>
<td align="center">7.393</td>
<td align="center">7.393</td>
</tr>
<tr>
<td align="left">Median</td>
<td>50.000</td>
<td>54.000</td>
<td>27.000</td>
<td>44.000</td>
<td>45.000</td>
<td>24.000</td>
<td align="center"/>
<td align="center"/>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p><sup>a</sup>RMSE and MAE are calculated from the ground truth and estimated values of the second measurement.</p></fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3e"><label>3.5.</label><title>Speech feature importance to the estimation of scale scores</title>
<p>Speech features contributing to scoring estimation for each scale were identified by SHAP values. MFCC1 was selected as the most important speech feature for model development, followed by Mel-spectogram10.</p>
</sec>
</sec>
<sec id="s4" sec-type="discussion"><label>4.</label><title>Discussion</title>
<p>Scale score estimation results from cross-sectional and longitudinal measurements and the speech features that contributed to the model development will be discussed based on each result. Finally, we discussed the challenges and future work for this study.</p>
<sec id="s4a"><label>4.1.</label><title>Estimation of scale scores by cross-sectional measurement</title>
<p>Comparing the mean RMSE of the three scale scores among the ten algorithms, the k-NN was the best, with 13.433. The RMSE and MAE of the model with k-NN were 14.361 for the training data and 13.304 for the test data. Better test scores than the training suggest that the trained k-NN model could estimate the subscale scores on unseen subjects.</p>
<p>In each fold, the closer the ground truth was to the median, the lower the RMSE and MAE. On the other hand, 8 folds had RMSE and MAE above 10. Among them, fold6 had a high RMSE and MAE of 38.200 (<xref ref-type="table" rid="T7">Table&#x00A0;7</xref>). Subjects were unable to perform daily activities such as housework due to delusions. fold 9 had RMSE and MAE of 32.867. The subject was hospitalized in a psychiatric ward after the first measurement. In both cases, the scores of subjects who required the most medical intervention tended to deviate from the overall trend.</p>
</sec>
<sec id="s4b"><label>4.2.</label><title>Estimation of scale scores by longitudinal measurement</title>
<p>Comparing the average RMSE of the three scale scores among the 10 algorithms, k-NN was the best with 13.301. The RMSE and MAE for the cross-sectional scales (&#x201C;Psychosocial&#x201D; 14.644, &#x201C;Motivation, and Energy&#x201D; 13.633, &#x201C;Symptoms and Side-effects&#x201D; 11.633) were above 10 for all subscales. However, the RMSE and MAE for the model with longitudinal measures were less than 10 for &#x201C;Psychosocial&#x201D; at 9.607, &#x201C;Motivation and Energy&#x201D; at 4.767, and &#x201C;Symptoms and Side-effects&#x201D; at 9.508.</p>
<p>In addition, in the model with cross-sectional measurement, 10 out of 18 folds had RMSE less than 10, while in the model with longitudinal measurement, 5 out of 6 folds had RMSE less than 10. In developing the model with the cross-sectional measurement, 1 fold was used as test data and 17 folds as training data. On the other hand, the longitudinal measurement used all data from 18 subjects as training data, which may be partly responsible for the increase in the number of training data.</p>
</sec>
<sec id="s4c"><label>4.3.</label><title>Speech features that contributed to the estimation of scale scores</title>
<p>MFCC commonly contributed to the estimation of the scores of the three subscales. MFCC has many advantages, such as high discriminative power and noise immunity (<xref ref-type="bibr" rid="B39">39</xref>). Furthermore, MFCC can accurately characterize the vocal tract and accurately represent the phonemes produced by the vocal tract (<xref ref-type="bibr" rid="B40">40</xref>). The JSQLS response options are &#x201C;Always,&#x201D; &#x201C;Often,&#x201D; &#x201C;Sometimes,&#x201D; &#x201C;Rarely,&#x201D; and &#x201C;Never&#x201D;. Therefore, it would have been possible to use the vocalization patterns of the different choices for identification.</p>
<p>Finally, as issues and future perspectives of this study, the number of patients diagnosed with schizophrenia eligible to subjects was small at the collaborating institutions. Therefore, it is necessary to seek the participation of more subjects who use medical institutions and welfare services. In addition, the study found that the scores for &#x201C;Symptoms and Side-effects&#x201D; were the lowest among the subscales. The fact that the subjects in this study were not hospitalized patients may be a factor. Therefore, it is necessary to examine whether there is a difference in scores between hospitalized and non-hospitalized patients and to consider model building. Next, the model was developed using data from 18 subjects. Attempts were made to estimate six subjects&#x0027; scale scores for the second measurement. The baseline of the scale scores differs depending on the individual conditions. Although only two measurements were used in this study, developing individual models based on continuous measurements may be helpful. In addition, it may be a more straightforward method to estimate subjective QoL by examining the possibility of estimating the scale scores using speech features of daily conversation with conversational agents (e.g., greetings).</p>
<p>A 3-year follow-up of schizophrenia patients in a previous study found that non-remitting patients had worse QoL and increased healthcare costs than remitting patients (<xref ref-type="bibr" rid="B41">41</xref>). The results of this study are considered a severe issue in psychiatric treatment in Japan, where the readmission rate is high and the length of hospital stay is extended. As one strategy, evaluating QoL using voice features enables continuous monitoring by applications and can be applied to telemedicine.</p>
</sec>
</sec>
<sec id="s5" sec-type="conclusions"><label>5.</label><title>Conclusion</title>
<p>In this study, a model was developed to estimate the three scale scores of the Japanese Schizophrenia Quality of Life Scale (JSQLS) using speech features. The ten different machine learning algorithms were compared, with k-NN being the best. The RMSE of the training data was 14.361 and the MAE of the test data was 13.361, suggesting the generality of the model. In the estimation for scale scores on individual subjects, the RMSE and MAE were higher if the scale scores were far from the median. In this study, RMSE and MAE values were higher in subjects with psychiatric symptoms that interfered with daily life and in subjects hospitalized after the measurement. In the longitudinal measurement, a model was developed using data from 18 subjects, and scale scores were estimated for six subjects measured twice. The results showed that RF was the best, with RMSE and MAE less than ten in five of the 6 folds. The speech feature most involved in model development was MFCC, which may be the result of identifying speech patterns according to question choice. Future studies should analyze more data sets and consider model development based on longitudinal measurements of individuals.</p>
</sec>
</body>
<back>
<sec id="s7" sec-type="data-availability"><title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/Supplementary Material. Further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="s8" sec-type="ethics-statement"><title>Ethics statement</title>
<p>The studies involving human participants were reviewed and approved by Research Review Committee for Human Subjects of Kyushu Institution of Technology Graduate School of Life Science and Systems Engineering. Clinical Research Review Committee, University of Occupational and Environmental Health. The patients/participants provided their written informed consent to participate in this study.</p>
</sec>
<sec id="s9" sec-type="author-contributions"><title>Author contributions</title>
<p>YS, TN, NO, RY and TS: conceived the study. JNV: developed the conversational agent. YS, TN, and NO: selected the research subjects. YS, TN and JNV: collected the data. NO: supervised the data collection process. YS and JNV: processed and analyzed the data. YS: wrote and revised the manuscript. JNV and TS: provided feedback on the manuscript. RY and TS: reviewed and supervised the overall study progress. All authors contributed to the article and approved the submitted version.</p>
</sec>
<ack><title>Acknowledgments</title>
<p>Acknowledgments go to the subjects who participated and the staff at the facilities who assisted in data collection.</p>
</ack>
<sec id="s10" sec-type="COI-statement"><title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s11" sec-type="disclaimer"><title>Publisher&#x0027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list><title>References</title>
<ref id="B1"><label>1.</label><citation citation-type="other"><collab>Ministry of Health, Labor and Welfare</collab>. <comment>Summary of 2020 Patient Survey (2020). Available at:</comment> <ext-link ext-link-type="uri" xlink:href="https://www.mhlw.go.jp/toukei/saikin/hw/kanja/20/index.html">https://www.mhlw.go.jp/toukei/saikin/hw/kanja/20/index.html</ext-link> <comment>(Accessed October 1, 2022)</comment>.</citation></ref>
<ref id="B2"><label>2.</label><citation citation-type="other"><collab>Ministry of Health, Labour and Welfare</collab>. <comment>Vision for Reform of Mental Health and Medical Welfare (Summary) (2004). Available at:</comment> <ext-link ext-link-type="uri" xlink:href="https://www.mhlw.go.jp/topics/2004/09/dl/tp0902-1a.pdf">https://www.mhlw.go.jp/topics/2004/09/dl/tp0902-1a.pdf</ext-link> <comment>(Accessed October 1.2022)</comment>.</citation></ref>
<ref id="B3"><label>3.</label><citation citation-type="other"><collab>Ministry of Health, Labor, and Welfare</collab>. <comment>Recent Trends in Mental Health and Medical Welfare Policy (2018). Available at:</comment> <ext-link ext-link-type="uri" xlink:href="https://www.mhlw.go.jp/content/12200000/000462293.pdf">https://www.mhlw.go.jp/content/12200000/000462293.pdf</ext-link> <comment>(Accessed October 1, 2022)</comment>.</citation></ref>
<ref id="B4"><label>4.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lambert</surname><given-names>M</given-names></name><name><surname>Schimmelmann</surname><given-names>BG</given-names></name><name><surname>Naber</surname><given-names>D</given-names></name><name><surname>Eich</surname><given-names>FX</given-names></name><name><surname>Schulz</surname><given-names>H</given-names></name><name><surname>Huber</surname><given-names>CG</given-names></name><etal/></person-group> <article-title>Early- and delayed antipsychotic response and prediction of outcome in 528 severely impaired patients with schizophrenia treated with amisulpride</article-title>. <source>Pharmacopsychiatry</source>. (<year>2009</year>) <volume>42</volume>(<issue>6</issue>):<fpage>77</fpage>&#x2013;<lpage>83</lpage>. <pub-id pub-id-type="doi">10.1055/s-0029-1234105</pub-id></citation></ref>
<ref id="B5"><label>5.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hofer</surname><given-names>A</given-names></name><name><surname>Baumgartner</surname><given-names>S</given-names></name><name><surname>Edlinger</surname><given-names>M</given-names></name><name><surname>Hummer</surname><given-names>M</given-names></name><name><surname>Kemmler</surname><given-names>G</given-names></name><name><surname>Rettenbacher</surname><given-names>MA</given-names></name><etal/></person-group> <article-title>Patient outcomes in schizophrenia I: correlates with sociodemographic variables, psychopathology, and side effects</article-title>. <source>Eur Psychiatry</source>. (<year>2005</year>) <volume>5&#x2013;6</volume>:<fpage>386</fpage>&#x2013;<lpage>94</lpage>. <pub-id pub-id-type="doi">10.1016/j.eurpsy.2005.02.005</pub-id></citation></ref>
<ref id="B6"><label>6.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Halyard</surname><given-names>MY</given-names></name><name><surname>Frost</surname><given-names>MH</given-names></name><name><surname>Dueck</surname><given-names>A</given-names></name><name><surname>Sloan</surname><given-names>JA</given-names></name></person-group>. <article-title>Is the use of QOL data really any different than other medical testing?</article-title> <source>Curr Probl Cancer</source>. (<year>2006</year>) <volume>30</volume>(<issue>6</issue>):<fpage>261</fpage>&#x2013;<lpage>71</lpage>. <pub-id pub-id-type="doi">10.1016/j.currproblcancer.2006.08.004</pub-id><pub-id pub-id-type="pmid">17123878</pub-id></citation></ref>
<ref id="B7"><label>7.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Morris</surname><given-names>J</given-names></name><name><surname>Perez</surname><given-names>D</given-names></name><name><surname>McNoe</surname><given-names>B</given-names></name></person-group>. <article-title>The use of quality of life data in clinical practice</article-title>. <source>Qual Life Res</source>. (<year>1997</year>) <volume>7</volume>(<issue>1</issue>):<fpage>85</fpage>&#x2013;<lpage>91</lpage>. <pub-id pub-id-type="doi">10.1023/a:1008893007068</pub-id></citation></ref>
<ref id="B8"><label>8.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Awad</surname><given-names>AG</given-names></name><name><surname>Voruganti</surname><given-names>LN</given-names></name></person-group>. <article-title>Measuring quality of life in patients with schizophrenia: an update</article-title>. <source>Pharmacoeconomics</source>. (<year>2012</year>) <volume>30</volume>(<issue>3</issue>):<fpage>183</fpage>&#x2013;<lpage>95</lpage>. <pub-id pub-id-type="doi">10.2165/11594470-000000000-00000</pub-id><pub-id pub-id-type="pmid">22263841</pub-id></citation></ref>
<ref id="B9"><label>9.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Boyer</surname><given-names>L</given-names></name><name><surname>Auquier</surname><given-names>P</given-names></name></person-group>. <article-title>The lack of impact of quality-of-life measures in schizophrenia: a shared responsibility?</article-title> <source>Pharmacoeconomics</source>. (<year>2012</year>) <volume>30</volume>(<issue>6</issue>):<fpage>531</fpage>&#x2013;<lpage>2</lpage>. <pub-id pub-id-type="doi">10.2165/11633640-000000000-00000</pub-id><pub-id pub-id-type="pmid">22551037</pub-id></citation></ref>
<ref id="B10"><label>10.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Skantze</surname><given-names>K</given-names></name><name><surname>Malm</surname><given-names>U</given-names></name><name><surname>Dencker</surname><given-names>SJ</given-names></name><name><surname>May</surname><given-names>PR</given-names></name><name><surname>Corrigan</surname><given-names>P</given-names></name></person-group>. <article-title>Comparison of quality of life with standard of living in schizophrenic out-patients</article-title>. <source>Br J Psychiatry</source>. (<year>1992</year>) <volume>161</volume>:<fpage>797</fpage>&#x2013;<lpage>801</lpage>. <pub-id pub-id-type="doi">10.1192/bjp.161.6.797</pub-id><pub-id pub-id-type="pmid">1483165</pub-id></citation></ref>
<ref id="B11"><label>11.</label><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Narayanan</surname><given-names>A</given-names></name><name><surname>Misra</surname><given-names>A</given-names></name><name><surname>Sim</surname><given-names>KC</given-names></name><name><surname>Pundak</surname><given-names>G</given-names></name><name><surname>Tripathi</surname><given-names>A</given-names></name><name><surname>Elfeky</surname><given-names>M</given-names></name><etal/></person-group> <conf-name>Toward domain-invariant speech recognition via large scale training</conf-name>. <conf-name>(2018) IEEE spoken language technology workshop (SLT)</conf-name> (<year>2018</year>). <pub-id pub-id-type="doi">10.1109/SLT.2018.8639610</pub-id></citation></ref>
<ref id="B12"><label>12.</label><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Zeinali</surname><given-names>H</given-names></name><name><surname>Burget</surname><given-names>L</given-names></name><name><surname>&#x010C;ernock&#x00FD;</surname><given-names>JH</given-names></name></person-group>. <conf-name>A multi purpose and large scale speech corpus in Persian and English for speaker and speech recognition: the DeepMine database</conf-name>. <conf-name>IEEE automatic speech recognition and understanding workshop (ASRU)</conf-name> (<year>2019</year>). p. <fpage>397</fpage>&#x2013;<lpage>402</lpage>. <pub-id pub-id-type="doi">10.1109/ASRU46091.2019.9003882</pub-id></citation></ref>
<ref id="B13"><label>13.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Issa</surname><given-names>D</given-names></name><name><surname>Demirci</surname><given-names>MF</given-names></name><name><surname>Yazici</surname><given-names>A</given-names></name></person-group>. <article-title>Speech emotion recognition with deep convolutional neural networks</article-title>. <source>Biomed Signal Process Control</source>. (<year>2020</year>) <volume>59</volume>:<fpage>1</fpage>&#x2013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1016/j.bspc.2020.101894</pub-id></citation></ref>
<ref id="B14"><label>14.</label><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Eyben</surname><given-names>F</given-names></name><name><surname>Huber</surname><given-names>B</given-names></name><name><surname>Marchi</surname><given-names>E</given-names></name><name><surname>Schuller</surname><given-names>D</given-names></name><name><surname>Schuller</surname><given-names>B</given-names></name></person-group>. <conf-name>Real-time robust recognition of speakers&#x2019; emotions and characteristics on mobile platforms</conf-name>. <conf-name>International conference on affective computing and intelligent interaction (ACII)</conf-name> (<year>2015</year>). p. <fpage>778</fpage>&#x2013;<lpage>80</lpage>. <pub-id pub-id-type="doi">10.1109/ACII.2015.7344658</pub-id></citation></ref>
<ref id="B15"><label>15.</label><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Sandulescu</surname><given-names>V</given-names></name><name><surname>Andrews</surname><given-names>S</given-names></name><name><surname>Ellis</surname><given-names>D</given-names></name><name><surname>Dobrescu</surname><given-names>R</given-names></name><name><surname>Mozos</surname><given-names>OM</given-names></name></person-group>. <conf-name>Mobile app for stress monitoring using voice features</conf-name>. <conf-name>The 5th conference on E-health and bioengineering</conf-name> (<year>2015</year>). <pub-id pub-id-type="doi">10.1109/EHB.2015.7391411</pub-id></citation></ref>
<ref id="B16"><label>16.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname><given-names>R</given-names></name><name><surname>Mei</surname><given-names>G</given-names></name><name><surname>Zhang</surname><given-names>G</given-names></name><name><surname>Gao</surname><given-names>P</given-names></name><name><surname>Judkins</surname><given-names>T</given-names></name><name><surname>Cannizzaro</surname><given-names>M</given-names></name><etal/></person-group> <article-title>A voice-based automated system for PTSD screening and monitoring</article-title>. <source>Stud Health Technol Inform</source>. (<year>2012</year>) <volume>173</volume>:<fpage>552</fpage>&#x2013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.3233/978-1-61499-022-2-552</pub-id><pub-id pub-id-type="pmid">22357057</pub-id></citation></ref>
<ref id="B17"><label>17.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ayadi</surname><given-names>ME</given-names></name><name><surname>Kamel</surname><given-names>MS</given-names></name><name><surname>Karray</surname><given-names>F</given-names></name></person-group>. <article-title>Survey on speech emotion recognition: features, classification schemes, and databases</article-title>. <source>Pattern Recogn</source>. (<year>2011</year>) <volume>44</volume>(<issue>3</issue>):<fpage>572</fpage>&#x2013;<lpage>87</lpage>. <pub-id pub-id-type="doi">10.1016/j.patcog.2010.09.020</pub-id></citation></ref>
<ref id="B18"><label>18.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kaneda</surname><given-names>Y</given-names></name><name><surname>Imakura</surname><given-names>A</given-names></name><name><surname>Ohmori</surname><given-names>T</given-names></name></person-group>. <article-title>The schizophrenia quality of life scale Japanese version (JSQLS)</article-title>. <source>Clin Psychiatry</source>. (<year>2004</year>) <volume>46</volume>(<issue>7</issue>):<fpage>737</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.11477/mf.1405100520</pub-id></citation></ref>
<ref id="B19"><label>19.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wilkinson</surname><given-names>G</given-names></name><name><surname>Hesdon</surname><given-names>B</given-names></name><name><surname>Wild</surname><given-names>D</given-names></name><name><surname>Cookson</surname><given-names>R</given-names></name><name><surname>Farina</surname><given-names>C</given-names></name><name><surname>Sharma</surname><given-names>V</given-names></name></person-group>. <article-title>Self-report quality of life measure for people with schizophrenia: the SQLS</article-title>. <source>The Br J Psychiatry</source>. (<year>2000</year>) <volume>177</volume>(<issue>1</issue>):<fpage>42</fpage>&#x2013;<lpage>6</lpage>. <pub-id pub-id-type="doi">10.1192/bjp.177.1.42</pub-id><pub-id pub-id-type="pmid">10945087</pub-id></citation></ref>
<ref id="B20"><label>20.</label><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Bocklisch</surname><given-names>T</given-names></name><name><surname>Faulkner</surname><given-names>J</given-names></name><name><surname>Pawlowski</surname><given-names>N</given-names></name><name><surname>Nichol</surname><given-names>A</given-names></name></person-group>. <conf-name>Rasa: open source language understanding and dialogue management</conf-name>. <conf-name>NIPS workshop on conversational AI</conf-name> (<year>2017</year>). <pub-id pub-id-type="doi">10.48550/arXiv.1712.05181</pub-id></citation></ref>
<ref id="B21"><label>21.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bartsch</surname><given-names>MA</given-names></name><name><surname>Wakefield</surname><given-names>GH</given-names></name></person-group>. <article-title>Audio thumbnailing of popular music using chroma-based representations</article-title>. <source>IEEE Trans Multimedia</source>. (<year>2005</year>) <volume>7</volume>(<issue>1</issue>):<fpage>96</fpage>&#x2013;<lpage>104</lpage>. <pub-id pub-id-type="doi">10.1109/TMM.2004.840597</pub-id></citation></ref>
<ref id="B22"><label>22.</label><citation citation-type="book"><person-group person-group-type="author"><name><surname>Wakefield</surname><given-names>GH</given-names></name></person-group>. <article-title>Chromagram visualization of the singing voice</article-title>. <source>Models and analysis of vocal emissions for biomedical applications</source>. <publisher-loc>Florence</publisher-loc>: <publisher-name>Firenze University Press</publisher-name> (<year>1999</year>). p. <fpage>24</fpage>&#x2013;<lpage>9</lpage>.</citation></ref>
<ref id="B23"><label>23.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>McDonald</surname><given-names>GC</given-names></name></person-group>. <article-title>Ridge regression</article-title>. <source>WIRE Comp Stats</source>. (<year>2009</year>) <volume>1</volume>(<issue>1</issue>):93&#x2013;100. <pub-id pub-id-type="doi">10.1002/wics.14</pub-id></citation></ref>
<ref id="B24"><label>24.</label><citation citation-type="other"><collab>Scikit learn</collab>. <comment>1.1.2. Ridge regression and classification (2022). Available at:</comment> <ext-link ext-link-type="uri" xlink:href="https://scikit-learn.org/stable/modules/generated/sklearn.linear_model.Ridge.html">https://scikit-learn.org/stable/modules/generated/sklearn.linear_model.Ridge.html</ext-link> <comment>(Accessed January 8, 2023)</comment>.</citation></ref>
<ref id="B25"><label>25.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tibshirani</surname><given-names>R</given-names></name></person-group>. <article-title>Regression shrinkage and selection via the Lasso</article-title>. <source>J R Stat Soc</source>. (<year>1996</year>) <volume>58</volume>(<issue>1</issue>):<fpage>267</fpage>&#x2013;<lpage>88</lpage>. <pub-id pub-id-type="doi">10.1111/j.2517-6161.1996.tb02080.x</pub-id></citation></ref>
<ref id="B26"><label>26.</label><citation citation-type="other"><collab>Scikit learn</collab>. <comment>1.1.3. Lasso (2022). Available at:</comment> <ext-link ext-link-type="uri" xlink:href="https://scikit-learn.org/stable/modules/linear_model.html?highlight=ridge+regression#lasso">https://scikit-learn.org/stable/modules/linear_model.html?highlight&#x003D;ridge&#x002B;regression&#x0023;lasso</ext-link> <comment>(Accessed January 8, 2023)</comment>.</citation></ref>
<ref id="B27"><label>27.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zou</surname><given-names>H</given-names></name><name><surname>Hastie</surname><given-names>T</given-names></name></person-group>. <article-title>Regularization and variable selection via the elastic net</article-title>. <source>J R Stat Soc</source>. (<year>2005</year>) <volume>67</volume>(<issue>2</issue>):<fpage>301</fpage>&#x2013;<lpage>20</lpage>. <pub-id pub-id-type="doi">10.1111/j.1467-9868.2005.00503.x</pub-id></citation></ref>
<ref id="B28"><label>28.</label><citation citation-type="other"><collab>Scikit learn</collab>. <comment>1.1.5. Elastic-Net (2022). Available at:</comment> <ext-link ext-link-type="uri" xlink:href="https://scikit-learn.org/stable/modules/linear_model.html?highlight=ridge+regression#elastic-net">https://scikit-learn.org/stable/modules/linear_model.html?highlight&#x003D;ridge&#x002B;regression&#x0023;elastic-net</ext-link> <comment>(Accessed January 8, 2023)</comment>.</citation></ref>
<ref id="B29"><label>29.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bajaj</surname><given-names>P</given-names></name><name><surname>Ray</surname><given-names>R</given-names></name><name><surname>Shedge</surname><given-names>S</given-names></name><name><surname>Vidhate</surname><given-names>S</given-names></name><name><surname>Shardoor</surname><given-names>S</given-names></name></person-group>. <article-title>Sales prediction using machine learning algorithms</article-title>. <source>Int Res J Eng Technol</source>. (<year>2020</year>) <volume>7</volume>(<issue>6</issue>):<fpage>3619</fpage>&#x2013;<lpage>25</lpage>.</citation></ref>
<ref id="B30"><label>30.</label><citation citation-type="book"><person-group person-group-type="author"><name><surname>M&#x00FC;ller</surname><given-names>CA</given-names></name><name><surname>Guido</surname><given-names>S</given-names></name></person-group>. <source>Introduction to machine learning with python a guide for data scientists</source>. <publisher-loc>Sebastopol, CA: Japan</publisher-loc>: <publisher-name>O&#x2019;REILLY</publisher-name> (<year>2017</year>). <fpage>70</fpage> p.</citation></ref>
<ref id="B31"><label>31.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Basak</surname><given-names>D</given-names></name><name><surname>Pal</surname><given-names>S</given-names></name><name><surname>Patranabis</surname><given-names>DC</given-names></name></person-group>. <article-title>Support vector regression</article-title>. <source>Stat Comput</source>. (<year>2007</year>) <volume>11</volume>(<issue>10</issue>):<fpage>203</fpage>&#x2013;<lpage>24</lpage>.</citation></ref>
<ref id="B32"><label>32.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ho</surname><given-names>CH</given-names></name><name><surname>Lin</surname><given-names>CJ</given-names></name></person-group>. <article-title>Large-scale linear support vector regression</article-title>. <source>J Mach Learn Res</source>. (<year>2012</year>) <volume>13</volume>(<issue>1</issue>):<fpage>3323</fpage>&#x2013;<lpage>48</lpage>. <pub-id pub-id-type="doi">10.5555/2503308.2503348</pub-id></citation></ref>
<ref id="B33"><label>33.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Breiman</surname><given-names>L</given-names></name></person-group>. <article-title>Random forests</article-title>. <source>Mach Learn</source>. (<year>2001</year>) <volume>45</volume>:<fpage>5</fpage>&#x2013;<lpage>32</lpage>. <pub-id pub-id-type="doi">10.1023/A:1010933404324</pub-id></citation></ref>
<ref id="B34"><label>34.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schapire</surname><given-names>RE</given-names></name></person-group>. <article-title>Explaining AdaBoost</article-title>. <source>Empirical Inference</source>. (<year>2013</year>):<fpage>37</fpage>&#x2013;<lpage>52</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-642-41136-6_5</pub-id></citation></ref>
<ref id="B35"><label>35.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xiao</surname><given-names>C</given-names></name><name><surname>Chen</surname><given-names>N</given-names></name><name><surname>Hu</surname><given-names>C</given-names></name><name><surname>Wang</surname><given-names>K</given-names></name><name><surname>Gong</surname><given-names>J</given-names></name><name><surname>Chen</surname><given-names>Z</given-names></name></person-group>. <article-title>Short and mid-term sea surface temperature prediction using time-series satellite data and LSTM-AdaBoost combination approach</article-title>. <source>Remote Sens Environ</source>. (<year>2019</year>) <volume>233</volume>:1&#x2013;44. <pub-id pub-id-type="doi">10.1016/j.rse.2019.111358</pub-id></citation></ref>
<ref id="B36"><label>36.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Friedman</surname><given-names>JH</given-names></name></person-group>. <article-title>Stochastic gradient boosting</article-title>. <source>Comput Stat Data Anal</source>. (<year>2002</year>) <volume>38</volume>(<issue>4</issue>):<fpage>367</fpage>&#x2013;<lpage>78</lpage>. <pub-id pub-id-type="doi">10.1016/S0167-9473(01)00065-2</pub-id></citation></ref>
<ref id="B37"><label>37.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Raschka</surname><given-names>S</given-names></name></person-group>. <article-title>Model evaluation, model selection, and algorithm selection in machine learning</article-title>. <source>arXiv</source>. (<year>2020</year>):1&#x2013;48. <pub-id pub-id-type="doi">10.48550/arXiv.1811.12808</pub-id></citation></ref>
<ref id="B38"><label>38.</label><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Lundberg</surname><given-names>SM</given-names></name><name><surname>Lee</surname><given-names>SI</given-names></name></person-group>. <conf-name>A unified approach to interpreting model predictions</conf-name>. <conf-name>31st international conference on neural information processing systems</conf-name> (<year>2017</year>). p. <fpage>4768</fpage>&#x2013;<lpage>77</lpage></citation></ref>
<ref id="B39"><label>39.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Milton</surname><given-names>A</given-names></name><name><surname>Roy</surname><given-names>SS</given-names></name><name><surname>Selvi</surname><given-names>ST</given-names></name></person-group>. <article-title>SVM scheme for speech emotion recognition using MFCC feature</article-title>. <source>Int J Comput Appl</source>. (<year>2013</year>) <volume>69</volume>(<issue>9</issue>):<fpage>34</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.5120/11872-7667</pub-id></citation></ref>
<ref id="B40"><label>40.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Abdulmajeed</surname><given-names>NQ</given-names></name><name><surname>Khateeb</surname><given-names>BA</given-names></name><name><surname>Mohammed</surname><given-names>MA</given-names></name></person-group>. <article-title>A review on voice pathology: taxonomy, diagnosis, medical procedures and detection techniques, open challenges, limitations, and recommendations for future directions</article-title>. <source>J Intell Syst</source>. (<year>2022</year>) <volume>31</volume>(<issue>1</issue>):<fpage>855</fpage>&#x2013;<lpage>75</lpage>. <pub-id pub-id-type="doi">10.1515/jisys-2022-0058</pub-id></citation></ref>
<ref id="B41"><label>41.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Haynes</surname><given-names>VS</given-names></name><name><surname>Zhu</surname><given-names>B</given-names></name><name><surname>Stauffer</surname><given-names>VL</given-names></name><name><surname>Kinon</surname><given-names>BJ</given-names></name><name><surname>Stensland</surname><given-names>MD</given-names></name><name><surname>Xu</surname><given-names>L</given-names></name></person-group>. <article-title>Long-term healthcare costs and functional outcomes associated with lack of remission in schizophrenia: a post-hoc analysis of a prospective observational study</article-title>. <source>BMC Psychiatry</source>. (<year>2012</year>) <volume>12</volume>:1&#x2013;10. <pub-id pub-id-type="doi">10.1186/1471-244X-12-222</pub-id><pub-id pub-id-type="pmid">23216976</pub-id></citation></ref></ref-list>
<app-group><app id="app1"><title>APPENDIX Hyperparameters used in model development</title>
<table-wrap position="anchor">
<table frame="hsides" rules="groups">
<colgroup>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th valign="top" align="left">Learning Algorithms</th>
<th valign="top" align="left">Hyperparameters</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Ridge</td>
<td valign="top" align="left">alpha&#x2009;&#x003D;&#x2009;1.0 tol&#x2009;&#x003D;&#x2009;0.0001</td>
</tr>
<tr>
<td valign="top" align="left">Lasso</td>
<td valign="top" align="left">alpha&#x2009;&#x003D;&#x2009;1.0 tol&#x2009;&#x003D;&#x2009;0.0001</td>
</tr>
<tr>
<td valign="top" align="left">Elastic Net</td>
<td valign="top" align="left">alpha&#x2009;&#x003D;&#x2009;1.0 tol&#x2009;&#x003D;&#x2009;0.0001 L1 ratio&#x2009;&#x003D;&#x2009;0.5</td>
</tr>
<tr>
<td valign="top" align="left">K-nearest neighbors (k-NN)</td>
<td valign="top" align="left">n neighbors&#x2009;&#x003D;&#x2009;5 weights&#x2009;&#x003D;&#x2009;uniform<break/>leaf size&#x2009;&#x003D;&#x2009;30 metric&#x2009;&#x003D;&#x2009;Minkowski<break/>power parameter for metric, <italic>p</italic>&#x2009;&#x003D;&#x2009;2</td>
</tr>
<tr>
<td valign="top" align="left">Decision Tree (DT)</td>
<td valign="top" align="left">criterion&#x2009;&#x003D;&#x2009;squared error, min sample split&#x2009;&#x003D;&#x2009;2<break/>max depth&#x2009;&#x003D;&#x2009;None, min sample leaf&#x2009;&#x003D;&#x2009;1<break/>min weight fraction leaf&#x2009;&#x003D;&#x2009;0<break/>min impurity decrease&#x2009;&#x003D;&#x2009;0.0<break/>ccpp alpha&#x2009;&#x003D;&#x2009;0.0</td>
</tr>
<tr>
<td valign="top" align="left">Support Vector Regression (SVR)</td>
<td valign="top" align="left">kernel&#x2009;&#x003D;&#x2009;RBF, degree&#x2009;&#x003D;&#x2009;3, gamma&#x2009;&#x003D;&#x2009;scale<break/>coef0&#x2009;&#x003D;&#x2009;0.0, tol&#x2009;&#x003D;&#x2009;0.001, C&#x2009;&#x003D;&#x2009;1.0, epsilon&#x2009;&#x003D;&#x2009;0.1</td>
</tr>
<tr>
<td valign="top" align="left">Linear SVR (L.SVR)</td>
<td valign="top" align="left">epsilon&#x2009;&#x003D;&#x2009;0.0, tol&#x2009;&#x003D;&#x2009;0.0001, C&#x2009;&#x003D;&#x2009;1.0<break/>loss&#x2009;&#x003D;&#x2009;epsilon insensitive, intercept scaling&#x2009;&#x003D;&#x2009;1.0</td>
</tr>
<tr>
<td valign="top" align="left">Random Forest (RF)</td>
<td valign="top" align="left">n estimators&#x2009;&#x003D;&#x2009;100, criterion&#x2009;&#x003D;&#x2009;squared error<break/>min sample split&#x2009;&#x003D;&#x2009;2, max depth&#x2009;&#x003D;&#x2009;None<break/>min sample leaf&#x2009;&#x003D;&#x2009;1, min weight fraction leaf&#x2009;&#x003D;&#x2009;0<break/>min impurity decrease&#x2009;&#x003D;&#x2009;0.0, ccpp alpha&#x2009;&#x003D;&#x2009;0.0</td>
</tr>
<tr>
<td valign="top" align="left">AdaBoost</td>
<td valign="top" align="left">estimator&#x2009;&#x003D;&#x2009;Decision Tree Regressor, max depth&#x2009;&#x003D;&#x2009;3<break/>n estimators&#x2009;&#x003D;&#x2009;50, learning rate&#x2009;&#x003D;&#x2009;1.0, loss&#x2009;&#x003D;&#x2009;linear</td>
</tr>
<tr>
<td valign="top" align="left">Gradient Boosting</td>
<td valign="top" align="left">loss&#x2009;&#x003D;&#x2009;squared error, learning rate&#x2009;&#x003D;&#x2009;0.1<break/>n estimators&#x2009;&#x003D;&#x2009;100, subsample&#x2009;&#x003D;&#x2009;1.0<break/>criterion&#x2009;&#x003D;&#x2009;Friedman MSE, min samples split&#x2009;&#x003D;&#x2009;2<break/>min samples leaf&#x2009;&#x003D;&#x2009;1, max depth&#x2009;&#x003D;&#x2009;3, alpha&#x2009;&#x003D;&#x2009;0.9</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="table-fn1"><p>Tol, tolerance.</p></fn>
<fn id="table-fn2"><p>All hyperparameters are available on the Scikit learn website.</p></fn>
</table-wrap-foot>
</table-wrap></app>
</app-group>
</back>
</article>