<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="brief-report" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Psychiatry</journal-id>
<journal-title>Frontiers in Psychiatry</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Psychiatry</abbrev-journal-title>
<issn pub-type="epub">1664-0640</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpsyt.2024.1520173</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Psychiatry</subject>
<subj-group>
<subject>Brief Research Report</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Predicting conversion to psychosis using machine learning: response to Cannon</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Smucny</surname>
<given-names>Jason</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/164745"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Cannon</surname>
<given-names>Tyrone D.</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/112440"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Bearden</surname>
<given-names>Carrie E.</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/15888"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Addington</surname>
<given-names>Jean</given-names>
</name>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/704786"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Cadenhead</surname>
<given-names>Kristen S.</given-names>
</name>
<xref ref-type="aff" rid="aff7">
<sup>7</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/91536"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Cornblatt</surname>
<given-names>Barbara A.</given-names>
</name>
<xref ref-type="aff" rid="aff8">
<sup>8</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2260990"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Keshavan</surname>
<given-names>Matcheri</given-names>
</name>
<xref ref-type="aff" rid="aff9">
<sup>9</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/5901"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Mathalon</surname>
<given-names>Daniel H.</given-names>
</name>
<xref ref-type="aff" rid="aff10">
<sup>10</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/5890"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Perkins</surname>
<given-names>Diana O.</given-names>
</name>
<xref ref-type="aff" rid="aff11">
<sup>11</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/667552"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Stone</surname>
<given-names>William</given-names>
</name>
<xref ref-type="aff" rid="aff9">
<sup>9</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Walker</surname>
<given-names>Elaine F.</given-names>
</name>
<xref ref-type="aff" rid="aff12">
<sup>12</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/9695"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Woods</surname>
<given-names>Scott W.</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/109962"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Davidson</surname>
<given-names>Ian</given-names>
</name>
<xref ref-type="aff" rid="aff13">
<sup>13</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1781838"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Carter</surname>
<given-names>Cameron S.</given-names>
</name>
<xref ref-type="aff" rid="aff14">
<sup>14</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/4684"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Department of Psychiatry, University of California, Davis</institution>, <addr-line>Davis, CA</addr-line>, <country>United States</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Psychology, Yale University</institution>, <addr-line>New Haven, CT</addr-line>, <country>United States</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Psychiatry, Yale University</institution>, <addr-line>New Haven, CT</addr-line>, <country>United States</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Department of Psychiatry, Semel Institute for Neuroscience and Human Behavior, University of California, Los Angeles</institution>, <addr-line>Los Angeles, CA</addr-line>, <country>United States</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Biobehavioral Sciences and Psychology, Semel Institute for Neuroscience and Human Behavior, University of California, Los Angeles</institution>, <addr-line>Los Angeles, CA</addr-line>, <country>United States</country>
</aff>
<aff id="aff6">
<sup>6</sup>
<institution>Department of Psychiatry, University of Calgary</institution>, <addr-line>Calgary, AB</addr-line>, <country>Canada</country>
</aff>
<aff id="aff7">
<sup>7</sup>
<institution>Department of Psychiatry, University of North Carolina Chapel Hill</institution>, <addr-line>Chapel Hill, NC</addr-line>, <country>United States</country>
</aff>
<aff id="aff8">
<sup>8</sup>
<institution>Department of Psychiatry Research, Zucker Hillside Hospital</institution>, <addr-line>New York, NY</addr-line>, <country>United States</country>
</aff>
<aff id="aff9">
<sup>9</sup>
<institution>Department of Psychiatry, Harvard University</institution>, <addr-line>Cambridge, MA</addr-line>, <country>United States</country>
</aff>
<aff id="aff10">
<sup>10</sup>
<institution>Department of Psychiatry, University of California, San Francisco</institution>, <addr-line>San Francisco, CA</addr-line>, <country>United States</country>
</aff>
<aff id="aff11">
<sup>11</sup>
<institution>Department of Psychiatry, University of San Diego</institution>, <addr-line>San Diego, CA</addr-line>, <country>United States</country>
</aff>
<aff id="aff12">
<sup>12</sup>
<institution>Department of Psychiatry, Emory University</institution>, <addr-line>Atlanta, GA</addr-line>, <country>United States</country>
</aff>
<aff id="aff13">
<sup>13</sup>
<institution>Department of Computer Science, University of California, Davis</institution>, <addr-line>Davis, CA</addr-line>, <country>United States</country>
</aff>
<aff id="aff14">
<sup>14</sup>
<institution>Department of Psychiatry, University of California, Irvine</institution>, <addr-line>Irvine, CA</addr-line>, <country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Jennifer Larimore, Agnes Scott College, United States</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Nikolaos Smyrnis, National and Kapodistrian University of Athens, Greece</p>
<p>Ayse Ulgen, Nottingham Trent University, United Kingdom</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Jason Smucny, <email xlink:href="mailto:jsmucny@ucdavis.edu">jsmucny@ucdavis.edu</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>15</day>
<month>01</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>15</volume>
<elocation-id>1520173</elocation-id>
<history>
<date date-type="received">
<day>30</day>
<month>10</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>18</day>
<month>12</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Smucny, Cannon, Bearden, Addington, Cadenhead, Cornblatt, Keshavan, Mathalon, Perkins, Stone, Walker, Woods, Davidson and Carter</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Smucny, Cannon, Bearden, Addington, Cadenhead, Cornblatt, Keshavan, Mathalon, Perkins, Stone, Walker, Woods, Davidson and Carter</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Background</title>
<p>We previously reported that machine learning could be used to predict conversion to psychosis in individuals at clinical high risk (CHR) for psychosis with up to 90% accuracy using the North American Prodrome Longitudinal Study-3 (NAPLS-3) dataset. A definitive test of our predictive model that was trained on the NAPLS-3 data, however, requires further support through implementation in an independent dataset. In this report we tested for model generalization using the previous iteration of NAPLS-3, the NAPLS-2, using the identical machine learning algorithms employed in our previous study.</p>
</sec>
<sec>
<title>Method</title>
<p>Standard machine learning algorithms were trained to predict conversion to psychosis in clinical high risk individuals on the NAPLS-3 dataset and tested on the NAPLS-2 dataset.</p>
</sec>
<sec>
<title>Results</title>
<p>NAPLS-2 and -3 individuals significantly differed on most features used in machine learning models. All models performed above chance, with Naive Bayes and random forest methods showing the best overall performance. Importantly, however, overall performance did not match those previously observed when using only NAPLS-3 data.</p>
</sec>
<sec>
<title>Conclusion</title>
<p>The results of this study suggest that a machine learning model trained to predict conversion to psychosis on one dataset can be used to train an independent dataset. Performance on the test set was not in the range necessary for clinical application, however. Possible reasons that limited performance are discussed.</p>
</sec>
</abstract>
<kwd-group>
<kwd>schizophrenia</kwd>
<kwd>clinical high risk (CHR)</kwd>
<kwd>NAPLS</kwd>
<kwd>out of sample evaluation</kwd>
<kwd>scale of psychosis risk symptoms</kwd>
<kwd>generalizability</kwd>
</kwd-group>
<counts>
<fig-count count="0"/>
<table-count count="2"/>
<equation-count count="0"/>
<ref-count count="19"/>
<page-count count="6"/>
<word-count count="2315"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Schizophrenia</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>Background</title>
<p>In a prior priority data letter in <italic>The American Journal of Psychiatry</italic>, we reported that machine learning could be used to predict conversion to psychosis in individuals at clinical high risk (CHR) for psychosis with up to 90% accuracy using the North American Prodrome Longitudinal Study-3 (NAPLS-3) dataset (<xref ref-type="bibr" rid="B1">1</xref>). As argued by Cannon (<xref ref-type="bibr" rid="B2">2</xref>), a definitive test of our predictive model that was trained on the NAPLS-3 data requires further support through implementation in an independent dataset. In this report, in collaboration with the primary investigators of the NAPLS consortium we tested for model generalization using the previous iteration of NAPLS-3, the NAPLS-2, using the identical machine learning algorithms employed in our previous study. We used the NAPLS-2 dataset as a test set because it used mostly the same measures that were used in the NAPLS-3 to predict conversion to psychosis in a similarly-aged sample.</p>
</sec>
<sec id="s2" sec-type="materials|methods">
<title>Materials and methods</title>
<sec id="s2_1">
<title>Participants</title>
<p>NAPLS-2 and 3 are NIMH-funded studies conducted at 8 and 9 sites, respectively. All participants provided written informed consent, including parental consent for minors. The study was approved by all sites&#x2019; Institutional Review Boards.</p>
<p>Detailed descriptions of NAPLS-2 and NAPLS-3 participants, including exclusion criteria, are provided in Addington et&#xa0;al. (<xref ref-type="bibr" rid="B3">3</xref>) and Addington et&#xa0;al. (<xref ref-type="bibr" rid="B4">4</xref>), respectively. Participants were between 12 and 30 years old and were followed up to 2 years. Predictors included those used by the NAPLS-2 calculator (riskcalc.org/napls, see <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>).</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Demographic and clinical information, excluding participants with missing data (<italic>N</italic> = 64).</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center"/>
<th valign="top" align="center">NAPLS3 Mean</th>
<th valign="top" align="center">SD</th>
<th valign="top" align="center">NAPLS2 Mean</th>
<th valign="top" align="center">SD</th>
<th valign="top" align="center">NAPLS2 vs. NAPLS3 t or &#x3c7;<sup>2</sup>
</th>
<th valign="top" align="center">
<italic>p</italic>
</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">
<italic>N</italic>
</td>
<td valign="top" align="center">598</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">596</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">&#x2013;</td>
</tr>
<tr>
<td valign="top" align="left">Age in Years*</td>
<td valign="top" align="center">18.79</td>
<td valign="top" align="center">4.03</td>
<td valign="top" align="center">18.51</td>
<td valign="top" align="center">4.27</td>
<td valign="top" align="center">-1.16</td>
<td valign="top" align="center">.25</td>
</tr>
<tr>
<td valign="top" align="left">
<italic>N</italic> With First Degree Relative with Psychosis*</td>
<td valign="top" align="center">117</td>
<td valign="top" rowspan="2" align="center">&#x2013;</td>
<td valign="top" align="center">96</td>
<td valign="top" rowspan="2" align="center">&#x2013;</td>
<td valign="top" rowspan="2" align="center">2.44</td>
<td valign="top" rowspan="2" align="center">.12</td>
</tr>
<tr>
<td valign="top" align="left">
<italic>N</italic> Without First Degree Relative with Psychosis*</td>
<td valign="top" align="center">481</td>
<td valign="top" align="center">500</td>
</tr>
<tr>
<td valign="top" align="left">BACS Symbol Coding Raw Score*</td>
<td valign="top" align="center">54.74</td>
<td valign="top" align="center">13.21</td>
<td valign="top" align="center">56.80</td>
<td valign="top" align="center">13.04</td>
<td valign="top" align="center">2.72</td>
<td valign="top" align="center">.007</td>
</tr>
<tr>
<td valign="top" align="left">HVLT Raw Score*</td>
<td valign="top" align="center">26.39</td>
<td valign="top" align="center">5.21</td>
<td valign="top" align="center">25.61</td>
<td valign="top" align="center">5.15</td>
<td valign="top" align="center">-2.60</td>
<td valign="top" align="center">.009</td>
</tr>
<tr>
<td valign="top" align="left">Number of Trauma Types*</td>
<td valign="top" align="center">1.87</td>
<td valign="top" align="center">1.61</td>
<td valign="top" align="center">2.08</td>
<td valign="top" align="center">1.71</td>
<td valign="top" align="center">2.20</td>
<td valign="top" align="center">.028</td>
</tr>
<tr>
<td valign="top" align="left">Decrease in Global Social Functioning Score Over the Past Year*</td>
<td valign="top" align="center">1.03</td>
<td valign="top" align="center">.97</td>
<td valign="top" align="center">.74</td>
<td valign="top" align="center">1.04</td>
<td valign="top" align="center">-4.98</td>
<td valign="top" align="center">&lt;.001</td>
</tr>
<tr>
<td valign="top" align="left">Number of Undesirable Life Events*</td>
<td valign="top" align="center">9.56</td>
<td valign="top" align="center">4.86</td>
<td valign="top" align="center">10.47</td>
<td valign="top" align="center">5.43</td>
<td valign="top" align="center">3.06</td>
<td valign="top" align="center">.002</td>
</tr>
<tr>
<td valign="top" align="left">Rescaled** SIPS Delusions + Suspicions*</td>
<td valign="top" align="center">3.10</td>
<td valign="top" align="center">1.46</td>
<td valign="top" align="center">2.61</td>
<td valign="top" align="center">1.57</td>
<td valign="top" align="center">-5.60</td>
<td valign="top" align="center">&lt;.001</td>
</tr>
<tr>
<td valign="top" align="left">SIPS Delusions + Suspicions*</td>
<td valign="top" align="center">6.80</td>
<td valign="top" align="center">1.87</td>
<td valign="top" align="center">6.08</td>
<td valign="top" align="center">2.23</td>
<td valign="top" align="center">-6.00</td>
<td valign="top" align="center">&lt;.001</td>
</tr>
<tr>
<td valign="top" align="left">
<italic>N</italic> Converters</td>
<td valign="top" align="center">62</td>
<td valign="top" rowspan="2" align="center">&#x2013;</td>
<td valign="top" align="center">84</td>
<td valign="top" rowspan="2" align="center">&#x2013;</td>
<td valign="top" rowspan="2" align="center">3.86</td>
<td valign="top" rowspan="2" align="center">.049</td>
</tr>
<tr>
<td valign="top" align="left">
<italic>N</italic> Non-Converters</td>
<td valign="top" align="center">536</td>
<td valign="top" align="center">512</td>
</tr>
<tr>
<td valign="top" align="left">Days from Baseline to Conversion</td>
<td valign="top" align="center">278.0</td>
<td valign="top" align="center">286.1</td>
<td valign="top" align="center">219.7</td>
<td valign="top" align="center">173.4</td>
<td valign="top" align="center">-1.42</td>
<td valign="top" align="center">.16</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Numbers in parentheses represent the standard deviation unless otherwise stated. BACS, Brief Assessment of Cognition in Schizophrenia; HC, Healthy Control; HVLT, Hopkins Verbal Learning Test; SIPS, Structured Interview for Psychosis-risk Syndromes. *Included as features in machine learning models. **Rescaled such that scores 0-2 = 0, 3 = 1, 4 = 2, 5 = 3, and 6 = 4.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Consistent with prior work (<xref ref-type="bibr" rid="B5">5</xref>), conversion to psychosis for a CHR individual was defined as meeting the Presence of Psychotic Symptoms criteria: One of the five Scale of Psychosis-Risk Symptoms (SOPS) positive symptoms must reach a psychotic level of intensity (rated 6) for &#x2265; 1 hour per day for 4 days per week during the past month, and/or these symptoms seriously impact their functioning. Machine learning models were tested with and without SOPS rescaling prior to analysis [see <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref> legend for rescaling procedure, based on Cannon et&#xa0;al. (<xref ref-type="bibr" rid="B6">6</xref>)].</p>
</sec>
<sec id="s2_2">
<title>Analyses</title>
<p>Standard machine learning algorithms were employed using Weka software (University of Waikato, New Zealand) and included logistic regression, naive Bayes, a 3 kernel support vector machine, KStar, J48 decision tree, random forest (max depth 2), decision stump (with 100 iterations of AdaBoost), and multilayer perceptron (see explanation of algorithms below). Classifiers were trained using NAPLS-3 data and tested on NAPLS-2 data. Individuals with missing data were excluded from the analyses. Because of class imbalance, prior to machine learning training data for the minority (converter) class was 800% upsampled using the Synthetic Minority Oversampling Technique (SMOTE) (<xref ref-type="bibr" rid="B7">7</xref>) to match the majority (nonconverter) class. By providing more training data of the minority class, this procedure helps to refine the machine learning model decision binary and improve performance (<xref ref-type="bibr" rid="B7">7</xref>). SMOTE <italic>k</italic> (<italic>n</italic> nearest neighbors) was set to 5.</p>
</sec>
<sec id="s2_3">
<title>Naive Bayes</title>
<p>Naive Bayes (<xref ref-type="bibr" rid="B8">8</xref>) compares the probability of observing an converter/non-converter for each test data point according to the equation <italic>P</italic> (<italic>C<sub>i</sub>
</italic> | <italic>x</italic>) = (<italic>P</italic> (<italic>C<sub>i</sub>
</italic>) * <italic>p</italic> (<italic>x</italic> | <italic>C</italic>
<sub>i</sub>)/<italic>p</italic> (<italic>x</italic>) where <italic>P</italic>(<italic>C<sub>i</sub>
</italic>) is the prior probability of class <italic>C<sub>i</sub>
</italic> (e.g., converter) occurring, <italic>p</italic> (<italic>x</italic> | <italic>C</italic>
<sub>i</sub>) is the conditional probability that class <italic>C<sub>i</sub>
</italic> is associated with feature observation <italic>x</italic>, and <italic>p</italic> (<italic>x</italic>) is the marginal probability that observation <italic>x</italic> is observed (effectively constant for any given dataset). The joint model (combining all features) can then be expressed as the product of the probabilities for all features, and the algorithm classifies unseen data as converter or non-converter based on the highest probability.</p>
</sec>
<sec id="s2_4">
<title>Support vector machine</title>
<p>SVM classifiers find the maximum-margin hyperplane using only those data instances closest to the separation boundary (i.e. &#x201c;support vectors&#x201d;) to determine classification boundaries (<xref ref-type="bibr" rid="B9">9</xref>). Both linear and non-linear (using a kernel) classifications can be performed. Polykernel SVM classifiers were evaluated starting with an exponent of 1 and increasing in size until average accuracy (over all 1000 allocations of test/training data) plateaued.</p>
</sec>
<sec id="s2_5">
<title>KStar (K-nearest neighbor)</title>
<p>The K* algorithm operates by assigning new data instances to the class that occurs most frequently amongst the <italic>k</italic>-nearest data points, <italic>y<sub>j</sub>
</italic>, where <italic>j</italic> = 1,2&#x2026;<italic>k</italic> (<xref ref-type="bibr" rid="B10">10</xref>). Distance is then used to retrieve the most similar instances from the data set. The K* function is operationalized as K* (<italic>y<sub>i</sub>
</italic>,<italic>x</italic>)= -<italic>ln P</italic>*(<italic>y<sub>i</sub>
</italic>,<italic>x</italic>) where <italic>P</italic>* is the probability of all transformational paths from instance <italic>x</italic> to <italic>y</italic>, i.e., the probability <italic>x</italic> will arrive at <italic>y</italic> via a random walk in feature space.</p>
</sec>
<sec id="s2_6">
<title>AdaBoost</title>
<p>AdaBoost operates by creating multiple weak classifiers that are weighed by their effectiveness at classifying data (<xref ref-type="bibr" rid="B11">11</xref>). Initially, a classifier is created with all instances weighted equally. Next, the weights of the incorrectly predicted instances are increased. The instances that are still misclassified are then selected and their weights increased as well, and so forth. After the complete classifier is constructed, each weak classifier then casts a weighted &#x201c;vote&#x201d; as to the class membership of each set of individual test data to make a classification decision.</p>
</sec>
<sec id="s2_7">
<title>J48 decision tree</title>
<p>Decision tree classifiers operate hierarchically, with each level representing a feature (e.g., age) (<xref ref-type="bibr" rid="B12">12</xref>, <xref ref-type="bibr" rid="B13">13</xref>). Based on the value of that feature the tree either classifies immediately or passes the information to the next level of the tree. The C4.5 algorithm (<xref ref-type="bibr" rid="B12">12</xref>) was used for the J48 decision tree, which uses a measure called &#x201c;information gain&#x201d; to select each attribute at each stage. In essence, the J48 tree first chooses the feature that most effectively splits the training data into one class or another using a measure called &#x201c;information gain&#x201d; (essentially, the effectiveness of feature at classifying data). After this split, the tree then chooses the next most effective feature to split each resulting partition. The process then iteratively repeats until all training data is classified. Performance of the resulting tree is then evaluated on test data.</p>
</sec>
<sec id="s2_8">
<title>Random forest</title>
<p>A random forest is a group of decision trees made up of random partitions of training data (<xref ref-type="bibr" rid="B14">14</xref>). Each tree casts a &#x201c;vote&#x201d; as to the classification of a testing instance and votes are counted to produce the final classification.</p>
</sec>
</sec>
<sec id="s3" sec-type="results">
<title>Results</title>
<p>Demographic and clinical information for and comparisons between NAPLS-2 and NAPLS-3 participants are provided in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>. NAPLS-2 and -3 individuals significantly differed on most features used in machine learning models. Examining conversion rates, 84 out of the 596 CHR individuals in NAPLS-2 (14%) and 62 out of the 598 CHR participants in NAPLS-3 (10%) with no missing data converted over the course of the follow-up period.</p>
<p>Machine learning performance metrics for each machine learning algorithm are provided in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>. Briefly, all models performed above chance as evidenced by concordance values [receiver operating characteristic area under the curve (ROC AUC)] values over 50%. Naive Bayes and random forest showed the best overall performance (AUC). Sensitivity and negative predictive values were relatively high and specificity and positive predictive values were relatively low for all models, however. Importantly, overall performance did not match those previously observed when using only NAPLS-3 data (performance metrics from the previous study are also provided in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>). For example, when exclusively using NAPLS-3 data (including non-rescaled SIPS scores) for training and testing, the most accurate method (random forest) showed 90% accuracy with 79% sensitivity and 96% specificity (<xref ref-type="bibr" rid="B1">1</xref>). When a NAPLS-3-based random forest model was tested on NAPLS-2 data, however, accuracy, sensitivity, and specificity were reduced to 77%, 41%, and 83% (respectively) (<xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>, bottom half).</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Predicting conversion to psychosis using NAPLS-3 clinical/demographic data for training and NAPLS-2 data for testing using various machine learning methods.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Method</th>
<th valign="top" colspan="2" align="center">Acc</th>
<th valign="top" colspan="2" align="center">Sn</th>
<th valign="top" colspan="2" align="center">Sp</th>
<th valign="top" colspan="2" align="center">PPV</th>
<th valign="top" colspan="2" align="center">NPV</th>
<th valign="top" colspan="2" align="center">ROC AUC</th>
</tr>
</thead>
<tbody>
<tr>
<th valign="top" colspan="13" align="left">SIPS Rescaled</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Decision Stump with AdaBoost</td>
<td valign="top" colspan="2" align="center">79</td>
<td valign="top" colspan="2" align="center">33</td>
<td valign="top" colspan="2" align="center">87</td>
<td valign="top" colspan="2" align="center">29</td>
<td valign="top" colspan="2" align="center">89</td>
<td valign="top" colspan="2" align="center">66</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;J48 Decision Tree</td>
<td valign="top" colspan="2" align="center">82</td>
<td valign="top" colspan="2" align="center">10</td>
<td valign="top" colspan="2" align="center">94</td>
<td valign="top" colspan="2" align="center">21</td>
<td valign="top" colspan="2" align="center">86</td>
<td valign="top" colspan="2" align="center">56</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;KStar</td>
<td valign="top" colspan="2" align="center">72</td>
<td valign="top" colspan="2" align="center">20</td>
<td valign="top" colspan="2" align="center">81</td>
<td valign="top" colspan="2" align="center">15</td>
<td valign="top" colspan="2" align="center">86</td>
<td valign="top" colspan="2" align="center">52</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Logistic Regression</td>
<td valign="top" colspan="2" align="center">75</td>
<td valign="top" colspan="2" align="center">51</td>
<td valign="top" colspan="2" align="center">79</td>
<td valign="top" colspan="2" align="center">29</td>
<td valign="top" colspan="2" align="center">91</td>
<td valign="top" colspan="2" align="center">67</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;MLP*</td>
<td valign="top" colspan="2" align="center">70</td>
<td valign="top" colspan="2" align="center">36</td>
<td valign="top" colspan="2" align="center">76</td>
<td valign="top" colspan="2" align="center">20</td>
<td valign="top" colspan="2" align="center">88</td>
<td valign="top" colspan="2" align="center">60</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Naive Bayes</td>
<td valign="top" colspan="2" align="center">75</td>
<td valign="top" colspan="2" align="center">58</td>
<td valign="top" colspan="2" align="center">78</td>
<td valign="top" colspan="2" align="center">30</td>
<td valign="top" colspan="2" align="center">92</td>
<td valign="top" colspan="2" align="center">70</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Random Forest</td>
<td valign="top" colspan="2" align="center">78</td>
<td valign="top" colspan="2" align="center">38</td>
<td valign="top" colspan="2" align="center">85</td>
<td valign="top" colspan="2" align="center">29</td>
<td valign="top" colspan="2" align="center">89</td>
<td valign="top" colspan="2" align="center">70</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;SVM (3 Kernel)</td>
<td valign="top" colspan="2" align="center">70</td>
<td valign="top" colspan="2" align="center">36</td>
<td valign="top" colspan="2" align="center">76</td>
<td valign="top" colspan="2" align="center">20</td>
<td valign="top" colspan="2" align="center">88</td>
<td valign="top" colspan="2" align="center">56</td>
</tr>
<tr>
<th valign="top" align="left">SIPS Non-Rescaled</th>
<th valign="top" colspan="6" align="center">NAPLS-3 Train, NAPLS-2 Test</th>
<th valign="top" colspan="6" align="center" style="background-color:#d9d9d9">
<italic>NAPLS-3 Train, NAPLS-3 Test</italic>
</th>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Decision Stump with AdaBoost</td>
<td valign="top" align="center">79</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>71</italic>
</bold>
</td>
<td valign="top" align="center">31</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>43</italic>
</bold>
</td>
<td valign="top" align="center">87</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>88</italic>
</bold>
</td>
<td valign="top" align="center">27</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>68</italic>
</bold>
</td>
<td valign="top" align="center">88</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>73</italic>
</bold>
</td>
<td valign="top" align="center">68</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>74</italic>
</bold>
</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;J48 Decision Tree</td>
<td valign="top" align="center">81</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>85</italic>
</bold>
</td>
<td valign="top" align="center">20</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>76</italic>
</bold>
</td>
<td valign="top" align="center">91</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>90</italic>
</bold>
</td>
<td valign="top" align="center">28</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>82</italic>
</bold>
</td>
<td valign="top" align="center">88</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>87</italic>
</bold>
</td>
<td valign="top" align="center">55</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>86</italic>
</bold>
</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;KStar</td>
<td valign="top" align="center">70</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>78</italic>
</bold>
</td>
<td valign="top" align="center">21</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>77</italic>
</bold>
</td>
<td valign="top" align="center">78</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>79</italic>
</bold>
</td>
<td valign="top" align="center">14</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>69</italic>
</bold>
</td>
<td valign="top" align="center">86</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>86</italic>
</bold>
</td>
<td valign="top" align="center">57</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>84</italic>
</bold>
</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Logistic Regression</td>
<td valign="top" align="center">72</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>68</italic>
</bold>
</td>
<td valign="top" align="center">51</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>35</italic>
</bold>
</td>
<td valign="top" align="center">75</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>87</italic>
</bold>
</td>
<td valign="top" align="center">25</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>61</italic>
</bold>
</td>
<td valign="top" align="center">90</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>70</italic>
</bold>
</td>
<td valign="top" align="center">66</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>68</italic>
</bold>
</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;MLP*</td>
<td valign="top" align="center">66</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>73</italic>
</bold>
</td>
<td valign="top" align="center">57</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>48</italic>
</bold>
</td>
<td valign="top" align="center">67</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>88</italic>
</bold>
</td>
<td valign="top" align="center">22</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>70</italic>
</bold>
</td>
<td valign="top" align="center">91</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>75</italic>
</bold>
</td>
<td valign="top" align="center">64</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>77</italic>
</bold>
</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Naive Bayes</td>
<td valign="top" align="center">66</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>70</italic>
</bold>
</td>
<td valign="top" align="center">55</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>45</italic>
</bold>
</td>
<td valign="top" align="center">68</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>84</italic>
</bold>
</td>
<td valign="top" align="center">22</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>62</italic>
</bold>
</td>
<td valign="top" align="center">90</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>73</italic>
</bold>
</td>
<td valign="top" align="center">65</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>68</italic>
</bold>
</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;Random Forest</td>
<td valign="top" align="center">77</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>90</italic>
</bold>
</td>
<td valign="top" align="center">41</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>79</italic>
</bold>
</td>
<td valign="top" align="center">83</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>96</italic>
</bold>
</td>
<td valign="top" align="center">28</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>92</italic>
</bold>
</td>
<td valign="top" align="center">89</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>89</italic>
</bold>
</td>
<td valign="top" align="center">69</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>96</italic>
</bold>
</td>
</tr>
<tr>
<td valign="top" align="left">&#x2003;SVM (3 Kernel)</td>
<td valign="top" align="center">69</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>74</italic>
</bold>
</td>
<td valign="top" align="center">45</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>41</italic>
</bold>
</td>
<td valign="top" align="center">73</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>93</italic>
</bold>
</td>
<td valign="top" align="center">22</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>78</italic>
</bold>
</td>
<td valign="top" align="center">89</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>73</italic>
</bold>
</td>
<td valign="top" align="center">59</td>
<td valign="top" align="center" style="background-color:#d9d9d9">
<bold>
<italic>67</italic>
</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Performance from our previous study (<xref ref-type="bibr" rid="B1">1</xref>) using NAPLS-3 data (only) for training and testing are provided in shaded bold italic to the right of each metric for comparison (SIPS were non-rescaled only for the previous analysis). Values are percentages. Acc, Accuracy; NPV, Negative Predictive Value; PPV, Positive Predictive Value; ROC AUC, Receiver Operating Characteristics Area Under the Curve; Sn, Sensitivity; Sp, Specificity. *Multilayer perceptron (MLP) with 2 hidden layers (5 nodes in the first and 2 in the second).</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s4" sec-type="discussion">
<title>Discussion</title>
<p>The results of this study suggest that a machine learning model trained to predict conversion to psychosis on one dataset can be used to train an independent dataset. In this analysis, however, we did not achieve the high level of accuracy as in the training sample and classification success was not in the range generally considered necessary for clinical application (80% positive and negative predictive value). Notably most models in this generalization analysis struggled to identify positive cases (converters), the minority outcome in both samples, although they did show high specificity and AUC scores that were above chance.</p>
<p>One likely reason for the discrepancy in performance is that models that exclusively use NAPLS-3 data may have overfit to that dataset. Chekroud et&#xa0;al. (<xref ref-type="bibr" rid="B15">15</xref>) have recently emphasized that machine learning faces a generalizability issue in outcome prediction in studies of psychiatric disorders in that models often do not perform well when tested on unseen data. Consistent with our findings, recent prior work that also sought to predict conversion to psychosis in CHRs observed a significant reduction in performance (from 85% accuracy on the trained sample to 73% on the independent test sample) when a trained model was tested on an independent dataset (<xref ref-type="bibr" rid="B16">16</xref>). Although all algorithms here did perform above chance, the fact that they did not perform as well as a model trained within one dataset as in our previous study is in conceptual agreement with the findings of Chekroud et&#xa0;al. (<xref ref-type="bibr" rid="B15">15</xref>). An additional potential reason for poor performance is that the assumption of stationarity, core to machine learning methods including random forest, was not met since the levels of key predictor variables differed significantly across the two samples (<xref ref-type="bibr" rid="B17">17</xref>) (as shown in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>). This presents a significant challenge for researchers seeking to develop generalizable prediction tools using machine learning approaches. One approach to improving generalizability that machine learning researchers are actively investigating is the application of <italic>domain adaptation</italic> methods, to apply in cases in which source and target domains have different distributions but identical underlying predictive features (<xref ref-type="bibr" rid="B18">18</xref>). As recently reviewed by Smucny et&#xa0;al. (<xref ref-type="bibr" rid="B19">19</xref>), domain adaptation methods have achieved some success when using magnetic resonance imaging data to predict various outcomes, although only a few studies thus far have been conducted and domain adaptation procedures are still being developed and refined. Finally, it should be noted that the power of the model may have been limited by the relatively low sample size of the minority class in the training dataset (<italic>n =</italic> 62). Hence, while we are not &#x201c;there yet&#x201d; (<xref ref-type="bibr" rid="B1">1</xref>), we believe that the field should eschew nihilism regarding the application of predictive analytics to unique and powerful datasets (such as those collected by the NAPLS consortium) in pursuit of a personalized medicine approach to early intervention for young people with psychosis.</p>
</sec>
</body>
<back>
<sec id="s5" sec-type="data-availability">
<title>Data availability statement</title>
<p>The data analyzed in this study is subject to the following licenses/restrictions: The NAPLS-2 dataset is only available to NAPLS-2 investigators and their collaborators. Requests to access these datasets should be directed to <email xlink:href="mailto:tyrone.cannon@yale.edu">tyrone.cannon@yale.edu</email>. NAPLS-3 data is available for download on the National Institutes of Mental Health Data Archive.</p>
</sec>
<sec id="s6" sec-type="ethics-statement">
<title>Ethics statement</title>
<p>Ethical approval was not required for the study involving humans in accordance with the local legislation and institutional requirements. Written informed consent to participate in this study was not required from the participants or the participants&#x2019; legal guardians/next of kin in accordance with the national legislation and the institutional requirements.</p>
</sec>
<sec id="s7" sec-type="author-contributions">
<title>Author contributions</title>
<p>JS: Conceptualization, Formal analysis, Investigation, Methodology, Project administration, Resources, Validation, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. TC: Data curation, Funding acquisition, Resources, Writing &#x2013; review &amp; editing. CB: Data curation, Writing &#x2013; review &amp; editing. JA: Data curation, Writing &#x2013; review &amp; editing. KC: Data curation, Writing &#x2013; review &amp; editing. BC: Data curation, Writing &#x2013; review &amp; editing. MK: Data curation, Writing &#x2013; review &amp; editing. DM: Data curation, Writing &#x2013; review &amp; editing. DP: Data curation, Writing &#x2013; review &amp; editing. WS: Data curation, Writing &#x2013; review &amp; editing. EW: Data curation, Writing &#x2013; review &amp; editing. SW: Data curation, Writing &#x2013; review &amp; editing. ID: Supervision, Writing &#x2013; review &amp; editing. CC: Supervision, Writing &#x2013; review &amp; editing.</p>
</sec>
<sec id="s8" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This work was supported by NIMH grants K01-MH125096 (JS), R01-MH122139 (CC), U01-MH081902 (TC), P50-MH066286 (CB), U01-MH081857 (BC), U01-MH082022 (SW), U01-MH066134 (JA), U01-MH081944 (KC), U01-MH066069 (DP), R01-MH076989 (DM), and U01-MH081988 to (EW).</p>
</sec>
<ack>
<title>Acknowledgments</title>
<p>Some of the data and/or research tools used in the preparation of this manuscript were obtained from the National Institute of Mental Health (NIMH) Data Archive (NDA). NDA is a collaborative informatics system created by the National Institutes of Health to provide a national resource to support and accelerate research in mental health. Dataset identifier: <italic>Predictors and Mechanisms of Conversion to Psychosis (NAPLS3).</italic> This manuscript reflects the views of the authors and may not reflect the opinions or views of the NIH or of the Submitters submitting original data to NDA.</p>
</ack>
<sec id="s9" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>TC has served as a consultant for Boehringer Ingelheim and Lundbeck A/S. DM reported personal fees from Neurocrine Biosciences outside the submitted work and has served as a consultant for Aptinyx, Boehringer Ingelheim, Cadent Therapeutics, and Greenwich Biosciences. DP has served as a consultant for Sunovion and Alkermes, has received research support from Boehringer Ingelheim, and has received royalties from American Psychiatric Association Publishing. SW has received personal fees from American Psychiatric Association and Medscape outside the submitted work; had a patent for US patent no. 8492418 B2 issued; has received investigator-initiated research support from Pfizer and sponsor-initiated research support from Auspex and Teva; served as a consultant for Biomedisyn unpaid, Boehringer Ingelheim, and Merck; served as an unpaid consultant to <italic>DSM-5;</italic> and received royalties from Oxford University Press.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
<p>The author(s) declared that they were an editorial board member of Frontiers, at the time of submission. This had no impact on the peer review process and the final decision.</p>
</sec>
<sec id="s10" sec-type="ai-statement">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec id="s11" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Smucny</surname> <given-names>J</given-names>
</name>
<name>
<surname>Davidson</surname> <given-names>I</given-names>
</name>
<name>
<surname>Carter</surname> <given-names>CS</given-names>
</name>
</person-group>. <article-title>Are we there yet? Predicting conversion to psychosis using machine learning</article-title>. <source>Am J Psychiatry</source>. (<year>2023</year>) <volume>180</volume>:<page-range>836&#x2013;40</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1176/appi.ajp.20220973</pub-id>
</citation>
</ref>
<ref id="B2">
<label>2</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cannon</surname> <given-names>TD</given-names>
</name>
</person-group>. <article-title>Predicting conversion to psychosis using machine learning: are we there yet</article-title>? <source>Am J Psychiatry</source>. (<year>2023</year>) <volume>180</volume>:<page-range>789&#x2013;91</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1176/appi.ajp.20220973</pub-id>
</citation>
</ref>
<ref id="B3">
<label>3</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Addington</surname> <given-names>J</given-names>
</name>
<name>
<surname>Cadenhead</surname> <given-names>KS</given-names>
</name>
<name>
<surname>Cornblatt</surname> <given-names>BA</given-names>
</name>
<name>
<surname>Mathalon</surname> <given-names>DH</given-names>
</name>
<name>
<surname>Mcglashan</surname> <given-names>TH</given-names>
</name>
<name>
<surname>Perkins</surname> <given-names>DO</given-names>
</name>
<etal/>
</person-group>. <article-title>North American Prodrome Longitudinal Study (NAPLS 2): overview and recruitment</article-title>. <source>Schizophr Res</source>. (<year>2012</year>) <volume>142</volume>:<fpage>77</fpage>&#x2013;<lpage>82</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.schres.2012.09.012</pub-id>
</citation>
</ref>
<ref id="B4">
<label>4</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Addington</surname> <given-names>J</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>L</given-names>
</name>
<name>
<surname>Brummitt</surname> <given-names>K</given-names>
</name>
<name>
<surname>Bearden</surname> <given-names>CE</given-names>
</name>
<name>
<surname>Cadenhead</surname> <given-names>KS</given-names>
</name>
<name>
<surname>Cornblatt</surname> <given-names>BA</given-names>
</name>
<etal/>
</person-group>. <article-title>North American Prodrome Longitudinal Study (NAPLS 3): Methods and baseline description</article-title>. <source>Schizophr Res</source>. (<year>2022</year>) <volume>243</volume>:<page-range>262&#x2013;7</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.schres.2020.04.010</pub-id>
</citation>
</ref>
<ref id="B5">
<label>5</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Addington</surname> <given-names>J</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>L</given-names>
</name>
<name>
<surname>Buchy</surname> <given-names>L</given-names>
</name>
<name>
<surname>Cadenhead</surname> <given-names>KS</given-names>
</name>
<name>
<surname>Cannon</surname> <given-names>TD</given-names>
</name>
<name>
<surname>Cornblatt</surname> <given-names>BA</given-names>
</name>
<etal/>
</person-group>. <article-title>North american prodrome longitudinal study (NAPLS 2): the prodromal symptoms</article-title>. <source>J Nerv Ment Dis</source>. (<year>2015</year>) <volume>203</volume>:<page-range>328&#x2013;35</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1097/NMD.0000000000000290</pub-id>
</citation>
</ref>
<ref id="B6">
<label>6</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cannon</surname> <given-names>TD</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>C</given-names>
</name>
<name>
<surname>Addington</surname> <given-names>J</given-names>
</name>
<name>
<surname>Bearden</surname> <given-names>CE</given-names>
</name>
<name>
<surname>Cadenhead</surname> <given-names>KS</given-names>
</name>
<name>
<surname>Cornblatt</surname> <given-names>BA</given-names>
</name>
<etal/>
</person-group>. <article-title>An individualized risk calculator for research in prodromal psychosis</article-title>. <source>Am J Psychiatry</source>. (<year>2016</year>) <volume>173</volume>:<page-range>980&#x2013;8</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1176/appi.ajp.2016.15070890</pub-id>
</citation>
</ref>
<ref id="B7">
<label>7</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nitesh</surname> <given-names>V</given-names>
</name>
<name>
<surname>Bowyer</surname> <given-names>KW</given-names>
</name>
<name>
<surname>Hall</surname> <given-names>LO</given-names>
</name>
<name>
<surname>Kegelmeyer</surname> <given-names>WP</given-names>
</name>
</person-group>. <article-title>Synthetic minority over-sampling technique</article-title>. <source>J Artif Intell Res</source>. (<year>2002</year>) <volume>16</volume>:<page-range>321&#x2013;57</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1613/jair.953</pub-id>
</citation>
</ref>
<ref id="B8">
<label>8</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Maron</surname> <given-names>ME</given-names>
</name>
</person-group>. <article-title>Automatic indexing: an experimental inquiry</article-title>. <source>J Assoc Computing Machinery</source>. (<year>1961</year>) <volume>8</volume>:<page-range>404&#x2013;17</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1145/321075.321084</pub-id>
</citation>
</ref>
<ref id="B9">
<label>9</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Burges</surname> <given-names>C</given-names>
</name>
</person-group>. <article-title>A tutorial on support vector machines for pattern recognition</article-title>. <source>Data Min Knowledge Discovery</source>. (<year>1998</year>) <volume>2</volume>:<page-range>121&#x2013;67</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1023/A:1009715923555</pub-id>
</citation>
</ref>
<ref id="B10">
<label>10</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hart</surname> <given-names>PE</given-names>
</name>
</person-group>. <article-title>The condensed nearest neighbour rule</article-title>. <source>IEEE Trans Inf Theory</source>. (<year>1968</year>) <volume>14</volume>:<page-range>515&#x2013;6</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TIT.1968.1054155</pub-id>
</citation>
</ref>
<ref id="B11">
<label>11</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Viola</surname> <given-names>P</given-names>
</name>
<name>
<surname>Jones</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>Rapid object detection using a boosted cascade of simple features</article-title>. In: <source>International conference on Computer Vision and Pattern Recognition</source>. <publisher-loc>New York, New York, USA</publisher-loc>: <publisher-name>Institute of Electrical and Electronics Engineers</publisher-name>. (<year>2001</year>). p. <page-range>511&#x2013;8</page-range>.</citation>
</ref>
<ref id="B12">
<label>12</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Quinlan</surname> <given-names>JR</given-names>
</name>
</person-group>. <source>C4.5: Programs for Machine Learning</source>. <publisher-loc>San Mateo, CA USA</publisher-loc>: <publisher-name>Morgan Kaufmann Publishers</publisher-name> (<year>1993</year>).</citation>
</ref>
<ref id="B13">
<label>13</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Alpaydin</surname> <given-names>E</given-names>
</name>
</person-group>. <source>Introduction to Machine Learning</source>. <publisher-loc>Cambridge, MA USA</publisher-loc>: <publisher-name>MIT Press</publisher-name> (<year>2004</year>).</citation>
</ref>
<ref id="B14">
<label>14</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Breiman</surname> <given-names>L</given-names>
</name>
</person-group>. <article-title>Random forests</article-title>. <source>Mach Learn</source>. (<year>2001</year>) <volume>45</volume>:<fpage>5</fpage>&#x2013;<lpage>32</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1023/A:1010933404324</pub-id>
</citation>
</ref>
<ref id="B15">
<label>15</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chekroud</surname> <given-names>AM</given-names>
</name>
<name>
<surname>Hawrilenko</surname> <given-names>M</given-names>
</name>
<name>
<surname>Loho</surname> <given-names>H</given-names>
</name>
<name>
<surname>Bondar</surname> <given-names>J</given-names>
</name>
<name>
<surname>Gueorguieva</surname> <given-names>R</given-names>
</name>
<name>
<surname>Hasan</surname> <given-names>A</given-names>
</name>
<etal/>
</person-group>. <article-title>Illusory generalizability of clinical prediction models</article-title>. <source>Science</source>. (<year>2024</year>) <volume>383</volume>:<page-range>164&#x2013;7</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1126/science.adg8538</pub-id>
</citation>
</ref>
<ref id="B16">
<label>16</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Maikusa</surname> <given-names>N</given-names>
</name>
<name>
<surname>Radua</surname> <given-names>J</given-names>
</name>
<name>
<surname>Samann</surname> <given-names>PG</given-names>
</name>
<name>
<surname>Fusar-Poli</surname> <given-names>P</given-names>
</name>
<name>
<surname>Agartz</surname> <given-names>I</given-names>
</name>
<etal/>
</person-group>. <article-title>Using brain structural neuroimaging measures to predict psychosis onset for individuals at clinical high-risk</article-title>. <source>Mol Psychiatry</source>. (<year>2024</year>) <volume>29</volume>:<page-range>1465&#x2013;77</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41380-024-02426-7</pub-id>
</citation>
</ref>
<ref id="B17">
<label>17</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Sugiyama</surname> <given-names>M</given-names>
</name>
<name>
<surname>Kawanabe</surname> <given-names>M</given-names>
</name>
</person-group>. <source>Machine learning in non-stationary environments: Introduction to covariate shift adaptation</source>. <publisher-loc>Cambridge, MA, USA</publisher-loc>: <publisher-name>MIT Press</publisher-name> (<year>2012</year>).</citation>
</ref>
<ref id="B18">
<label>18</label>
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Kouw</surname> <given-names>WM</given-names>
</name>
<name>
<surname>Loog</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>An introduction to domain adaptation and transfer learning</article-title> (<year>2018</year>). Available online at: <uri xlink:href="https://ui.adsabs.harvard.edu/abs/2018arXiv181211806K">https://ui.adsabs.harvard.edu/abs/2018arXiv181211806K</uri> (Accessed <access-date>December 01, 2018</access-date>).</citation>
</ref>
<ref id="B19">
<label>19</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Smucny</surname> <given-names>J</given-names>
</name>
<name>
<surname>Shi</surname> <given-names>G</given-names>
</name>
<name>
<surname>Davidson</surname> <given-names>I</given-names>
</name>
</person-group>. <article-title>Deep learning in neuroimaging: overcoming challenges with emerging approaches</article-title>. <source>Front Psychiatry</source>. (<year>2022</year>) <volume>13</volume>:<elocation-id>912600</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpsyt.2022.912600</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>