<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<?covid-19-tdm?>
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Psychiatry</journal-id>
<journal-title>Frontiers in Psychiatry</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Psychiatry</abbrev-journal-title>
<issn pub-type="epub">1664-0640</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpsyt.2024.1376784</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Psychiatry</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Machine learning models predict the emergence of depression in Argentinean college students during periods of COVID-19 quarantine</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>L&#xf3;pez Steinmetz</surname>
<given-names>Lorena Cecilia</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/967659"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Sison</surname>
<given-names>Margarita</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2667063"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Zhumagambetov</surname>
<given-names>Rustam</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Godoy</surname>
<given-names>Juan Carlos</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2683490"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Haufe</surname>
<given-names>Stefan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/17505"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Inverse Modeling and Machine Learning, Chair of Uncertainty, Institute of Software Engineering and Theoretical Computer Science, Faculty IV Electrical Engineering and Computer Science, Technische Universit&#xe4;t Berlin</institution>, <addr-line>Berlin</addr-line>, <country>Germany</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Instituto de Investigaciones Psicol&#xf3;gicas (IIPsi), Facultad de Psicolog&#xed;a, Consejo Nacional de Investigaciones Cient&#xed;ficas y T&#xe9;cnicas (CONICET), Universidad Nacional de C&#xf3;rdoba (UNC)</institution>, <addr-line>C&#xf3;rdoba</addr-line>, <country>Argentina</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Berlin Center for Advanced Neuroimaging (BCAN), Charit&#xe9; &#x2013; Universit&#xe4;tsmedizin Berlin</institution>, <addr-line>Berlin</addr-line>, <country>Germany</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Working Group 8.44 Machine Learning and Uncertainty, Mathematical Modelling and Data Analysis Department, Physikalisch-Technische Bundesanstalt Braunschweig und Berlin</institution>, <addr-line>Berlin</addr-line>, <country>Germany</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Institute for Medical Informatics, Charit&#xe9; &#x2013; Universit&#xe4;tsmedizin Berlin</institution>, <addr-line>Berlin</addr-line>, <country>Germany</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Jianjun Ou, The Second Xiangya Hospital of Central South University, China</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Caglar Uyulan, Izmir K&#xe2;tip &#xc7;elebi University, T&#xfc;rkiye</p>
<p>Seyed-Ali Sadegh-Zadeh, Staffordshire University, United Kingdom</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Lorena Cecilia L&#xf3;pez Steinmetz, <email xlink:href="mailto:lopez.steinmetz@tu-berlin.de">lopez.steinmetz@tu-berlin.de</email>; <email xlink:href="mailto:cecilialopezsteinmetz@unc.edu.ar">cecilialopezsteinmetz@unc.edu.ar</email>; Stefan Haufe, <email xlink:href="mailto:haufe@tu-berlin.de">haufe@tu-berlin.de</email>
</p>
</fn>
<fn fn-type="equal" id="fn003">
<p>&#x2020;These authors have contributed equally to this work</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>16</day>
<month>04</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>15</volume>
<elocation-id>1376784</elocation-id>
<history>
<date date-type="received">
<day>26</day>
<month>01</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>29</day>
<month>03</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 L&#xf3;pez Steinmetz, Sison, Zhumagambetov, Godoy and Haufe</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>L&#xf3;pez Steinmetz, Sison, Zhumagambetov, Godoy and Haufe</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>The COVID-19 pandemic has exacerbated mental health challenges, particularly depression among college students. Detecting at-risk students early is crucial but remains challenging, particularly in developing countries. Utilizing data-driven predictive models presents a viable solution to address this pressing need.</p>
</sec>
<sec>
<title>Aims</title>
<p>1) To develop and compare machine learning (ML) models for predicting depression in Argentinean students during the pandemic. 2) To assess the performance of classification and regression models using appropriate metrics. 3) To identify key features driving depression prediction.</p>
</sec>
<sec>
<title>Methods</title>
<p>A longitudinal dataset (N = 1492 college students) captured T1 and T2 measurements during the Argentinean COVID-19 quarantine. ML models, including linear logistic regression classifiers/ridge regression (LogReg/RR), random forest classifiers/regressors, and support vector machines/regressors (SVM/SVR), are employed. Assessed features encompass depression and anxiety scores (at T1), mental disorder/suicidal behavior history, quarantine sub-period information, sex, and age. For classification, models&#x2019; performance on test data is evaluated using Area Under the Precision-Recall Curve (AUPRC), Area Under the Receiver Operating Characteristic curve, Balanced Accuracy, F1 score, and Brier loss. For regression, R-squared (R2), Mean Absolute Error, and Mean Squared Error are assessed. Univariate analyses are conducted to assess the predictive strength of each individual feature with respect to the target variable. The performance of multi- vs univariate models is compared using the mean AUPRC score for classifiers and the R2 score for regressors.</p>
</sec>
<sec>
<title>Results</title>
<p>The highest performance is achieved by SVM and LogReg (e.g., AUPRC: 0.76, 95% CI: 0.69, 0.81) and SVR and RR models (e.g., R2 for SVR and RR: 0.56, 95% CI: 0.45, 0.64 and 0.45, 0.63, respectively). Univariate models, particularly LogReg and SVM using depression (AUPRC: 0.72, 95% CI: 0.64, 0.79) or anxiety scores (AUPRC: 0.71, 95% CI: 0.64, 0.78) and RR using depression scores (R2: 0.48, 95% CI: 0.39, 0.57) exhibit performance levels close to those of the multivariate models, which include all features.</p>
</sec>
<sec>
<title>Discussion</title>
<p>These findings highlight the relevance of pre-existing depression and anxiety conditions in predicting depression during quarantine, underscoring their comorbidity. ML models, particularly SVM/SVR and LogReg/RR, demonstrate potential in the timely detection of at-risk students. However, further studies are needed before clinical implementation.</p>
</sec>
</abstract>
<kwd-group>
<kwd>depression prediction</kwd>
<kwd>COVID-19 pandemic</kwd>
<kwd>machine learning</kwd>
<kwd>classification</kwd>
<kwd>regression</kwd>
<kwd>college students</kwd>
<kwd>longitudinal survey</kwd>
<kwd>Argentina</kwd>
</kwd-group>
<contract-sponsor id="cn001">H2020 European Research Council<named-content content-type="fundref-id">10.13039/100010663</named-content>
</contract-sponsor>
<counts>
<fig-count count="5"/>
<table-count count="0"/>
<equation-count count="0"/>
<ref-count count="68"/>
<page-count count="15"/>
<word-count count="8034"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Mood Disorders</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<title>Introduction</title>
<p>The COVID-19 pandemic has presented significant global challenges to mental health, particularly among college students. The sudden transition to remote learning, social isolation, and economic instability has exacerbated mental health issues, with depression becoming a prevalent concern for this population (<xref ref-type="bibr" rid="B1">1</xref>, <xref ref-type="bibr" rid="B2">2</xref>). Depression can negatively impact academic performance, social relationships, and overall quality of life. Therefore, early identification of depression and its risk factors is crucial for timely interventions and prevention of further negative outcomes. However, identifying individuals who may be at risk of developing depression remains a significant challenge in this field.</p>
<p>The literature consistently identifies key psychological and demographic variables linked to or predicting depression. Anxiety often emerges as a predictor for depression (<xref ref-type="bibr" rid="B3">3</xref>, <xref ref-type="bibr" rid="B4">4</xref>), notably among college students (<xref ref-type="bibr" rid="B5">5</xref>). Moreover, histories of diagnosed mental disorders (<xref ref-type="bibr" rid="B6">6</xref>) and suicidal behavior (<xref ref-type="bibr" rid="B7">7</xref>) closely correlate with college student depression. While sex (<xref ref-type="bibr" rid="B8">8</xref>) and age (<xref ref-type="bibr" rid="B9">9</xref>) differences in depression are well-documented, sex-based variations in depression may display age-specific trends (<xref ref-type="bibr" rid="B10">10</xref>). Amid COVID-19 lockdowns, mental health, particularly depression, was anticipated to deteriorate with extended measures, impacting well-being during and after implementation (<xref ref-type="bibr" rid="B11">11</xref>). Yet, achieving a comprehensive integration of these factors into accurate predictive models for depression remains incomplete and challenging.</p>
<p>Existing literature on depression has extensively analyzed variability between groups and within individuals (e.g., <xref ref-type="bibr" rid="B12">12</xref>). Mixed-effects modeling (MEM), or hierarchical linear modeling, is a widely used statistical approach that allows for the estimation of both fixed and random effects, accounting for variability between groups and within individuals. MEM has a major strength in its ability to identify individual and group-level effects, as well as interactions between them, which is particularly relevant in fields like psychology where individual differences are of interest.</p>
<p>However, MEM is primarily designed for inference rather than prediction, and offers limited flexibility for the handling of high-dimensional data, compromising their predictive accuracy compared to machine learning (ML) models. ML models are well-suited for prediction tasks, handling complex relationships between variables, and relying less on strong data assumptions (<xref ref-type="bibr" rid="B13">13</xref>&#x2013;<xref ref-type="bibr" rid="B16">16</xref>). MEM, in particular, relies on assumptions about predictor-response relationships (i.e., linearity), normality, and independence of residuals, which can impact their predictive accuracy for new data. Conversely, ML is a data-driven approach that uses algorithms to identify patterns in the data for prediction. However, ML models require extensive data for training and can be prone to overfitting with too complex models or small datasets. Therefore, while ML has emerged as an alternative or complementary approach to traditional statistics in mental health research, it is an open challenge to leverage ML for predicting individuals at risk for depression.</p>
<p>In the field of mental health assessment and ML applications, the challenge of diagnosis prediction can be tackled using a dual methodology, encompassing both classification and regression approaches. In the context of binary classification, the goal is to categorize individuals into two groups: those with depression and those without. This approach yields a straightforward determination of depression presence, aiding in identifying individuals requiring additional evaluation. Conversely, in the context of regression, the aim is to estimate or forecast numerical depression scores. This approach offers a continuous prediction of depression severity, affording a deeper insight into the condition and potentially enabling personalized treatment strategies.</p>
<p>In a prior study, we utilized MEM to investigate the aforementioned key psychological and demographic factors linked to or predictive of depression in college students during Argentina&#x2019;s COVID-19 pandemic quarantine, which was one of the most rigorous and extended lockdowns globally (<xref ref-type="bibr" rid="B17">17</xref>). In the current study, we seek to reassess the same dataset using ML algorithms to evaluate their potential as an alternative or complementary technique to MEM for predicting depression diagnosis. Through this, we hope to enhance predictive precision and effectively identify individuals with depression, while gaining deeper insights into the intricate interplay of factors contributing to depression during pandemics. Furthermore, we expect to determine whether the key factors commonly acknowledged as depression predictors remain as significant input features when utilizing ML algorithms for data analysis, in contrast to other widely used statistical methodologies in psychology. The objectives of this study are: 1) To develop and compare various ML models, including linear logistic regression classifiers/ridge regression, random forest classifiers/regressors, and support vector machines/support vector regression models, to predict depression in college students during the pandemic based on psychological inventory scores, basic clinical information, quarantine sub-period information, and demographics as features. 2) To assess the performance of classification and regression models using appropriate metrics. 3) To identify the key features that drive the prediction of depression by applying univariate methods.</p>
<p>A recent review following the PRISMA guidelines identified only 33 peer-reviewed studies in the domain of utilizing ML algorithms for predicting mental health diagnoses, with only 31% specifically focusing on depression/anxiety disorders. Additionally, most studies have been conducted on clinical samples, primarily consisting of adults or older patients (<xref ref-type="bibr" rid="B18">18</xref>). It is worth noting that many of these studies rely on unavailable facilities and resources for diagnosis, particularly in developing countries, such as MRI or blood samples. Therefore, one novel aspect of this study is its focus on exploring the potential of ML algorithms to identify risk factors for depression, during a pandemic, in a large longitudinal sample of quarantined and apparently healthy college students from a developing country. By doing so, it aims to provide further insights into this important area of research by enhancing the understanding of key determinants in depression detection and may have significant implications for the development of more effective screening tools and interventions for depression in college students during pandemics.</p>
</sec>
<sec id="s2" sec-type="materials|methods">
<title>Materials and methods</title>
<sec id="s2_1">
<title>Research design and dataset</title>
<p>This study employs a longitudinal dataset featuring two-repeated measurements in college students during the Argentinean COVID-19 quarantine period. The first measurement (T1) was taken across various quarantine sub-periods, each characterized by varying levels of restrictions, and spanning up to 106 days. The follow-up measurement (T2) occurred one month later (as depicted in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S1</bold>
</xref>).</p>
<p>The choice of this longitudinal research design arises from its inherent capability to capture dynamic changes over time, a crucial aspect when investigating the effects of a rapidly evolving event like the COVID-19 pandemic. This approach transcends mere cross-sectional snapshots, providing a more thorough understanding of the intricate interplay between psychological states and external factors amid a quarantine. By incorporating two measurements at distinct time points, it becomes possible to discern short-term fluctuations and potential enduring effects. The T1 measurement establishes a baseline and assesses the immediate effects of quarantine sub-periods lasting up to 106 days. The T2 measurement, conducted a month later, offers insights into the persistence or evolution of mental health patterns, shedding light on potential longer-term consequences. This temporal depth captures nuances that a single-time assessment might overlook, providing a more nuanced portrayal of the challenges college students faced during this unprecedented period.</p>
<p>The dataset comprises responses from 1492 college students who completed an online survey during the mandatory restrictive quarantine. It encompasses assessments of depression, anxiety-trait, basic clinical information, demographics, and quarantine sub-periods. This dataset corresponds to the one utilized in our previous study (<xref ref-type="bibr" rid="B17">17</xref>). Further information on the research design and data collection can be found there and in L&#xf3;pez Steinmetz (<xref ref-type="bibr" rid="B19">19</xref>).</p>
</sec>
<sec id="s2_2">
<title>Input features</title>
<p>The prediction models were developed using the following input features: age, sex (female, male), history of diagnosed mental disorder (absent, present), history of suicidal attempt and/or ideation (absent, present), depression scores from T1 using the Argentinean validation (<xref ref-type="bibr" rid="B20">20</xref>) of the Beck Depression Inventory (BDI) (<xref ref-type="bibr" rid="B21">21</xref>), anxiety-trait scores from T1 using the Spanish version of the State-Trait Anxiety Inventory (<xref ref-type="bibr" rid="B22">22</xref>), and three quarantine sub-periods to which participants&#x2019; responses were chronologically assigned. These sub-periods were categorized according to decreasing levels of COVID-related restrictions over the 106-day duration, and participants were classified into one of three quarantine sub-periods based on the date of their response for measurement T1 (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S1</bold>
</xref>).</p>
</sec>
<sec id="s2_3">
<title>Target variable</title>
<p>The target variable, depression, was assessed using the Argentinean validation (<xref ref-type="bibr" rid="B20">20</xref>) of the BDI (<xref ref-type="bibr" rid="B21">21</xref>). For the classification task, depression scores from T2 were labeled as either according to the absence or presence of clinically relevant levels of depression, respectively. The standardized cut-off score of &gt;20, indicative of depression presence in non-clinical populations (<xref ref-type="bibr" rid="B23">23</xref>), was employed. For the regression task, raw depression scores were used as a continuous variable.</p>
</sec>
<sec id="s2_4">
<title>Software</title>
<p>The analysis was performed using Python software (version 3.11.1) and relevant libraries, including scikit-learn (<xref ref-type="bibr" rid="B24">24</xref>), pandas (<xref ref-type="bibr" rid="B25">25</xref>), and numpy (<xref ref-type="bibr" rid="B26">26</xref>), among others.</p>
</sec>
<sec id="s2_5">
<title>Data analysis</title>
<sec id="s2_5_1">
<title>Preprocessing</title>
<p>The dataset does not contain missing data. Before performing classification and regression tasks, the data were randomly divided into a training (75%) and a test (25%) set. In the classification task, the proportion of negative to positive labels was maintained in both sets. Categorical variables were encoded using a one-hot schema. Each variable was transformed into as many binary 0/1 variables as there were different values. Continuous features such as depression and anxiety scores from T1 and age were scaled using a quantile transformation, specifically, we used the QuantileTransformer method (<xref ref-type="bibr" rid="B24">24</xref>) setting the output distribution as normal, to reduce skewness and to enhance model performance. Principal component analysis (PCA) was applied for dimensionality reduction. PCA was set to retain the number of components essential to account for 95% of the variance present in the original data. Consequently, a total of 3 components were retained. Both feature scaling and dimensionality reduction were fitted on the training set and subsequently applied to both sets. The target variable in the regression task was scaled using the same quantile transform function.</p>
</sec>
<sec id="s2_5_2">
<title>Hyperparameter tuning in the classification and regression tasks</title>
<p>Both in the classification task and in the regression task hyperparameter tuning was conducted using the GridSearchCV function of the scikit-learn library (<xref ref-type="bibr" rid="B24">24</xref>) on the training set for each model. Stratified 10-fold cross-validation, which is typically used when dealing with imbalanced datasets, was applied for robust evaluation during hyperparameter tuning, that is, the training data were split into ten stratified subsets, where in each fold one of the subsets served as an inner validation set and nine as an inner training set. Three ML models for the classification task and three ML models for the regression task with different hyperparameter choices were optimized on each inner training set and tested on each inner validation set. In the classification task, the Average Precision (AP) score was used as the evaluation metric to select model hyperparameters within the inner cross-validation. For the regression task, the R-squared (R2) score was employed as the evaluation metric for hyperparameter selection within the inner cross-validation process. The state of the random number generator was set to 0 for reproducibility in all cases. The final model for each ML model was then trained on the entire training set using the selected hyperparameters. This comprehensive approach aimed to enhance the performance and robustness of the models in both classification and regression tasks.</p>
</sec>
<sec id="s2_5_3">
<title>Classification models</title>
<p>For the classification task, three ML algorithms are employed: Linear logistic regression classifier (LogReg) (<xref ref-type="bibr" rid="B13">13</xref>), random forest (RF) classifier (<xref ref-type="bibr" rid="B27">27</xref>), and support vector machine (SVM) (<xref ref-type="bibr" rid="B28">28</xref>). LogReg models are designed for predicting probabilities of <italic>K</italic> classes or categories in the classification problem via linear functions of input features <italic>x</italic>, ensuring they sum to one and stay within a valid probability range of [0, 1]. The model is expressed in terms of <italic>K</italic> &#x2212; 1 log-odds or logit transformations and uses the last class as the denominator in the odds-ratios, with a special simplicity in binary classification scenarios (<xref ref-type="bibr" rid="B13">13</xref>). RF is an ensemble learning method that builds multiple decision trees by randomly sampling with replacement from the training data. Each tree is grown by recursively selecting random feature subsets and identifying optimal split points. For classification, the final output is determined by the majority vote of individual tree predictions during prediction. Thus, the model output is an ensemble of trees that collectively make predictions. Hyperparameters include the number of trees in the forest, the maximum depth of the trees, and the minimum number of samples required to split a node (<xref ref-type="bibr" rid="B13">13</xref>). The SVM algorithm classifies data by transforming each data point into a k-dimensional feature space, where k &gt;&gt; <italic>n</italic> and <italic>n</italic> is the number of features. It identifies a hyperplane that maximizes the margin between classes, minimizing classification errors. The margin is the distance between the decision hyperplane and the nearest instance of each class (<xref ref-type="bibr" rid="B29">29</xref>).</p>
<p>These algorithms were chosen due to their well-established competitive performance in classification tasks and adaptability to diverse scenarios. For instance, Uddin et&#xa0;al. (<xref ref-type="bibr" rid="B29">29</xref>) provide a comprehensive overview of the relative performance of different supervised ML algorithms for disease prediction, including LogReg, RF, and SVM. Their findings, despite variations in frequency and performance, underscore the potential of these algorithmic families in disease prediction.</p>
<p>Certain model hyperparameters were fixed for each algorithm while others were optimized using cross-validation. For the <bold>LogReg</bold> algorithm, an intercept term, which represents the log-odds of the baseline class, was included in the model by the default setting of scikit-learn. The parameter class_weight was set as balanced to adjust the weights assigned to classes during the training process. This indicates that the algorithm automatically adjusts the weights of the classes inversely proportional to their frequencies. The maximum number of iterations taken for the solver to converge was set to 500. Solver here refers to the optimization algorithm used to find the optimal values for the coefficients of the linear logistic regression model. L2-norm regularization was applied to prevent overfitting and improve the generalization of the model. Regularization involves adding a penalty term to the loss function, and in L2-norm regularization, the penalty is proportional to the square of the magnitudes of the coefficients. The regularization parameter C was set to values 0.0001, 0.001, 0.01, 0.1, 1, 10, and 1000. The optimal hyperparameter as identified using cross-validation is C = 0.1.</p>
<p>For the RF classifier, several model hyperparameters were tuned. The number of features to consider for the best split (&#x2018;max_features&#x2019;) was set to the square root of the total number of features. The function to measure the quality of a split (&#x2018;criterion&#x2019;) was configured to use the Gini diversity index. The Gini index is employed by each decision tree in the RF ensemble as the measure for determining optimal split points during the training process. The number of trees (&#x2018;n_estimators&#x2019;) was set to values 50, 100, 500, and 1000. The maximum depth of the tree (&#x2018;max_depth&#x2019;) was set to values 5, 10, and 15. The minimum number of samples required to split an internal node (&#x2018;min_samples_split&#x2019;) was set to values 2, 5, and 8. The minimum number of samples required to be at a leaf node (&#x2018;min_samples_leaf&#x2019;) was set to values 1, 2, and 3. For tree building, a bootstrap method was used, involving sampling with replacement instead of using the whole training set to build trees. The optimal combination of hyperparameters as identified using cross-validation are n_estimators: 100, max_depth: 5, min_samples_split: 5, and min_samples_leaf: 2.</p>
<p>For the SVM algorithm, the parameter class_weight was set as balanced. Two variants of the SVM algorithm were tested. The first one used a non-linear radial basis function (rbf) kernel with kernel width gamma (kernel: rbf). Tested hyperparameter values included: C: 0.01, 0.1, 1, 10, 100, 500, 1000; gamma: 0.00001, 0.0001, 0.001, 0.01, 0.1, 1, 10. The second variant used a parameterless linear kernel. Tested hyperparameter values for this choice included: C: 0.01, 0.1, 1, 10, 100, 500, 1000; kernel: linear. The optimal combination of hyperparameters as identified using cross-validation are C: 1000, gamma: 1e-05, kernel: rbf.</p>
<p>Results of the hyperparameter tuning and grid search for each classifier can be found in the <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Material (Supplementary Tables S1&#x2013;S3, Supplementary Figures S2, S3)</bold>
</xref>.</p>
<p>Benchmarking was conducted using dummy models representing baseline references. Baseline models serve as the null hypothesis, simulating scenarios where models possess minimal knowledge about individual sample labels, essentially reflecting the label distribution in the training data at best. These models are instrumental for assessing whether a sophisticated model&#x2019;s performance surpasses the &#x201c;null&#x201d; or &#x201c;chance&#x201d; level, providing a benchmark to evaluate the effectiveness of advanced algorithms.</p>
<p>Baseline models consisted of a uniform random baseline (randomly assigning class labels with equal probability, i.e., 50% for each class, without considering any input features), a most frequent baseline (it assigns the majority class label to all instances), and a stratified random baseline (which randomly assigns labels proportionally to their relative frequency in the training data) (<xref ref-type="bibr" rid="B14">14</xref>).</p>
</sec>
<sec id="s2_5_4">
<title>Regression models</title>
<p>For the regression task, three ML algorithms are utilized: Ridge regression (RR) (<xref ref-type="bibr" rid="B13">13</xref>), RF regressor (<xref ref-type="bibr" rid="B27">27</xref>), and support vector regressor (SVR) (<xref ref-type="bibr" rid="B28">28</xref>). RR, often referred to as L2-norm regularized least-squares regression, adds a regularization term to the standard linear regression objective function. In RR, the objective function to be minimized is the sum of the squared differences between the observed and predicted values (least squares term), and a penalty term that discourages overly complex models by adding the L2-norm of the coefficients multiplied by a regularization parameter (alpha). The regularization term penalizes large coefficients, preventing overfitting and promoting a more stable and generalizable model. The strength of this penalty is controlled by the hyperparameter alpha (<xref ref-type="bibr" rid="B13">13</xref>). RF constructs multiple decision trees by bootstrapping the data and introducing randomness in variable selection at each node. The final prediction is an aggregation of predictions from all individual trees. For regression, the final output is determined by the average of individual tree predictions during prediction (<xref ref-type="bibr" rid="B13">13</xref>). SVR uses the principles of SVM for regression tasks. Similar to the latter, SVR introduces the concept of a margin. In SVR, the margin represents a range of values within which errors are tolerable. SVR aims to fit the function <italic>f</italic>(<italic>x</italic>) in such a way that the differences between the predicted values and the actual values (errors) fall within the specified margin. SVR minimizes the errors while staying within the margin. The loss function penalizes deviations from the actual values but allows for a certain amount of error within the margin. To prevent overfitting, SVR includes a regularization term controlled by the parameter C, which is a user-defined hyperparameter. The regularization term controls the smoothness of the learned function (<xref ref-type="bibr" rid="B13">13</xref>).</p>
<p>Certain model hyperparameters were fixed for each algorithm while others were optimized using cross-validation. For the <bold>RR</bold> algorithm the regularization strength parameter, denoted as alpha, was set to values 0.0001, 0.001, 0.01, 0.1, 1, 10, 100, and 1000. The optimal model&#x2019;s alpha as identified using cross-validation is 100.</p>
<p>For the RF regressor, several model hyperparameters were tuned. The number of features to consider for the best split (&#x2018;max_features&#x2019;) was set to 1, as empirically justified in Geurts et&#xa0;al. (<xref ref-type="bibr" rid="B30">30</xref>). The function to measure the quality of a split (&#x2018;criterion&#x2019;) was configured to use the mean squared error. The number of trees (&#x2018;n_estimators&#x2019;) was set to values 50, 100, 500, and 1000. The maximum depth of the tree (&#x2018;max_depth&#x2019;) was set to values 5, 10, and 15. The minimum number of samples required to split an internal node (&#x2018;min_samples_split&#x2019;) was set to values 2, 5, and 8. The minimum number of samples required to be at a leaf node (&#x2018;min_samples_leaf&#x2019;) was set to values 1, 2, and 3. For tree building, a bootstrap method was used, involving sampling with replacement instead of using the whole training set to build trees. The optimal combination of hyperparameters as identified using cross-validation are n_estimators: 50, max_depth: 5, min_samples_split: 8, and min_samples_leaf: 3.</p>
<p>For the SVR training, the shrinking trick described by (<xref ref-type="bibr" rid="B31">31</xref>) was employed. Shrinking attempts to reduce the optimization problem by removing the elements that have already been bound. Two variants of the SVR algorithm were tested. The first one used a rbf kernel and the hyperparameters C, epsilon, and gamma are varied. The tested values included: C: 0.01, 0.1, 1, 10, 100, 500, 1000; epsilon: 0.001, 0.01, 0.1, 1; gamma: 0.00001, 0.0001, 0.001, 0.01, 0.1, 1, 10; kernel: rbf. The second variant used a linear kernel and only the hyperparameters C and epsilon are varied. The hyperparameter values for this choice included: C: 0.01, 0.1, 1, 10, 100, 500, 1000; epsilon: 0.001, 0.01, 0.1, 1; kernel: linear. The optimal hyperparameters combination as identified using cross-validation are C: 500, epsilon: 0.1, gamma: 0.00001, kernel: rbf.</p>
<p>Results of the hyperparameter tuning and grid search for each regressor can be found in the <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Materials (Supplementary Tables S4, S5, Supplementary Figures S4, S5)</bold>
</xref>.</p>
<p>Similar to the classification models, dummy models were included as benchmark references for regression tasks. These models encompassed a randomly shuffled baseline (predicting the target variable by randomly shuffling actual target values), a mean and a median baseline (predicting the arithmetic average and the median value of the target variable, respectively, for all instances, without considering any input features or patterns) (<xref ref-type="bibr" rid="B14">14</xref>).</p>
</sec>
<sec id="s2_5_5">
<title>Evaluation of classification model performance</title>
<p>The performance of the classification models on test data was assessed using five metrics: Area Under the Precision-Recall Curve (AUPRC), Area Under the Receiver Operating Characteristic curve (AUROC), Balanced Accuracy, F1 score, and Brier loss. These metrics all range from 0 to 1, where 0 indicates worst and 1 indicates best performance. An exception is Brier loss, for which 0 indicates perfect model performance and uncertainty calibration, while a loss of 1 indicates the worst performance.</p>
<p>Precision measures a classifier&#x2019;s ability to avoid incorrectly labeling negative examples as positives, while Recall (also known as &#x201c;sensitivity&#x201d; or &#x201c;true positive rate&#x201d;) measures its ability to identify all positive examples correctly as positives (<xref ref-type="bibr" rid="B32">32</xref>). The AUPRC represents the tradeoff between Precision and Recall over all possible classifier thresholds as quantified by the area under the Precision curve integrated over all Recall values. The AUROC represents the tradeoff between Recall and the true negative rate (also known as &#x201c;specificity&#x201d;, <xref ref-type="bibr" rid="B33">33</xref>, <xref ref-type="bibr" rid="B34">34</xref>). The Balanced Accuracy score calculates the average of sensitivity (true positive rate) and specificity and it is used to ensure that the model&#x2019;s performance on imbalanced datasets is not overestimated. The F1 score is the harmonic mean of Precision and Recall scores (<xref ref-type="bibr" rid="B32">32</xref>). The Brier loss is calculated as the mean squared difference between the predicted probability and the true binary label (<xref ref-type="bibr" rid="B35">35</xref>). It measures whether the probabilities provided by the model are well-calibrated. These predicted probabilities are derived from the model&#x2019;s internal mechanisms, which assess the input features and generate probability estimates for each instance in the dataset.</p>
</sec>
<sec id="s2_5_6">
<title>Evaluation of regression model performance</title>
<p>To assess the performance of the regression models on test data, three key metrics were employed: R2, Mean Absolute Error (MAE), and Mean Squared Error (MSE). The R2 score measures the proportion of variance in the target variable explained by the model and is calculated as 1 minus the ratio of the model&#x2019;s MSE to that of a mean baseline model. The R2 score is a coefficient of determination and, theoretically, it can range from -&#x221e; to 1, where 1 indicates perfect prediction, 0 indicates no improvement over the mean model, and values less than 0 indicate poorer performance than the baseline (<xref ref-type="bibr" rid="B16">16</xref>, <xref ref-type="bibr" rid="B36">36</xref>). MSE and MAE represent the average squared and absolute differences between predicted and actual scores, respectively, and are used to assess the model&#x2019;s prediction accuracy (<xref ref-type="bibr" rid="B16">16</xref>, <xref ref-type="bibr" rid="B36">36</xref>). For both MSE and MAE, lower values signify better model performance.</p>
</sec>
<sec id="s2_5_7">
<title>Bootstrapping with replacement</title>
<p>To enhance the reliability of performance evaluation, bootstrapping with replacement was utilized. Specifically, we resampled the test set 100 times with replacement to create distributions of performance estimates. Each resampled sample had the same dimensions as the original sample, and mean performance estimates with 95% confidence intervals (CI) were derived using this method (<xref ref-type="bibr" rid="B37">37</xref>).</p>
</sec>
<sec id="s2_5_8">
<title>Feature importance analysis: comparing multi- vs. univariate models for depression prediction</title>
<p>Feature importance analysis was conducted to assess the predictive strength of each individual feature with respect to the target variable, depression. For this analysis, all ML algorithms previously employed for the classification (LogReg, RF, and SVM classifiers) and regression (RR, RF regressor, and SVR) tasks were applied to single features. Using different algorithms can provide a more comprehensive view of feature importance. The purpose of this analysis was to determine the extent to which individual features are predictive of depression.</p>
<p>Here, the performance of multivariate models, which incorporate all features simultaneously, is compared with that of univariate models, each including a single feature. The metrics used for this comparative analysis are the mean AUPRC score (with 95% CI) for classifiers and the mean R2 score (with 95% CI) for regressors. In this assessment, each multivariate model, whether for classification or regression tasks, serves as a benchmark. This enables an evaluation of whether the collective impact of features enhances prediction accuracy.</p>
</sec>
</sec>
<sec id="s2_6">
<title>Ethical considerations</title>
<p>This study was approved by the Ethics Committee of the Psychological Research Institute, Universidad Nacional de C&#xf3;rdoba (CEIIPsi-UNC-CONICET; comite.etica.iipsi@psicologia.unc.edu.ar), 14/02/20&#x2013;23/03/20. During data collection, participants&#x2019; names were not requested. However, email addresses were obtained solely for survey submission during the follow-up measurement, after which they were promptly deleted from the database. Only the principal researcher had access to the survey system and the raw dataset. The privacy and confidentiality of participants&#x2019; data were ensured, and informed consent was obtained from all participants before their participation. Guidelines and regulations regarding the use of human subjects in research were also followed. Additionally, the current study utilized the dataset that is available in open access.</p>
</sec>
</sec>
<sec id="s3" sec-type="results">
<title>Results</title>
<sec id="s3_1">
<title>Classification models: identifying depression presence</title>
<p>The performance of ML classifiers for predicting depression in college students during the COVID-19 pandemic is evaluated across five metrics and compared to the metrics of dummy baseline models. The baseline models serve as sanity checks for evaluating the performance of the ML classifiers.</p>
<p>
<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref> summarizes the performance metrics for both the trained and tested ML models and the tested baseline models. As expected and across all metrics, the ML models during the test phase achieve a superior predictive ability using the provided input features compared to random chance.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Performance of machine learning classifiers on the training and test sets as well as of baseline classifiers on the test set. The bar charts depict the performance of machine learning classifiers (linear logistic regression, random forest classifier, and support vector machine) on the training set and the test set and of baseline classifiers (uniform random, most frequent, and stratified random) on the test set using five performance metrics: <bold>(A)</bold> AUPRC, <bold>(B)</bold> AUROC, <bold>(C)</bold> Balanced accuracy, <bold>(D)</bold> Brier loss, and <bold>(E)</bold> F1. Light blue bars represent performance scores on the training set, while blue bars with error bars represent mean test set scores with 95% confidence intervals. To ensure that higher values consistently indicate better performance across all methods, the Brier loss is plotted as 1-Brier. AUPRC, Area Under the Precision-Recall Curve; AUROC, Area Under the Receiver Operating Characteristic Curve; LR, Linear logistic regression; RF, Random forest classifier; SVM, Support vector machine; UNI BL, Uniform random baseline; MFREQ BL, Most frequent baseline; STRAT BL, Stratified random baseline.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpsyt-15-1376784-g001.tif"/>
</fig>
<p>The SVM classifier is on par with or outperforms both alternative classifiers, attaining the highest AUPRC score of 0.76 (95% CI: 0.69, 0.81) and the lowest Brier score loss of 0.15 (0.14, 0.17). The LogReg classifier closely follows, displaying an identical AUPRC score of 0.76 (0.69, 0.81) and a slightly higher Brier score loss of 0.16 (0.14, 0.18).</p>
<p>The RF classifier generally achieves lower performance than SVM and LogReg, attaining an AUPRC score of 0.73 (0.66, 0.80) and a Brier score loss of 0.16 (0.15, 0.18). RF also shows some evidence of overfitting, as indicated by a higher drop in AUPRC performance (from 0.86 on the training set to 0.73 on the test set) compared to both alternative classifiers. Refer to <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref> and <xref ref-type="supplementary-material" rid="SM1">
<bold>Table S6</bold>
</xref> for a visualization of these and the remaining performance metrics (i.e., AUROC, Balanced accuracy, and F1 scores) employed in this study for each ML classifier.</p>
<p>Refer to <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref> for the Precision-Recall Curves and Receiver-Operating Characteristic depicting depression prediction. This plot aids in comparing models&#x2019; ability to distinguish between positive and negative classes (Depressed and Non-Depressed, respectively) at different clinically relevant operating points such as high sensitivity and high specificity). Confusion matrices for each ML classifier are displayed in the <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary File (Supplementary Figure S6)</bold>
</xref>.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Classifier performance comparison: Precision-Recall Curves and Receiver-Operating Characteristic Curves for depression prediction. These curves characterize the predictive performance of trained models (linear logistic regression, random forest classifier, and support vector machine) and dummy classifiers (uniform random, most frequent, and stratified random baselines) for depression prediction for varying classifier thresholds, where the positive class indicates depression presence. <bold>(A)</bold> Precision-Recall Curve (PRC) displays how precision changes with different recall levels for various models/baselines. <bold>(B)</bold> Receiver-Operating Characteristic (ROC) curves illustrate the balance between true positive rate (sensitivity) and false positive rate (1-specificity) as probability thresholds change. Each curve depicts how sensitivity changes with different specificity levels for different models/baselines. SVM, Support vector machine. AP: Precision-Recall curve values (the Precision-Recall curve display in scikit-learn uses the term &#x201c;average precision&#x201d; (AP) to refer to the area under the precision-recall curve). AUC, Area Under the Receiver Operating Characteristic values.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpsyt-15-1376784-g002.tif"/>
</fig>
</sec>
<sec id="s3_2">
<title>Regression models: quantifying depression severity</title>
<p>The performance of ML regressors in predicting depression among college students during the COVID-19 pandemic is evaluated across three key metrics and compared to the results of these metrics on dummy baseline models. The baseline models serve as benchmarks for evaluating the performance of the ML regressors.</p>
<p>
<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref> provides an overview of the performance metrics for both the trained and tested ML models, along with the tested baseline models. As expected, the tested ML models consistently demonstrate superior predictive capabilities using the provided input features compared to random chance, as evidenced by all metrics.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Performance comparison of machine learning regressors on the training and test sets and baselines on the test set. The bar charts compare the performance of machine learning regressors (ridge regression, random forest regressor, and support vector regressor) on the training set and the test set and of baseline regressors (randomly shuffled baseline, mean baseline, and median baseline) on the test set using three performance metrics: <bold>(A)</bold> R-squared, <bold>(B)</bold> Mean Absolute Error, <bold>(C)</bold> Mean Squared Error. Light blue bars represent performance scores on the training set, while blue bars with error bars represent mean test set scores with 95% confidence intervals. R2, R-squared; RR, Ridge Regression; RF, Random Forest Regressor; SVR, Support Vector Regressor; RAND SHUFF, Randomly Shuffled Baseline; MEAN, Mean Baseline; MDN, Median Baseline.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpsyt-15-1376784-g003.tif"/>
</fig>
<p>SVR and RR achieve the highest R2 score of 0.56 (95% CI: SVR: 0.45, 0.64 and RR: 0.45, 0.63). This R2 value indicates that the models can account for approximately 56% of the variability in the target variable. The RF regressor achieves a lower R2 value of 0.50 (0.40, 0.59).</p>
<p>The RF regressor also exhibits inferior MAE and MSE performance on the test set compared to the SVR and RR models. Both the MAE and MSE metrics for the RR algorithm mirror the SVR&#x2019;s performance, underscoring their comparable predictive capabilities (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>, <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S7</bold>
</xref>). The correlations between actual and predicted values for each ML algorithm in the test set are depicted in the <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary File (Supplementary Figure S7</bold>
</xref>).</p>
</sec>
<sec id="s3_3">
<title>Evaluation of feature importance: comparison of multivariate to univariate models</title>
<sec id="s3_3_1">
<title>Comparison of multivariate to univariate models in the classification task</title>
<p>In the classification task for predicting depression presence, the multivariate model generally outperforms univariate models, except for a small difference in favor of the RF classifier using depression scores (at T1) as a single feature (AUPRC multivariate RF classifier: 0.73, 95% CI: 0.66, 0.80; AUPRC univariate RF classifier: 0.74, 95% CI: 0.68, 0.79) (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>, <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S8</bold>
</xref>). However, it&#x2019;s crucial to note that the multivariate model for the RF classifier shows some signs of overfitting compared to both multivariate alternative classifiers (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>).</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Mean AUPRC scores for multivariate versus univariate machine learning models predicting depression in the classification task. Linear logistic regression, random forest, and support vector machine classifiers were trained either as multivariate models (encompassing all features) or univariate models (each incorporating just a single feature). Error bars represent 95% confidence intervals. AUPRC, Average Precision-Recall Curve; All, Multivariate model with all features included; Univariate models: DEP, Depression scores measured at time 1 as the single feature; ANX, Anxiety scores as the single feature; SUBP, Quarantine sub-periods as the single feature; Sex, Biological sex as the single feature; Age, Age as the single feature; MDH, Mental disorder history as the single feature; SH, Suicidal behavior history as the single feature.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpsyt-15-1376784-g004.tif"/>
</fig>
<p>Univariate classifiers, in particular LogReg and SVM using depression (at T1) or anxiety scores as single features, exhibit comparative performance levels (in terms of mean test AUPRC) that are close to those of the corresponding multivariate model that include all features. On the other hand, all other univariate models perform significantly worse, with mean AUPRC scores slightly above chance-level performance. Importantly, this trend is consistent across the three ML algorithms tested (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>, <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S8</bold>
</xref>).</p>
</sec>
<sec id="s3_3_2">
<title>Comparison of multivariate to univariate models in the regression task</title>
<p>In the regression task for predicting depression scores, the multivariate model consistently outperforms all univariate models. Notably, when considering all univariate models, depression (at T1) and anxiety scores emerge as the most predictive single features, achieving the highest performance levels (mean R2 scores), particularly when using the RR algorithm. It is noteworthy that univariate models excluding both depression (at T1) and anxiety scores exhibit performance levels slightly above (e.g., suicidal behavior history) or comparable (e.g., quarantine sub-period) to chance-level. Importantly, this pattern is consistent across the three ML algorithms. Contrary to the results on the comparison between multivariate classifiers&#x2019; performance, when comparing univariate models&#x2019; performance, RF models perform better than SVR models (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>, <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S9</bold>
</xref>).</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Comparative analysis of mean R-squared scores for multivariate versus univariate machine learning models predicting depression in the regression task. Bar plot illustrating the comparison of mean R2 scores for various machine learning models (ordinary least-squares regression, random forest regressor, and support vector regressor) in the regression task of predicting depression. The models evaluated include multivariate models (encompassing all features) and univariate models (each incorporating a single feature). The bar plots include error bars representing 95% confidence intervals and using markers to highlight data points. All: Multivariate model with all features included; Univariate models: DEP, Depression scores measured at time 1 as the single feature; ANX, Anxiety scores as the single feature; SUBP, Quarantine sub-periods as the single feature; Sex, Biological sex as the single feature; Age, Age as the single feature; MDH, Mental disorder history as the single feature; SH, Suicidal behavior history as the single feature.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpsyt-15-1376784-g005.tif"/>
</fig>
</sec>
</sec>
</sec>
<sec id="s4" sec-type="discussion">
<title>Discussion</title>
<p>This study leveraged a longitudinal dataset for the prediction of depression among college students amidst the challenging backdrop of the COVID-19 pandemic, allowing for the examination of evolving mental health patterns over time. The results indicate a gain in predictive efficacy of multivariate ML algorithms as compared to their univariate counterparts.</p>
<p>As previously mentioned, the dataset analyzed using ML in this study underwent scrutiny through MEM in a prior study (<xref ref-type="bibr" rid="B17">17</xref>). In that study, variables such as sex (female), age (younger), mental disorder history (presence), and suicidal behavior history (presence) were identified as having significant main effects on depression. However, following Cohen&#x2019;s conventions for effect sizes (ES; <xref ref-type="bibr" rid="B38">38</xref>), these effects were deemed small for sex (ES = 0.09), age (ES = 0.09), and mental disorder history (ES = 0.17), while being of medium magnitude for suicidal behavior history (ES = 0.34) (<xref ref-type="bibr" rid="B17">17</xref>). Notwithstanding, the outcomes of that prior MEM study align with existing literature, underscoring that variables associated with adverse mental health effects after and during the COVID-19 pandemic encompass baseline depression, female sex, younger age, the presence of mental disorders, and student status, among others (<xref ref-type="bibr" rid="B39">39</xref>, <xref ref-type="bibr" rid="B40">40</xref>). In the current study, the application of ML models enabled the exploration of various features&#x2019; predictive potential, aligning with those examined in the previous study (<xref ref-type="bibr" rid="B17">17</xref>). Notably, anxiety, a feature included in this ML-based approach but absent in the previous study, was revealed as one of the most relevant features, alongside depression at T1, for identifying depression presence and quantifying depression severity. The significance of anxiety as a predictor of depression underscores the interconnectedness of these mental health conditions. It aligns with existing literature that emphasizes the comorbidity and shared features between anxiety and depression (<xref ref-type="bibr" rid="B41">41</xref>&#x2013;<xref ref-type="bibr" rid="B43">43</xref>) particularly in college students during COVID-19 (<xref ref-type="bibr" rid="B44">44</xref>, <xref ref-type="bibr" rid="B45">45</xref>). As for depression at T1, these findings suggest that depressive symptoms persist over time during COVID-19. This aligns with research conducted in college students (<xref ref-type="bibr" rid="B46">46</xref>) and general populations in developed countries (<xref ref-type="bibr" rid="B47">47</xref>) and Argentina (<xref ref-type="bibr" rid="B48">48</xref>), demonstrating the enduring nature of depression symptoms amid the pandemic.</p>
<p>As for the ML algorithms&#x2019; performance in this study, both SVM/SVR and LogReg/RR demonstrated the best results in both classification and regression tasks. SVM has consistently been identified as a high-performing algorithm in predicting depression among students (see, e.g., <xref ref-type="bibr" rid="B49">49</xref>&#x2013;<xref ref-type="bibr" rid="B51">51</xref>). However, RF also emerged as the top-performing algorithm in some studies (see, e.g., <xref ref-type="bibr" rid="B50">50</xref>&#x2013;<xref ref-type="bibr" rid="B52">52</xref>), which contrasts with the present findings where it displayed some signs of overfitting and exhibited comparative lower performance. The observed overfitting in the RF algorithm in both classification and regression tasks within this study can be attributed to various factors. Although RFs are renowned for their adaptability and capacity to discern intricate data relationships, this characteristic may lead to overfitting, particularly if the model&#x2019;s complexity surpasses the dataset&#x2019;s requirements. The profusion of decision trees and their interactions may contribute to an overfitted model. Additionally, the hyperparameters of the RF play pivotal roles. In this study, a set of hyperparameters was explored, including the number of trees, the maximum depth of the tree, the minimum number of samples required to split an internal node, and the minimum number of samples required to be at a leaf node. For tree building, a bootstrap method involving sampling with replacement was used. Likewise, RFs automatically perform feature selection by considering a subset of features at each split. While advantageous, this process can lead to overfitting if the model emphasizes noise or irrelevant features. However, as demonstrated in this study, an examination of multi- versus univariate models revealed two dominant features (depression at T1 and anxiety), with the remaining features exhibiting limited significance. Just like SVM/SVR and LogReg/RR models, RF multi- and univariate models also captured these key features. As for the discrepancies in algorithm performance between different studies, these may be ascribed to several factors. To mention some of them, the diverse array of features integrated into the models can significantly impact performance when comparing studies. Additionally, variations in study designs, predominantly the widespread use of a cross-sectional design in studies analyzing questionnaire-collected data, further contribute to these differences. Depression assessment methods, relying on self-reported and clinician-administered questionnaires, exhibit inherent limitations (<xref ref-type="bibr" rid="B53">53</xref>). However, it is crucial to emphasize that, particularly in developing countries, accessing potent features capable of identifying biomarkers, such as those related to neuroimaging (<xref ref-type="bibr" rid="B54">54</xref>), is often impeded by their high cost. Consequently, there is a pressing need to identify reliable and cost-effective predictors to develop risk prediction models capable of enhancing clinical decision-making in depression diagnosis and care. This study represents a first exploratory step toward addressing this goal.</p>
<p>In addition, addressing the challenge of interpretability in ML models is crucial for their effective deployment. This challenge is particularly pronounced in the transparency of decision-making processes within advanced models, often referred to as &#x201c;black-boxes&#x201d; (<xref ref-type="bibr" rid="B55">55</xref>). Notably, complex models like RF, which leverage numerous decision trees in a non-linear interaction, exacerbate the difficulty in comprehending underlying mechanisms (<xref ref-type="bibr" rid="B27">27</xref>). In contrast, linear models such as logistic regression and RR are frequently considered to be easier to interpret due to their transparent mathematical formulations (<xref ref-type="bibr" rid="B56">56</xref>). Nevertheless, even the interpretation of multivariate linear models and decision trees can be highly misleading (<xref ref-type="bibr" rid="B57">57</xref>&#x2013;<xref ref-type="bibr" rid="B59">59</xref>). Moreover, additional challenges persist in high-dimensional datasets (<xref ref-type="bibr" rid="B60">60</xref>). For these reasons, we here resort to univariate analyses to unanimously clarify the aptitude of individual features for prediction (<xref ref-type="bibr" rid="B57">57</xref>). In this study, while multivariate ML models encompassing all features demonstrate robust performance in both classification and regression tasks, particularly with LogReg/RR and SVM/SVR, a closer examination reveals a reliance on two key features: depression (at T1) and anxiety. The relevance of these variables in identifying depression aligns with existing literature (<xref ref-type="bibr" rid="B3">3</xref>&#x2013;<xref ref-type="bibr" rid="B5">5</xref>, <xref ref-type="bibr" rid="B39">39</xref>).</p>
<p>However, this study&#x2019;s outcomes also emphasize the limited significance of certain variables, encompassing quarantine duration and restrictiveness (referred to as quarantine sub-periods), sex, age, mental disorder history, and suicidal behavior history, not only in identifying depression presence (i.e., classification task) but particularly in quantifying depression severity (i.e., regression task). This departure from the established psychological literature, which traditionally links these variables to depression prediction and diagnosis both before and during pandemics (e.g., <xref ref-type="bibr" rid="B6">6</xref>&#x2013;<xref ref-type="bibr" rid="B9">9</xref>, <xref ref-type="bibr" rid="B11">11</xref>, <xref ref-type="bibr" rid="B61">61</xref>), is noteworthy. Nevertheless, it is also worth noting that some recent research, such as a systematic review and meta-analysis, found no significant association between sex and depression among undergraduates (<xref ref-type="bibr" rid="B62">62</xref>). Likewise, it is important to consider that much of the established psychological literature is derived from studies analyzing samples in developed countries, often leading to underrepresentation from developing southern countries. Consequently, the role and impact of these variables in predicting depression can vary across different contexts and populations. Overall, the presented findings suggest that, when considered in isolation, these variables may not reliably predict depression within this population. This underscores the need for a more holistic approach to depression prediction, emphasizing the intricate interplay of multiple factors. Additionally, there is a call for further studies to explore additional factors that may be linked with depression prediction, particularly in samples from developing southern countries.</p>
<p>The observed disparities among these current ML-based results, prior MEM analysis, and existing literature bear implications for translational research, as well as for health decision-makers and policymakers striving to improve mental health, particularly to offer scalable and effective evidence-based interventions addressing depression (<xref ref-type="bibr" rid="B63">63</xref>). Reliable predictors and prediction models are crucial for effective mental health interventions, particularly those targeting depression prevention (<xref ref-type="bibr" rid="B64">64</xref>). While internal validation provides strong support for the performance of the models tested in this study, it is imperative to underscore the necessity for additional external validation and replication studies to solidify the implications of these findings. The prevailing landscape of developing clinical prediction models related to mental disorders calls for more rigorous validation procedures. The current studies in this field often exhibit a deficiency in both external and internal validation (<xref ref-type="bibr" rid="B63">63</xref>), emphasizing a critical weakness that needs to be addressed in future research.</p>
<p>Especially in the post-COVID-19 era, there is a growing imperative to leverage advanced technologies like ML algorithms and artificial intelligence to enhance the precision of depression screening among college students. This, in turn, can enable proactive prevention efforts and contribute significantly to improving mental health outcomes (<xref ref-type="bibr" rid="B65">65</xref>). However, the applicability of ML algorithms to depression prediction and the development of evidence-based tools also entails ethical challenges. These include optimizing predictions across diverse datasets, ensuring data protection and privacy, assessing feasibility in clinical practice, addressing issues of fairness, accountability, and transparency, as well as identifying and mitigating potential biases of ML models. Many of these challenges have not yet been fully addressed or solved by research studies (<xref ref-type="bibr" rid="B66">66</xref>&#x2013;<xref ref-type="bibr" rid="B68">68</xref>). Likewise, as discussed above, the non-trivial interpretability and explainability of ML algorithms is a well-recognized problem. Furthermore, potential future application of these findings, such as widespread depression screening using ML and automatic diagnosis of disorders, raise additional questions. These include whether clinicians need to confirm diagnoses and eventually communicate them to individuals. Thus, the involvement of these algorithms in diagnosis decision-making has to be accompanied by ethical reflection (<xref ref-type="bibr" rid="B67">67</xref>).</p>
<p>This study comprises a valuable exploration of using ML models to predict depression among college students during the COVID-19 quarantine. However, the interpretation of these results is subject to certain limitations. The focus on a specific population and cultural context limits the generalizability and applicability of the developed ML models. Future studies should endeavor to incorporate data from diverse demographics and cultural backgrounds. Additionally, potential biases introduced by the data collection process, which are discussed elsewhere (<xref ref-type="bibr" rid="B17">17</xref>), should also be considered. Moreover, the limitations of the ML models, as acknowledged above, should be considered.</p>
<p>In conclusion, this study not only advances our understanding of depression prediction in college students from a developing country using ML algorithms but also highlights the need for further exploration and validation of predictive models in diverse populations. Future research should aim to refine models, consider additional psychological and socio-environmental factors, and enhance external validation to ensure the broad applicability of these findings. For instance, further studies should include larger and more diverse datasets, incorporating the features analyzed in this study as well as other potentially relevant ones. Specifically, future studies may involve additional psychological measurements, such as capturing somatic symptoms of coronavirus-related anxiety, and analyze specific mental disorder diagnoses instead of examining a general category of mental disorder background, as done in this study. Additionally, a comparison with another similar cohort should be conducted as part of future work. Moreover, future research should focus on improving the interpretability of multivariate models. As suggested in this study, comparing multivariate and univariate models may help elucidate the interpretability of the former. This way, the implementation of scalable and effective interventions for diagnosing depression may become more attainable. This initial research study serves to explore the viability of data-driven algorithms in detecting depression among college students. While this study reveals interesting correlations, these findings do not necessarily imply immediate practical consequences. To thoroughly understand and establish causal relationships between anxiety and depression, or to develop evidence-based tools, further investigation is needed. In the future, models similar to those presented in this study may be employed for online pre-screening of students for depression at home.</p>
</sec>
<sec id="s5" sec-type="data-availability">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: The dataset is available in the Open Science Framework (OSF) repository, <uri xlink:href="https://doi.org/10.17605/OSF.IO/2V84N">https://doi.org/10.17605/OSF.IO/2V84N</uri>. The reproducible Python code is available in the Open Science Framework (OSF) repository, <uri xlink:href="https://doi.org/10.17605/OSF.IO/QT2GU">https://doi.org/10.17605/OSF.IO/QT2GU</uri>, and in the GitHub repository, <uri xlink:href="https://github.com/cecilost/machine-learning-to-predict-depression-in-college-students-during-the-COVID-19-pandemic">https://github.com/cecilost/machine-learning-to-predict-depression-in-college-students-during-the-COVID-19-pandemic</uri>.</p>
</sec>
<sec id="s6" sec-type="ethics-statement">
<title>Ethics statement</title>
<p>The studies involving humans were approved by Ethics Committee of the Psychological Research Institute, Universidad Nacional de C&#xf3;rdoba (CEIIPsi-UNC-CONICET). The studies were conducted in accordance with the local legislation and institutional requirements. The participants provided their written informed consent to participate in this study.</p>
</sec>
<sec id="s7" sec-type="author-contributions">
<title>Author contributions</title>
<p>LL: Writing &#x2013; review &amp; editing, Writing &#x2013; original draft, Visualization, Validation, Supervision, Resources, Project administration, Methodology, Investigation, Formal analysis, Data curation, Conceptualization. MS: Writing &#x2013; original draft, Formal analysis. RZ: Writing &#x2013; original draft, Formal analysis. JG: Writing &#x2013; review &amp; editing, Supervision, Resources. SH: Writing &#x2013; review &amp; editing, Writing &#x2013; original draft, Supervision, Resources, Project administration, Funding acquisition.</p>
</sec>
</body>
<back>
<sec id="s8" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. This result is part of a project that has received funding from the European Research Council (ERC) under the European Union&#x2019;s Horizon 2020 research and innovation programme (Grant agreement No. 758985).</p>
</sec>
<sec id="s9" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s10" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s11" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fpsyt.2024.1376784/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fpsyt.2024.1376784/full#supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet_1.docx" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>A</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Han</surname> <given-names>N</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>H</given-names>
</name>
</person-group>. <article-title>Impact of the COVID-19 pandemic on the mental health of college students: a systematic review and meta-analysis</article-title>. <source>Front Psychol</source>. (<year>2021</year>) <volume>12</volume>:<elocation-id>669119</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpsyg.2021.669119</pub-id>
</citation>
</ref>
<ref id="B2">
<label>2</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>C</given-names>
</name>
<name>
<surname>Wen</surname> <given-names>W</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>H</given-names>
</name>
<name>
<surname>Ni</surname> <given-names>J</given-names>
</name>
<name>
<surname>Jiang</surname> <given-names>J</given-names>
</name>
<name>
<surname>Cheng</surname> <given-names>Y</given-names>
</name>
<etal/>
</person-group>. <article-title>Anxiety, depression, and stress prevalence among college students during the COVID-19 pandemic: a systematic review and meta-analysis</article-title>. <source>J Am Coll Health</source>. (<year>2021</year>) <volume>1</volume>:<fpage>1</fpage>&#x2013;<lpage>8</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1080/07448481.2021.1960849</pub-id>
</citation>
</ref>
<ref id="B3">
<label>3</label>
<citation citation-type="book">
<person-group person-group-type="author">
<collab>American Psychiatric Association</collab>
</person-group>. <source>Diagnostic and statistical manual of mental disorders</source>. <edition>4th ed</edition>. <publisher-loc>Washington, DC</publisher-loc>: <publisher-name>American Psychiatric Association</publisher-name> (<year>2000</year>).</citation>
</ref>
<ref id="B4">
<label>4</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Horn</surname> <given-names>PJ</given-names>
</name>
<name>
<surname>Wuyek</surname> <given-names>LA</given-names>
</name>
</person-group>. <article-title>Anxiety disorders as a risk factor for subsequent depression</article-title>. <source>Int J Psychiatry Clin Pract</source>. (<year>2010</year>) <volume>14</volume>:<page-range>244&#x2013;7</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.3109/13651501.2010.487979</pub-id>
</citation>
</ref>
<ref id="B5">
<label>5</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cassady</surname> <given-names>JC</given-names>
</name>
<name>
<surname>Pierson</surname> <given-names>EE</given-names>
</name>
<name>
<surname>Starling</surname> <given-names>JM</given-names>
</name>
</person-group>. <article-title>Predicting student depression with measures of general and academic anxieties</article-title>. <source>Front Educ</source>. (<year>2019</year>) <volume>4</volume>:<elocation-id>11</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/feduc.2019.00011</pub-id>
</citation>
</ref>
<ref id="B6">
<label>6</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sheldon</surname> <given-names>E</given-names>
</name>
<name>
<surname>Simmonds-Buckley</surname> <given-names>M</given-names>
</name>
<name>
<surname>Bone</surname> <given-names>C</given-names>
</name>
<name>
<surname>Mascarenhas</surname> <given-names>T</given-names>
</name>
<name>
<surname>Chan</surname> <given-names>N</given-names>
</name>
<name>
<surname>Wincott</surname> <given-names>M</given-names>
</name>
<etal/>
</person-group>. <article-title>Prevalence and risk factors for mental health problems in university undergraduate students: a systematic review with meta-analysis</article-title>. <source>J Affect Disord</source>. (<year>2021</year>) <volume>287</volume>:<page-range>282&#x2013;92</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jad.2021.03.054</pub-id>
</citation>
</ref>
<ref id="B7">
<label>7</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kisch</surname> <given-names>J</given-names>
</name>
<name>
<surname>Leino</surname> <given-names>EV</given-names>
</name>
<name>
<surname>Silverman</surname> <given-names>MM</given-names>
</name>
</person-group>. <article-title>Aspects of suicidal behavior, depression, and treatment in college students: results from the spring 2000 national college health assessment survey</article-title>. <source>Suicide Life Threat Behav</source>. (<year>2005</year>) <volume>35</volume>:<fpage>3</fpage>&#x2013;<lpage>13</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1521/suli.35.1.3.59263</pub-id>
</citation>
</ref>
<ref id="B8">
<label>8</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Piccinelli</surname> <given-names>M</given-names>
</name>
<name>
<surname>Wilkinson</surname> <given-names>G</given-names>
</name>
</person-group>. <article-title>Gender differences in depression. Critical review</article-title>. <source>Br J Psychiatry</source>. (<year>2000</year>) <volume>177</volume>:<page-range>486&#x2013;92</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1192/bjp.177.6.486</pub-id>
</citation>
</ref>
<ref id="B9">
<label>9</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nolen-Hoeksema</surname> <given-names>S</given-names>
</name>
<name>
<surname>Aldao</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>Gender and age differences in emotion regulation strategies and their relationship to depressive symptoms</article-title>. <source>Pers Individ Differ</source>. (<year>2011</year>) <volume>51</volume>:<page-range>704&#x2013;8</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.paid.2011.06.012</pub-id>
</citation>
</ref>
<ref id="B10">
<label>10</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jorm</surname> <given-names>AF</given-names>
</name>
</person-group>. <article-title>Sex and age differences in depression: a quantitative synthesis of published research</article-title>. <source>Aust N Z J Psychiatry</source>. (<year>1987</year>) <volume>21</volume>:<fpage>46</fpage>&#x2013;<lpage>53</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3109/00048678709160898</pub-id>
</citation>
</ref>
<ref id="B11">
<label>11</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brooks</surname> <given-names>S</given-names>
</name>
<name>
<surname>Webster</surname> <given-names>RK</given-names>
</name>
<name>
<surname>Smith</surname> <given-names>LE</given-names>
</name>
<name>
<surname>Woodland</surname> <given-names>L</given-names>
</name>
<name>
<surname>Wessely</surname> <given-names>S</given-names>
</name>
<name>
<surname>Greenberg</surname> <given-names>N</given-names>
</name>
<etal/>
</person-group>. <article-title>The psychological impact of quarantine and how to reduce it: rapid review of the evidence</article-title>. <source>Lancet</source>. (<year>2020</year>) <volume>395</volume>:<page-range>912&#x2013;20</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/S0140-6736(20)30460-8</pub-id>
</citation>
</ref>
<ref id="B12">
<label>12</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jacobson</surname> <given-names>NC</given-names>
</name>
<name>
<surname>Newman</surname> <given-names>MG</given-names>
</name>
</person-group>. <article-title>Anxiety and depression as bidirectional risk factors for one another: a meta-analysis of longitudinal studies</article-title>. <source>Psychol Bull</source>. (<year>2017</year>) <volume>143</volume>:<page-range>1155&#x2013;200</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1037/bul0000111</pub-id>
</citation>
</ref>
<ref id="B13">
<label>13</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Hastie</surname> <given-names>T</given-names>
</name>
<name>
<surname>Tibshirani</surname> <given-names>R</given-names>
</name>
<name>
<surname>Friedman</surname> <given-names>J</given-names>
</name>
</person-group>. <source>The elements of statistical learning: data mining, inference, and prediction</source>. <edition>2nd ed</edition>. <publisher-loc>New York, US</publisher-loc>: <publisher-name>Springer Science &amp; Business Media</publisher-name> (<year>2009</year>).</citation>
</ref>
<ref id="B14">
<label>14</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Murphy</surname> <given-names>KP</given-names>
</name>
</person-group>. <source>Machine learning: a probabilistic perspective</source>. <publisher-loc>Cambridge, MA</publisher-loc>: <publisher-name>MIT press</publisher-name> (<year>2012</year>).</citation>
</ref>
<ref id="B15">
<label>15</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kuhn</surname> <given-names>M</given-names>
</name>
<name>
<surname>Johnson</surname> <given-names>K</given-names>
</name>
</person-group>. <source>Applied predictive modeling</source>. <publisher-loc>New York</publisher-loc>: <publisher-name>Springer</publisher-name> (<year>2013</year>).</citation>
</ref>
<ref id="B16">
<label>16</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>James</surname> <given-names>G</given-names>
</name>
<name>
<surname>Witten</surname> <given-names>D</given-names>
</name>
<name>
<surname>Hastie</surname> <given-names>T</given-names>
</name>
<name>
<surname>Tibshirani</surname> <given-names>R</given-names>
</name>
</person-group>. <source>An introduction to statistical learning: with applications in R</source>. <publisher-loc>New York</publisher-loc>: <publisher-name>Springer</publisher-name> (<year>2017</year>).</citation>
</ref>
<ref id="B17">
<label>17</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>L&#xf3;pez Steinmetz</surname> <given-names>LC</given-names>
</name>
<name>
<surname>Godoy</surname> <given-names>JC</given-names>
</name>
<name>
<surname>Fong</surname> <given-names>SB</given-names>
</name>
</person-group>. <article-title>A longitudinal study on depression and anxiety in college students during the first 106-days of the lengthy Argentinean quarantine for the COVID-19 pandemic</article-title>. <source>J Ment Health</source>. (<year>2023</year>) <volume>32</volume>:<page-range>1030&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1080/09638237.2021.1952952</pub-id>
</citation>
</ref>
<ref id="B18">
<label>18</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Iyortsuun</surname> <given-names>NK</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>SH</given-names>
</name>
<name>
<surname>Jhon</surname> <given-names>M</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>HJ</given-names>
</name>
<name>
<surname>Pant</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>A review of machine learning and deep learning approaches on mental health diagnosis</article-title>. <source>Healthcare (Basel)</source>. (<year>2023</year>) <volume>11</volume>:<elocation-id>285</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/healthcare11030285</pub-id>
</citation>
</ref>
<ref id="B19">
<label>19</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>L&#xf3;pez Steinmetz</surname> <given-names>LC</given-names>
</name>
</person-group>. <article-title>Dataset and R Code for: A longitudinal study on depression and anxiety in college students during the first 106-days of the lengthy Argentinean quarantine for the COVID-19 pandemic</article-title>. <source>OSF Repository</source>. (<year>2023</year>). doi:&#xa0;<pub-id pub-id-type="doi">10.17605/OSF.IO/2V84N</pub-id>
</citation>
</ref>
<ref id="B20">
<label>20</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Brenlla</surname> <given-names>ME</given-names>
</name>
<name>
<surname>Rodr&#xed;guez</surname> <given-names>CM</given-names>
</name>
</person-group>. <article-title>Adaptacion Argentina del Inventario de Depresion de Beck (BDI-II) [Argentinean adaptation of the Beck Depression Inventory (BDI-II)]</article-title>. In: <person-group person-group-type="editor">
<name>
<surname>Beck</surname> <given-names>AT</given-names>
</name>
<name>
<surname>Steer</surname> <given-names>RA</given-names>
</name>
<name>
<surname>Brown</surname> <given-names>GK</given-names>
</name>
</person-group> (editors) <source>Inventario de Depresion de Beck, BDIII [Beck Depression Inventory, BDI-II]</source> (<year>2006</year>). p. <fpage>11</fpage>&#x2013;<lpage>37</lpage>. <publisher-name>Buenos Aires</publisher-name>, <publisher-loc>Paid&#xf3;s</publisher-loc>.</citation>
</ref>
<ref id="B21">
<label>21</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Beck</surname> <given-names>AT</given-names>
</name>
<name>
<surname>Steer</surname> <given-names>RA</given-names>
</name>
<name>
<surname>Brown</surname> <given-names>GK</given-names>
</name>
</person-group>. <source>Manual for the Beck depression inventory II</source>. (<publisher-loc>San Antonio</publisher-loc>: <publisher-name>Psychol Corporation</publisher-name>). (<year>1996</year>).</citation>
</ref>
<ref id="B22">
<label>22</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Spielberger</surname> <given-names>C</given-names>
</name>
<name>
<surname>Gorsuch</surname> <given-names>RL</given-names>
</name>
<name>
<surname>Lushene</surname> <given-names>RE</given-names>
</name>
</person-group>. <source>Manual for the state-trait anxiety inventory. STAI (Form Y), self-evaluation questionnaire</source>. (<publisher-loc>Palo Alto, California</publisher-loc>: <publisher-name>Consulting Psychologists Press</publisher-name>) (<year>1983</year>).</citation>
</ref>
<ref id="B23">
<label>23</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kendall</surname> <given-names>PC</given-names>
</name>
<name>
<surname>Hollon</surname> <given-names>SD</given-names>
</name>
<name>
<surname>Beck</surname> <given-names>AT</given-names>
</name>
<name>
<surname>Hammen</surname> <given-names>CL</given-names>
</name>
<name>
<surname>Ingram</surname> <given-names>RE</given-names>
</name>
</person-group>. <article-title>Issues and recommendations regarding use of the Beck Depression Inventory</article-title>. <source>Cogn Ther Res</source>. (<year>1987</year>) <volume>11</volume>:<page-range>289&#x2013;99</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/BF01186280</pub-id>
</citation>
</ref>
<ref id="B24">
<label>24</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pedregosa</surname> <given-names>F</given-names>
</name>
<name>
<surname>Varoquaux</surname> <given-names>G</given-names>
</name>
<name>
<surname>Gramfort</surname> <given-names>A</given-names>
</name>
<name>
<surname>Michel</surname> <given-names>V</given-names>
</name>
<name>
<surname>Thirion</surname> <given-names>B</given-names>
</name>
<name>
<surname>Grisel</surname> <given-names>O</given-names>
</name>
<etal/>
</person-group>. <article-title>Scikit-learn: machine learning in Python</article-title>. <source>J Mach Learn Res</source>. (<year>2011</year>) <volume>12</volume>, <page-range>2825&#x2013;30</page-range>.</citation>
</ref>
<ref id="B25">
<label>25</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>McKinney</surname> <given-names>W</given-names>
</name>
</person-group>. (<year>2010</year>). <article-title>Data structures for statistical computing in Python</article-title>, in: <conf-name>Proceedings of the 9th Python in Science Conference</conf-name>. pp. <page-range>56&#x2013;61</page-range>. doi: <pub-id pub-id-type="doi">10.25080/Majora-92bf1922-00a</pub-id>
</citation>
</ref>
<ref id="B26">
<label>26</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>van der Walt</surname> <given-names>S</given-names>
</name>
<name>
<surname>Colbert</surname> <given-names>SC</given-names>
</name>
<name>
<surname>Varoquaux</surname> <given-names>G</given-names>
</name>
</person-group>. <article-title>The NumPy array: a structure for efficient numerical computation</article-title>. <source>Comput Sci Eng</source>. (<year>2011</year>) <volume>13</volume>:<fpage>22</fpage>&#x2013;<lpage>30</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/MCSE.2011.37</pub-id>
</citation>
</ref>
<ref id="B27">
<label>27</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Breiman</surname> <given-names>L</given-names>
</name>
</person-group>. <article-title>Random forests</article-title>. <source>Mach Learn</source>. (<year>2001</year>) <volume>45</volume>:<fpage>5</fpage>&#x2013;<lpage>32</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1023/A:1010933404324</pub-id>
</citation>
</ref>
<ref id="B28">
<label>28</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cortes</surname> <given-names>C</given-names>
</name>
<name>
<surname>Vapnik</surname> <given-names>V</given-names>
</name>
</person-group>. <article-title>Support-vector networks</article-title>. <source>Mach Learn</source>. (<year>1995</year>) <volume>20</volume>:<page-range>273&#x2013;97</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/BF00994018</pub-id>
</citation>
</ref>
<ref id="B29">
<label>29</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Uddin</surname> <given-names>S</given-names>
</name>
<name>
<surname>Khan</surname> <given-names>A</given-names>
</name>
<name>
<surname>Hossain</surname> <given-names>ME</given-names>
</name>
<name>
<surname>Moni</surname> <given-names>MA</given-names>
</name>
</person-group>. <article-title>Comparing different supervised machine learning algorithms for disease prediction</article-title>. <source>BMC Med Inform Decis Mak</source>. (<year>2019</year>) <volume>19</volume>:<fpage>281</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12911-019-1004-8</pub-id>
</citation>
</ref>
<ref id="B30">
<label>30</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Geurts</surname> <given-names>P</given-names>
</name>
<name>
<surname>Ernst</surname> <given-names>D</given-names>
</name>
<name>
<surname>Wehenkel</surname> <given-names>L</given-names>
</name>
</person-group>. <article-title>Extremely randomized trees</article-title>. <source>Mach Learn</source>. (<year>2006</year>) <volume>63</volume>:<fpage>3</fpage>&#x2013;<lpage>42</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s10994-006-6226-1</pub-id>
</citation>
</ref>
<ref id="B31">
<label>31</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chang</surname> <given-names>CC</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>CJ</given-names>
</name>
</person-group>. <article-title>LIBSVM: A library for support vector machines</article-title>. <source>ACM Trans Intell Syst Technol</source>. (<year>2011</year>) <volume>2</volume>(<issue>3</issue>):<fpage>1</fpage>&#x2013;<lpage>27</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1145/1961189.1961199</pub-id>
</citation>
</ref>
<ref id="B32">
<label>32</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Powers</surname> <given-names>DM</given-names>
</name>
</person-group>. <article-title>Evaluation: from precision, recall and f-measure to ROC, informedness, markedness and correlation</article-title>. <source>J Mach Learn Technol</source>. (<year>2011</year>) <volume>2</volume>:<fpage>37</fpage>&#x2013;<lpage>63</lpage>.</citation>
</ref>
<ref id="B33">
<label>33</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Davis</surname> <given-names>J</given-names>
</name>
<name>
<surname>Goadrich</surname> <given-names>M</given-names>
</name>
</person-group>. (<year>2006</year>). <article-title>The relationship between Precision-Recall and ROC curves</article-title>, in: <conf-name>Proceedings of the 23rd international conference on Machine learning</conf-name>. <publisher-loc>New York</publisher-loc>: <publisher-name>ACM (Association for Computing Machinery)</publisher-name>. pp. <page-range>233&#x2013;40</page-range>.</citation>
</ref>
<ref id="B34">
<label>34</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fawcett</surname> <given-names>T</given-names>
</name>
</person-group>. <article-title>An introduction to ROC analysis</article-title>. <source>Pattern Recognit Lett</source>. (<year>2006</year>) <volume>27</volume>:<page-range>861&#x2013;74</page-range>. doi: <pub-id pub-id-type="doi">10.1016/j.patrec.2005.10.010</pub-id>
</citation>
</ref>
<ref id="B35">
<label>35</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brier</surname> <given-names>GW</given-names>
</name>
</person-group>. <article-title>Verification of forecasts expressed in terms of probability</article-title>. <source>Mon Weather Rev</source>. (<year>1950</year>) <volume>78</volume>:<fpage>1</fpage>&#x2013;<lpage>3</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1175/1520-0493(1950)078%3C0001:VOFEIT%3E2.0.CO;2</pub-id>
</citation>
</ref>
<ref id="B36">
<label>36</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Bishop</surname> <given-names>CM</given-names>
</name>
</person-group>. <source>Pattern recognition and machine learning</source> Vol. <volume>4</volume>. (<publisher-loc>New York</publisher-loc>: <publisher-name>Springer</publisher-name>) (<year>2006</year>).</citation>
</ref>
<ref id="B37">
<label>37</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Davison</surname> <given-names>AC</given-names>
</name>
<name>
<surname>Hinkley</surname> <given-names>DV</given-names>
</name>
</person-group>. <source>Bootstrap methods and their application</source>. <publisher-loc>New York</publisher-loc>: <publisher-name>Cambridge University Press</publisher-name> (<year>1997</year>).</citation>
</ref>
<ref id="B38">
<label>38</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Cohen</surname> <given-names>J</given-names>
</name>
</person-group>. <source>Statistical power analysis for the behavioral sciences</source>. <edition>2nd Edition</edition>. <publisher-loc>New York</publisher-loc>: <publisher-name>Lawrence Erlbaum Associates</publisher-name> (<year>1988</year>).</citation>
</ref>
<ref id="B39">
<label>39</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>N</given-names>
</name>
<name>
<surname>Bao</surname> <given-names>G</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Ji</surname> <given-names>B</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>Y</given-names>
</name>
<etal/>
</person-group>. <article-title>Predictors of depressive symptoms in college students: a systematic review and meta-analysis of cohort studies</article-title>. <source>J Affect Disord</source>. (<year>2019</year>) <volume>244</volume>:<fpage>196</fpage>&#x2013;<lpage>208</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jad.2018.10.084</pub-id>
</citation>
</ref>
<ref id="B40">
<label>40</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xiong</surname> <given-names>J</given-names>
</name>
<name>
<surname>Lipsitz</surname> <given-names>O</given-names>
</name>
<name>
<surname>Nasri</surname> <given-names>F</given-names>
</name>
<name>
<surname>Lui</surname> <given-names>LMW</given-names>
</name>
<name>
<surname>Gill</surname> <given-names>H</given-names>
</name>
<name>
<surname>Phan</surname> <given-names>L</given-names>
</name>
<etal/>
</person-group>. <article-title>Impact of COVID-19 pandemic on mental health in the general population: a systematic review</article-title>. <source>J Affect Disord</source>. (<year>2020</year>) <volume>277</volume>:<fpage>55</fpage>&#x2013;<lpage>64</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jad.2020.08.001</pub-id>
</citation>
</ref>
<ref id="B41">
<label>41</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brown</surname> <given-names>TA</given-names>
</name>
<name>
<surname>Campbell</surname> <given-names>LA</given-names>
</name>
<name>
<surname>Lehman</surname> <given-names>CL</given-names>
</name>
<name>
<surname>Grisham</surname> <given-names>JR</given-names>
</name>
<name>
<surname>Mancill</surname> <given-names>RB</given-names>
</name>
</person-group>. <article-title>Current and lifetime comorbidity of the DSM-IV anxiety and mood disorders in a large clinical sample</article-title>. <source>J Abnorm Psychol</source>. (<year>2001</year>) <volume>110</volume>:<page-range>585&#x2013;99</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1037//0021-843x.110.4.585</pub-id>
</citation>
</ref>
<ref id="B42">
<label>42</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lamers</surname> <given-names>F</given-names>
</name>
<name>
<surname>van Oppen</surname> <given-names>P</given-names>
</name>
<name>
<surname>Comijs</surname> <given-names>HC</given-names>
</name>
<name>
<surname>Smit</surname> <given-names>JH</given-names>
</name>
<name>
<surname>Spinhoven</surname> <given-names>P</given-names>
</name>
<name>
<surname>van Balkom</surname> <given-names>AJ</given-names>
</name>
<etal/>
</person-group>. <article-title>Comorbidity patterns of anxiety and depressive disorders in a large cohort study: the Netherlands Study of Depression and Anxiety (NESDA)</article-title>. <source>J Clin Psychiatry</source>. (<year>2011</year>) <volume>72</volume>:<page-range>341&#x2013;8</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.4088/JCP.10m06176blu</pub-id>
</citation>
</ref>
<ref id="B43">
<label>43</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kalin</surname> <given-names>NH</given-names>
</name>
</person-group>. <article-title>The critical relationship between anxiety and depression</article-title>. <source>Am J Psychiatry</source>. (<year>2020</year>) <volume>177</volume>:<page-range>365&#x2013;7</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1176/appi.ajp.2020.20030305</pub-id>
</citation>
</ref>
<ref id="B44">
<label>44</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bai</surname> <given-names>W</given-names>
</name>
<name>
<surname>Cai</surname> <given-names>H</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>S</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>X</given-names>
</name>
<name>
<surname>Sha</surname> <given-names>S</given-names>
</name>
<name>
<surname>Cheung</surname> <given-names>T</given-names>
</name>
<etal/>
</person-group>. <article-title>Anxiety and depressive symptoms in college students during the late stage of the COVID-19 outbreak: a network approach</article-title>. <source>Transl Psychiatry</source>. (<year>2021</year>) <volume>11</volume>:<fpage>638</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41398-021-01738-4</pub-id>
</citation>
</ref>
<ref id="B45">
<label>45</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chang</surname> <given-names>JJ</given-names>
</name>
<name>
<surname>Ji</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Li</surname> <given-names>YH</given-names>
</name>
<name>
<surname>Pan</surname> <given-names>HF</given-names>
</name>
<name>
<surname>Su</surname> <given-names>PY</given-names>
</name>
</person-group>. <article-title>Prevalence of anxiety symptom and depressive symptom among college students during COVID-19 pandemic: a meta-analysis</article-title>. <source>J Affect Disord</source>. (<year>2021</year>) <volume>292</volume>:<page-range>242&#x2013;54</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jad.2021.05.109</pub-id>
</citation>
</ref>
<ref id="B46">
<label>46</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Liang</surname> <given-names>K</given-names>
</name>
<name>
<surname>Qu</surname> <given-names>D</given-names>
</name>
<name>
<surname>He</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Wei</surname> <given-names>X</given-names>
</name>
<name>
<surname>Chi</surname> <given-names>X</given-names>
</name>
</person-group>. <article-title>The longitudinal features of depressive symptoms during the COVID-19 pandemic among Chinese college students: a network perspective</article-title>. <source>J Youth Adolesc</source>. (<year>2023</year>) <volume>52</volume>:<page-range>2031&#x2013;44</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s10964-023-01802-w</pub-id>
</citation>
</ref>
<ref id="B47">
<label>47</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ettman</surname> <given-names>CK</given-names>
</name>
<name>
<surname>Cohen</surname> <given-names>GH</given-names>
</name>
<name>
<surname>Abdalla</surname> <given-names>SM</given-names>
</name>
<name>
<surname>Sampson</surname> <given-names>L</given-names>
</name>
<name>
<surname>Trinquart</surname> <given-names>L</given-names>
</name>
<name>
<surname>Castrucci</surname> <given-names>BC</given-names>
</name>
<etal/>
</person-group>. <article-title>Persistent depressive symptoms during COVID-19: a national, population-representative, longitudinal study of U.S. adults</article-title>. <source>Lancet Reg Health Am</source>. (<year>2022</year>) <volume>5</volume>:<elocation-id>100091</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.lana.2021.100091</pub-id>
</citation>
</ref>
<ref id="B48">
<label>48</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Canet-Juric</surname> <given-names>L</given-names>
</name>
<name>
<surname>Andr&#xe9;s</surname> <given-names>ML</given-names>
</name>
<name>
<surname>Del Valle</surname> <given-names>M</given-names>
</name>
<name>
<surname>L&#xf3;pez-Morales</surname> <given-names>H</given-names>
</name>
<name>
<surname>Po&#xf3;</surname> <given-names>F</given-names>
</name>
<name>
<surname>Galli</surname> <given-names>JI</given-names>
</name>
<etal/>
</person-group>. <article-title>A longitudinal study on the emotional impact cause by the COVID-19 pandemic quarantine on general population</article-title>. <source>Front Psychol</source>. (<year>2020</year>) <volume>11</volume>:<elocation-id>565688</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpsyg.2020.565688</pub-id>
</citation>
</ref>
<ref id="B49">
<label>49</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Choudhury</surname> <given-names>AA</given-names>
</name>
<name>
<surname>Khan</surname> <given-names>MRH</given-names>
</name>
<name>
<surname>Nahim</surname> <given-names>NZ</given-names>
</name>
<name>
<surname>Tulon</surname> <given-names>SR</given-names>
</name>
<name>
<surname>Islam</surname> <given-names>S</given-names>
</name>
<name>
<surname>Chakrabarty</surname> <given-names>A</given-names>
</name>
</person-group>. (<year>2019</year>). <article-title>Predicting depression in Bangladeshi undergraduates using machine learning</article-title>, in: <conf-name>2019 IEEE Region 10 Symposium (TENSYMP)</conf-name>. <publisher-loc>New York</publisher-loc>: <publisher-name>IEEE (Institute of Electrical and Electronics Engineers)</publisher-name> pp. <page-range>789&#x2013;94</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TENSYMP46218.2019.8971369</pub-id>
</citation>
</ref>
<ref id="B50">
<label>50</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gil</surname> <given-names>M</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>SS</given-names>
</name>
<name>
<surname>Min</surname> <given-names>EJ</given-names>
</name>
</person-group>. <article-title>Machine learning models for predicting risk of depression in Korean college students: identifying family and individual factors</article-title>. <source>Front Public Health</source>. (<year>2022</year>) <volume>10</volume>:<elocation-id>1023010</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpubh.2022.1023010</pub-id>
</citation>
</ref>
<ref id="B51">
<label>51</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qasrawi</surname> <given-names>R</given-names>
</name>
<name>
<surname>Vicuna Polo</surname> <given-names>SP</given-names>
</name>
<name>
<surname>Abu Al-Halawa</surname> <given-names>D</given-names>
</name>
<name>
<surname>Hallaq</surname> <given-names>S</given-names>
</name>
<name>
<surname>Abdeen</surname> <given-names>Z</given-names>
</name>
</person-group>. <article-title>assessment and prediction of depression and anxiety risk factors in schoolchildren: machine learning techniques performance analysis</article-title>. <source>JMIR Form Res</source>. (<year>2022</year>) <volume>6</volume>:<elocation-id>e32736</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.2196/32736</pub-id>
</citation>
</ref>
<ref id="B52">
<label>52</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rois</surname> <given-names>R</given-names>
</name>
<name>
<surname>Ray</surname> <given-names>M</given-names>
</name>
<name>
<surname>Rahman</surname> <given-names>A</given-names>
</name>
<name>
<surname>Roy</surname> <given-names>SK</given-names>
</name>
</person-group>. <article-title>Prevalence and predicting factors of perceived stress among Bangladeshi university students using machine learning algorithms</article-title>. <source>J Health Popul Nutr</source>. (<year>2021</year>) <volume>40</volume>:<fpage>50</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s41043-021-00276-5</pub-id>
</citation>
</ref>
<ref id="B53">
<label>53</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fried</surname> <given-names>EI</given-names>
</name>
<name>
<surname>Nesse</surname> <given-names>RM</given-names>
</name>
</person-group>. <article-title>Depression is not a consistent syndrome: an investigation of unique symptom patterns in the STAR*D study</article-title>. <source>J Affect Disord</source>. (<year>2015</year>) <volume>172</volume>:<fpage>96</fpage>&#x2013;<lpage>102</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jad.2014.10.010</pub-id>
</citation>
</ref>
<ref id="B54">
<label>54</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wise</surname> <given-names>T</given-names>
</name>
<name>
<surname>Cleare</surname> <given-names>AJ</given-names>
</name>
<name>
<surname>Herane</surname> <given-names>A</given-names>
</name>
<name>
<surname>Young</surname> <given-names>AH</given-names>
</name>
<name>
<surname>Arnone</surname> <given-names>D</given-names>
</name>
</person-group>. <article-title>Diagnostic and therapeutic utility of neuroimaging in depression: an overview</article-title>. <source>Neuropsychiatr Dis Treat</source>. (<year>2014</year>) <volume>10</volume>:<page-range>1509&#x2013;22</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.2147/NDT.S50156</pub-id>
</citation>
</ref>
<ref id="B55">
<label>55</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ribeiro</surname> <given-names>MT</given-names>
</name>
<name>
<surname>Singh</surname> <given-names>S</given-names>
</name>
<name>
<surname>Guestrin</surname> <given-names>C</given-names>
</name>
</person-group>. (<year>2016</year>). <article-title>&#x201c;Why should I trust you?&#x201d; Explaining the predictions of any classifier</article-title>, in: <conf-name>Proceedings of the 22nd ACM SIGKDD international conference on knowledge discovery and data mining</conf-name>. <publisher-loc>New York</publisher-loc>: <publisher-name>ACM</publisher-name> pp. <page-range>1135&#x2013;44</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1145/2939672.2939778</pub-id>
</citation>
</ref>
<ref id="B56">
<label>56</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Hosmer</surname> <given-names>Jr. DW</given-names>
</name>
<name>
<surname>Lemeshow</surname> <given-names>S</given-names>
</name>
</person-group>. <source>Applied logistic regression</source>. <edition>2nd Ed</edition>. <publisher-loc>New York</publisher-loc>: <publisher-name>John Wiley &amp; Sons</publisher-name> (<year>2013</year>).</citation>
</ref>
<ref id="B57">
<label>57</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Haufe</surname> <given-names>S</given-names>
</name>
<name>
<surname>Meinecke</surname> <given-names>F</given-names>
</name>
<name>
<surname>G&#xf6;rgen</surname> <given-names>K</given-names>
</name>
<name>
<surname>D&#xe4;hne</surname> <given-names>S</given-names>
</name>
<name>
<surname>Haynes</surname> <given-names>JD</given-names>
</name>
<name>
<surname>Blankertz</surname> <given-names>B</given-names>
</name>
<etal/>
</person-group>. <article-title>On the interpretation of weight vectors of linear models in multivariate neuroimaging</article-title>. <source>Neuroimage</source>. (<year>2014</year>) <volume>87</volume>:<fpage>96</fpage>&#x2013;<lpage>110</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.neuroimage.2013.10.067</pub-id>
</citation>
</ref>
<ref id="B58">
<label>58</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wilming</surname> <given-names>R</given-names>
</name>
<name>
<surname>Budding</surname> <given-names>C</given-names>
</name>
<name>
<surname>M&#xfc;ller</surname> <given-names>KR</given-names>
</name>
<name>
<surname>Haufe</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>Scrutinizing XAI using linear ground-truth data with suppressor variables</article-title>. <source>Mach Learn</source>. (<year>2022</year>) <volume>111</volume>:<page-range>1903&#x2013;23</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s10994-022-06167-y</pub-id>
</citation>
</ref>
<ref id="B59">
<label>59</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Wilming</surname> <given-names>R</given-names>
</name>
<name>
<surname>Kieslich</surname> <given-names>L</given-names>
</name>
<name>
<surname>Clark</surname> <given-names>B</given-names>
</name>
<name>
<surname>Haufe</surname> <given-names>S</given-names>
</name>
</person-group>. (<year>2023</year>). <article-title>Theoretical behavior of XAI methods in the presence of suppressor variables</article-title>, in: <conf-name>Proceedings of the 40th International Conference on Machine Learning</conf-name>. <publisher-name>ML Research Press, PMLR</publisher-name> Vol. <volume>202</volume>. pp. <page-range>37091&#x2013;107</page-range>.</citation>
</ref>
<ref id="B60">
<label>60</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Belloni</surname> <given-names>A</given-names>
</name>
<name>
<surname>Chernozhukov</surname> <given-names>V</given-names>
</name>
<name>
<surname>Hansen</surname> <given-names>C</given-names>
</name>
</person-group>. <article-title>High-dimensional methods and inference on structural and treatment effects</article-title>. <source>J Econ Perspect</source>. (<year>2014</year>) <volume>28</volume>:<fpage>29</fpage>&#x2013;<lpage>50</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1257/jep.28.2.29</pub-id>
</citation>
</ref>
<ref id="B61">
<label>61</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Solomou</surname> <given-names>I</given-names>
</name>
<name>
<surname>Constantinidou</surname> <given-names>F</given-names>
</name>
</person-group>. <article-title>Prevalence and predictors of anxiety and depression symptoms during the COVID-19 pandemic and compliance with precautionary measures: age and sex matter</article-title>. <source>Int J Environ Res Public Health</source>. (<year>2020</year>) <volume>17</volume>:<elocation-id>4924</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/ijerph17144924</pub-id>
</citation>
</ref>
<ref id="B62">
<label>62</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname> <given-names>L</given-names>
</name>
<name>
<surname>Yuan</surname> <given-names>J</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>H</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>J</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Y</given-names>
</name>
</person-group>. <article-title>Influencing factors of depressive symptoms among undergraduates: a systematic review and meta-analysis</article-title>. <source>PloS One</source>. (<year>2023</year>) <volume>18</volume>:<elocation-id>e0279050</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pone.0279050</pub-id>
</citation>
</ref>
<ref id="B63">
<label>63</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Meehan</surname> <given-names>AJ</given-names>
</name>
<name>
<surname>Lewis</surname> <given-names>SJ</given-names>
</name>
<name>
<surname>Fazel</surname> <given-names>S</given-names>
</name>
<name>
<surname>Fusar-Poli</surname> <given-names>P</given-names>
</name>
<name>
<surname>Steyerberg</surname> <given-names>EW</given-names>
</name>
<name>
<surname>Stahl</surname> <given-names>D</given-names>
</name>
<etal/>
</person-group>. <article-title>Clinical prediction models in psychiatry: a systematic review of two decades of progress and challenges</article-title>. <source>Mol Psychiatry</source>. (<year>2022</year>) <volume>27</volume>:<page-range>2700&#x2013;8</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41380-022-01528-4</pub-id>
</citation>
</ref>
<ref id="B64">
<label>64</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bernardini</surname> <given-names>F</given-names>
</name>
<name>
<surname>Attademo</surname> <given-names>L</given-names>
</name>
<name>
<surname>Cleary</surname> <given-names>SD</given-names>
</name>
<name>
<surname>Luther</surname> <given-names>C</given-names>
</name>
<name>
<surname>Shim</surname> <given-names>RS</given-names>
</name>
<name>
<surname>Quartesan</surname> <given-names>R</given-names>
</name>
<etal/>
</person-group>. <article-title>Risk prediction models in psychiatry: toward a new frontier for the prevention of mental illnesses</article-title>. <source>J Clin Psychiatry</source>. (<year>2017</year>) <volume>78</volume>:<page-range>572&#x2013;83</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.4088/JCP.15r10003</pub-id>
</citation>
</ref>
<ref id="B65">
<label>65</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>XQ</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>YX</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>WJ</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>WJ</given-names>
</name>
</person-group>. <article-title>Influencing factors, prediction and prevention of depression in college students: a literature review</article-title>. <source>World J Psychiatry</source>. (<year>2022</year>) <volume>12</volume>:<page-range>860&#x2013;73</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.5498/wjp.v12.i7.860</pub-id>
</citation>
</ref>
<ref id="B66">
<label>66</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lawrie</surname> <given-names>SM</given-names>
</name>
<name>
<surname>Fletcher-Watson</surname> <given-names>S</given-names>
</name>
<name>
<surname>Whalley</surname> <given-names>HC</given-names>
</name>
<name>
<surname>McIntosh</surname> <given-names>AM</given-names>
</name>
</person-group>. <article-title>Predicting major mental illness: ethical and practical considerations</article-title>. <source>BJPsych Open</source>. (<year>2019</year>) <volume>5</volume>:<elocation-id>e30</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1192/bjo.2019.11</pub-id>
</citation>
</ref>
<ref id="B67">
<label>67</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Grote</surname> <given-names>T</given-names>
</name>
<name>
<surname>Berens</surname> <given-names>P</given-names>
</name>
</person-group>. <article-title>On the ethics of algorithmic decision-making in healthcare</article-title>. <source>J Med Ethics</source>. (<year>2020</year>) <volume>46</volume>:<page-range>205&#x2013;11</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1136/medethics-2019-105586</pub-id>
</citation>
</ref>
<ref id="B68">
<label>68</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fusar-Poli</surname> <given-names>P</given-names>
</name>
<name>
<surname>Manchia</surname> <given-names>M</given-names>
</name>
<name>
<surname>Koutsouleris</surname> <given-names>N</given-names>
</name>
<name>
<surname>Leslie</surname> <given-names>D</given-names>
</name>
<name>
<surname>Woopen</surname> <given-names>C</given-names>
</name>
<name>
<surname>Calkins</surname> <given-names>ME</given-names>
</name>
<etal/>
</person-group>. <article-title>Ethical considerations for precision psychiatry: A roadmap for research and clinical practice</article-title>. <source>Eur Neuropsychopharmacol</source>. (<year>2022</year>) <volume>63</volume>:<fpage>17</fpage>&#x2013;<lpage>34</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.euroneuro.2022.08.001</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>