<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Psychol.</journal-id>
<journal-title>Frontiers in Psychology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Psychol.</abbrev-journal-title>
<issn pub-type="epub">1664-1078</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpsyg.2024.1392240</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Psychology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Identifying the most crucial factors associated with depression based on interpretable machine learning: a case study from CHARLS</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author"><name><surname>Li</surname> <given-names>Rulin</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2787564/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
</contrib>
<contrib contrib-type="author"><name><surname>Wang</surname> <given-names>Xueyan</given-names></name><xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2791572/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes"><name><surname>Luo</surname> <given-names>Lanjun</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref><xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2046755/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author"><name><surname>Yuan</surname> <given-names>Youwei</given-names></name><xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2041691/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>School of Management, North Sichuan Medical College</institution>, <addr-line>Nanchong</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>Information Centre, Affiliated Hospital of North Sichuan Medical College</institution>, <addr-line>Nanchong</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>School of Management, Huazhong University of Science and Technology</institution>, <addr-line>Wuhan</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by" id="fn0001">
<p>Edited by: Taolin Chen, Sichuan University, China</p>
</fn>
<fn fn-type="edited-by" id="fn0002">
<p>Reviewed by: Julian Mutz, King&#x2019;s College London, United Kingdom</p>
<p>Yan Liu, Jining Medical University, China</p>
</fn>
<corresp id="c001">&#x002A;Correspondence: Lanjun Luo, <email>lanjun@nsmc.edu.cn</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>25</day>
<month>07</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>15</volume>
<elocation-id>1392240</elocation-id>
<history>
<date date-type="received">
<day>27</day>
<month>02</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>08</day>
<month>07</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2024 Li, Wang, Luo and Yuan.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Li, Wang, Luo and Yuan</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec id="sec1">
<title>Background</title>
<p>Depression is one of the most common mental illnesses among middle-aged and older adults in China. It is of great importance to find the crucial factors that lead to depression and to effectively control and reduce the risk of depression. Currently, there are limited methods available to accurately predict the risk of depression and identify the crucial factors that influence it.</p>
</sec>
<sec id="sec2">
<title>Methods</title>
<p>We collected data from 25,586 samples from the harmonized China Health and Retirement Longitudinal Study (CHARLS), and the latest records from 2018 were included in the current cross-sectional analysis. Ninety-three input variables in the survey were considered as potential influential features. Five machine learning (ML) models were utilized, including CatBoost and eXtreme Gradient Boosting (XGBoost), Gradient Boosting decision tree (GBDT), Random Forest (RF), Light Gradient Boosting Machine (LightGBM). The models were compared to the traditional multivariable Linear Regression (LR) model. Simultaneously, SHapley Additive exPlanations (SHAP) were used to identify key influencing factors at the global level and explain individual heterogeneity through instance-level analysis. To explore how different factors are non-linearly associated with the risk of depression, we employed the Accumulated Local Effects (ALE) approach to analyze the identified critical variables while controlling other covariates.</p>
</sec>
<sec id="sec3">
<title>Results</title>
<p>CatBoost outperformed other machine learning models in terms of MAE, MSE, MedAE, and R<sup>2</sup>metrics. The top three crucial factors identified by the SHAP were r4satlife, r4slfmem, and r4shlta, representing life satisfaction, self-reported memory, and health status levels, respectively.</p>
</sec>
<sec id="sec4">
<title>Conclusion</title>
<p>This study demonstrates that the CatBoost model is an appropriate choice for predicting depression among middle-aged and older adults in Harmonized CHARLS. The SHAP and ALE interpretable methods have identified crucial factors and the nonlinear relationship with depression, which require the attention of domain experts.</p>
</sec>
</abstract>
<kwd-group>
<kwd>depression</kwd>
<kwd>interpretable machine learning</kwd>
<kwd>middle-aged and older adults</kwd>
<kwd>crucial factors</kwd>
<kwd>CHARLS</kwd>
</kwd-group>
<counts>
<fig-count count="8"/>
<table-count count="2"/>
<equation-count count="4"/>
<ref-count count="55"/>
<page-count count="12"/>
<word-count count="7591"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Health Psychology</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec5">
<label>1</label>
<title>Introduction</title>
<p>Population aging is one of the most concerned public health issues. It is predicted that by 2050, China&#x2019;s elderly population will increase substantially, with more than 400 million people over 65 (<xref ref-type="bibr" rid="ref41">Zeng, 2012</xref>). Aging leads to a variety of negative health effects, which include an increased risk of depression (<xref ref-type="bibr" rid="ref32">Qiu et al., 2020</xref>). A report conducted by the World Health Organization showed that depression was ranked as the single largest contributor to non-fatal health loss, and 7% of old adults suffered from depression (<xref ref-type="bibr" rid="ref9001">World Health Organization, 2017</xref>). Moreover, depression can lead to a decline in physical function, lower quality of life, and increased health costs, which will seriously affect the physical and psychiatric health of middle-aged and older adults. In this context, predicting the risk of depression and analyzing the critical predictors of risk formation is crucial for disease control and improving quality of life.</p>
<p>At present, there have been many empirical methods focusing on factors related to depression in middle-aged and older adults (<xref ref-type="bibr" rid="ref40">Yunming et al., 2012</xref>). Typically, <xref ref-type="bibr" rid="ref26">Luo et al. (2023)</xref> used binary logistic regression to examine the correlation between dietary diversity, exercise, and depressive symptoms. They showed that a score on the Dietary Diversity Scale and qualified physical activity were protective factors against depressive symptoms among middle-aged women. <xref ref-type="bibr" rid="ref16">Jung et al. (2015)</xref> conducted multivariable logistic regression and calculated odds ratios to assess potential interactions between depression prevalence and the ages of menarche and menopause. The results found that the odds ratio of depression decreased with increasing age of menopause and duration of reproductive years. However, the above studies have mainly focused on the relationship between physiological aspects and depression.</p>
<p>In addition, some scholars have focused on the relationship between family factors and depression. <xref ref-type="bibr" rid="ref28">McMunn et al. (2021)</xref> examined the relationship between work-family and middle-aged psychological distress by multivariable logistic regression. They found that middle-aged people with weaker long-term employment relationships had poorer mental health and well-being. <xref ref-type="bibr" rid="ref11">Giannelis et al. (2021)</xref> used logistic regression to estimate the relationship between family status and depression; the results showed that a higher number of children and lack of cohabitation with a spouse or partner were associated with a greater likelihood of depression. Therefore, since depression has multiple and complex influences, this study comprehensively considered multiple factors such as demographic variables, health status, cognitive status, income, family structure, stress, and life satisfaction.</p>
<p>At the same time, several related studies also focused on the construction of risk prediction models for depression. For example, <xref ref-type="bibr" rid="ref48">Zhou et al. (2023)</xref> utilized the Kaplan&#x2013;Meier method and Cox proportional hazards regression model to evaluate the relationship between baseline chronic disease and depression. They found that suffering from different degrees of chronic diseases increased the risk of depression in middle-aged and older adults. <xref ref-type="bibr" rid="ref24">Luo et al. (2018)</xref> used the Cox proportional hazards model to estimate the relationship between obesity status and depression. They discovered a negative correlation between the relationship between body weight and depression. <xref ref-type="bibr" rid="ref19">Kimmel et al. (2000)</xref> used Cox proportional hazards regression analysis to predict the mortality hazard associated with depression in chronic hemodialysis outpatients. They found that medical deterioration factors may increase depression. <xref ref-type="bibr" rid="ref49">Zhou et al. (2021)</xref> used Cox proportional hazards regression models to examine the association between socioeconomic status and the incidence of depression. They found that participants with the highest level of household income had a 20% reduction in risk of depression. <xref ref-type="bibr" rid="ref38">Xiang and Wang (2021)</xref> used competing risks regression analysis to examine the relationship between childhood adversity and major depression in older adults and found that childhood adversity increased the risk of major depression in later life, especially for those who experienced physical abuse. In the above studies, the risk regression model was able to assess the relationship between depression and risk factors.</p>
<p>However, while parametric models can well illustrate the association between depression and risk factors, many risk-analysis studies have found that these models are still suboptimal when faced with complex relationships and high-dimensional inputs, and machine learning is relatively better (<xref ref-type="bibr" rid="ref13">Hao et al., 2019</xref>; <xref ref-type="bibr" rid="ref25">Luo and Qi, 2021</xref>). Recent studies have demonstrated that machine learning methods outperformed traditional linear models regarding nonlinear fitting performance (<xref ref-type="bibr" rid="ref43">Zhang et al., 2018</xref>). Machine learning is better equipped to identify the relationship between input and output when dealing with multiple inputs or explanatory variables, resulting in more outstanding performance. For example, <xref ref-type="bibr" rid="ref22">Lin et al. (2023)</xref> classified the trajectories of depressive symptoms via the latent class growth model and growth mixture model. Based on the identified trajectory patterns, three ML classification algorithms, i.e., gradient-enhanced decision tree, support vector machine, and random forest, demonstrated that machine learning could be very robust in predicting depressive symptom onset and developmental trajectory. However, machine learning is often criticized as the &#x2018;black box&#x2019; due to its numerous parameters and complex calculation process, which makes the decision-making process not transparent (<xref ref-type="bibr" rid="ref34">Tjoa and Guan, 2021</xref>).</p>
<p>Therefore, in this study, we propose an idea based on identifying the most critical factors affecting depression using interpretable machine learning. For this purpose, we adopted five machine learning prediction models: Random forest (RF) (<xref ref-type="bibr" rid="ref9006">Breiman, 2001</xref>; <xref ref-type="bibr" rid="ref10">Fawagreh et al., 2014</xref>), CatBoost (<xref ref-type="bibr" rid="ref9002">Prokhorenkova et al., 2018</xref>; <xref ref-type="bibr" rid="ref14">Ibrahim et al., 2020</xref>), eXtreme Gradient Boosting (XGBoost) (<xref ref-type="bibr" rid="ref7">Chen and Guestrin, 2016</xref>), Light Gradient Boosting Machine (LightGBM) (<xref ref-type="bibr" rid="ref18">Ke et al., 2017</xref>), and Gradient Boosting decision tree (GBDT) (<xref ref-type="bibr" rid="ref9003">Friedman, 2001</xref>; <xref ref-type="bibr" rid="ref44">Zhang and Jung, 2019</xref>) for depression risk prediction. First, we utilized demographics, cognitive, income, stress, family structure, health status, life satisfaction, and CESD-10 data collected from the China Health and Retirement Longitudinal Study (CHARLS). Then, continuous values of depression in middle-aged and older adults were predicted using demographic variables, health variables, cognitive, income, pressure, family structure, and life satisfaction from CHARLS data. Then, in the baseline experimental study, we randomly divided the dataset into 80% training dataset for model construction and 20% testing dataset for model testing and compared them with the traditional multiple linear regression (LR) model. Based on this, we used SHapley Additive exPlanations (SHAP) (<xref ref-type="bibr" rid="ref23">Lundberg and Lee, 2017</xref>) to make an in-depth interpretation of the input variables in the prediction model and clarify the importance of each feature and their net impact on different individual cases. We further adopted the Accumulated Local Effects (ALE) (<xref ref-type="bibr" rid="ref9007">Apley and Zhu, 2020</xref>) method to estimate the nonlinear association between crucial variables and predicted depression.</p>
<p>The contribution of this research can be summarized into the following three aspects: (1) We proposed a depression risk prediction method under multiple factors based on the CatBoost model which outperforms traditional parametric model; (2) We proposed interpretable methods based on SHAP and ALE, and explain the decision-making process of the CatBoost model, overcoming the problem of &#x2018;black box&#x2019;; (3) We identified the most critical risk factors of depression in a hypothesis-free manner based on the interpretable machine learning framework, demonstrating the most crucial position of life satisfaction.</p>
</sec>
<sec sec-type="materials|methods" id="sec6">
<label>2</label>
<title>Materials and methods</title>
<sec id="sec7">
<label>2.1</label>
<title>Dataset</title>
<p>The data and information used in this study are obtained from the Harmonized CHARLS dataset and Codebook. The development of the Harmonized CHARLS was funded by the National Institute on Aging, and detailed descriptions of the survey design and procedures were reported in the original study documentation (<xref ref-type="bibr" rid="ref46">Zhao et al., 2014</xref>).</p>
<p>As shown in <xref ref-type="fig" rid="fig1">Figure 1</xref>, in this study, all samples were selected from CHARLS fourth wave (<inline-formula>
<mml:math id="M1">
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>25586</mml:mn>
</mml:math>
</inline-formula>), and variables with missing values exceeding 30% have been excluded. Finally, ninety-four variables with the proportion of missing values within 30% were used in this study. For these variables, we adopted the KNearest Neighbours (KNN) algorithm (<xref ref-type="bibr" rid="ref35">Troyanskaya et al., 2001</xref>) to impute the missing values. Based on this, 93 explanatory variables are selected as inputs, and one variable is the prediction target.</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption>
<p>Flow diagram of sample selection.</p>
</caption>
<graphic xlink:href="fpsyg-15-1392240-g001.tif"/>
</fig>
</sec>
<sec id="sec8">
<label>2.2</label>
<title>Outcome variables and input variables</title>
<p>In this study, the outcome variable is <inline-formula>
<mml:math id="M2">
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mn>4</mml:mn>
<mml:mi mathvariant="normal">cesd</mml:mi>
<mml:mn>10</mml:mn>
</mml:math>
</inline-formula>, which records the continuous value of depression by using the 10-item Center for Epidemiology Studies Depression Scale (CESD-10), which has good reliability and validity among Chinese adults (<xref ref-type="bibr" rid="ref9">Chin et al., 2015</xref>) and also showed good internal consistency among older respondents (<xref ref-type="bibr" rid="ref15">Jiang et al., 2020</xref>). CESD-10 focuses on 10 questions about the experience in the past week: feeling depressed, feeling everything they did was effortful, feeling restless sleep, feeling happy, feeling lonely, feeling bothered, feeling they could not get &#x201C;going,&#x201D; feeling hopeful for the future, feeling fearful, having trouble in remember what was done. The total CESD-10 score ranged from 0 to 30, with higher scores indicating more severe depressive symptoms. In this study, the CESD-10 depression values of all the respondents were included. A respondent with depression scores of not less than 10 was considered to have depressive symptoms (<xref ref-type="bibr" rid="ref8">Chen et al., 2023</xref>).</p>
<p>According to the research needs, a total of 93 input variables were selected in the fourth wave of Harmonized CHARLS, including seven categories:</p>
<list list-type="order">
<list-item>
<p>Demographic: this category contains eight variables, including <inline-formula>
<mml:math id="M3">
<mml:mi mathvariant="normal">rabplace</mml:mi>
<mml:mo>:</mml:mo>
<mml:mi mathvariant="normal">C</mml:mi>
</mml:math>
</inline-formula> (birthplace), <inline-formula>
<mml:math id="M4">
<mml:mi mathvariant="normal">raeduc</mml:mi>
<mml:mo>:</mml:mo>
<mml:mi mathvariant="normal">C</mml:mi>
</mml:math>
</inline-formula>(education), r<inline-formula>
<mml:math id="M5">
<mml:mn>4</mml:mn>
<mml:mi mathvariant="normal">mnev</mml:mi>
</mml:math>
</inline-formula>(never married), <inline-formula>
<mml:math id="M6">
<mml:mi mathvariant="normal">ragender</mml:mi>
</mml:math>
</inline-formula>(gender), r4hukou, h4cpl (a couple household), h4rural (urban or rural), r4agey (age).</p>
</list-item>
<list-item>
<p>Health status: this category contains 48 variables, including: r4psyche (ever had any emotional, nervous, or psychiatric problems), <inline-formula>
<mml:math id="M7">
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mn>4</mml:mn>
<mml:mi mathvariant="normal">asthmae</mml:mi>
</mml:math>
</inline-formula> (ever had asthma), r4memrye (ever had memory-related disease), r4cancre (ever had cancer), r4kidneye (ever had a kidney disease), r4stroke (ever had stroke disease), r4livere (ever had liver disease), r4diabe (ever had diabetes disease), r4lung (ever had lunge disease), r4digeste (ever had a digestive disease), r4hearte (ever had heart disease), r4hibpe (ever had high blood pressure disease), r4dyslipe (ever had dyslipidemia disease), r4arthre (ever had arthritis disease), r4drinkev (ever drinks any alcohol), r4smokev (smoking), r4vgact_c (any vigorous physical activity), r4vgactx_c (the number of days of vigorous activity), r4mdact_c (any moderately physical activity), r4mdactx_c (the number of days of moderate activity), r4ltact_c (any light physical activity), r4ltactx_c (the number of days of light activity), r4mealsa (preparing meals), r4phonea (making phone calls), r4moneya (managing money), r4medsa (taking medications), r4shopa (shopping), r4housewka (cleaning house), r4lowermob (lower body mobility), r4walk1kms (walking 1KM), r4chaira (getting up from a chair), r4mobilsev (7 item mobility), r4dressa (dressing), r4uppermob (3-item summary of any difficulty with upper-body mobility activities), r4stoopa (stooping, kneeling or crouching), r4adlfive (5-item summary of any difficulty with activities of daily living), r4adlab_c (6-item summary), r4adla_c (4-item summary), <inline-formula>
<mml:math id="M8">
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mn>4</mml:mn>
<mml:mi mathvariant="normal">armsa</mml:mi>
</mml:math>
</inline-formula>(reaching arms above shoulder level), r<inline-formula>
<mml:math id="M9">
<mml:mn>4</mml:mn>
<mml:mi mathvariant="normal">dlmeas</mml:mi>
</mml:math>
</inline-formula>(picking up a coin from the table), r<inline-formula>
<mml:math id="M10">
<mml:mn>4</mml:mn>
<mml:mi mathvariant="normal">lifta</mml:mi>
</mml:math>
</inline-formula>(lifting or carrying weights over ten jins), r<inline-formula>
<mml:math id="M11">
<mml:mn>4</mml:mn>
<mml:mi mathvariant="normal">climsa</mml:mi>
</mml:math>
</inline-formula> (climbing several flights of stairs without resting),<inline-formula>
<mml:math id="M12">
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mn>4</mml:mn>
<mml:mi mathvariant="normal">toilta</mml:mi>
</mml:math>
</inline-formula>(using the toile), r4eata (eating), r4urina (controlling urination and defecation), r4batha (bathing and showering), r4beda (getting in and out of bed), r4shlta (self-reported health).</p>
</list-item>
<list-item>
<p>Family structure: this category contains 23 variables, including: h4coresd (any children with them), rameduc_c (mother&#x2019;s education level), rafeduc_c (father&#x2019;s education level), h4dchild (total number of deceased children), h4child (number of living children), r4dadliv (father alive), h4kcnt (contact with their children), h4lvnear (live near children), r4livpar (number of living parents), r4dadoccup_c (father&#x2019;s occupation), r4momliv (the respondent&#x2019;s mother is alive), h4fcamt (amount of transfers from children/grandchildren), h4tcamt (amount of transfers to children/grandchildren), h4fpamt (amount of transfers from parents/parents-in-law), h4tpamt (amount of transfers to parents/parents-in-law), h4foamt (the number of transfers from others), h4toamt (the number of transfers to others), h4frec (the total amount of transfers received), h4tgiv (the total amount of transfers given), h4ftot (net value of financial transfers), r4decsib (number of deceased siblings), r4livsib (number of living siblings), r4socwk (social activities).</p>
</list-item>
<list-item>
<p>Income: r4ipen (private pension).</p>
</list-item>
<list-item>
<p>Stress: this category contains 11 variables, including ramomdrug (female guardian had an alcohol and drug), rapadrug (guardians had alcoholism nor had a drug problem), ramwarm_c (female guardian warmth summary mean score), ramomgrela (good relationship with female guardian), ramomeft (female guardian put effort into watching over), ramomatt_c (received female guardian&#x2019;s love), radaddrug (male guardian had an alcohol and drug), radadgrela (good relationship with male guardian), rafinacom (self-rated family financial situation before age 17), Rahltcom (health condition compared to other children), r4chdeathe (experienced death of own child),</p>
</list-item>
<list-item>
<p>Cognitive: r4slfmem (self-reported memory).</p>
</list-item>
<list-item>
<p>Life Satisfaction: r4satlife.</p>
</list-item>
</list>
<p>The details, including meanings and descriptive analysis of the input and outcome variables, can be found in the <xref ref-type="supplementary-material" rid="SM1">Supplementary material</xref>.</p>
</sec>
<sec id="sec9">
<label>2.3</label>
<title>Predictive model development and evaluation</title>
<p>In this study, compared to the traditional multiple linear regression model, machine learning models of the XGBoost, GBDT, RF, CatBoost, and LightGBM were used to predict the risk of depression and generate six sets of data. Among them, the LR Model is a classic statistical algorithm. XGBoost, LightGBM, and CatBoost are the current most used tree-based algorithms, which can also be classified into the gradient-boosting decision tree algorithm series.</p>
<p>For the hyperparameter setting, Catboost was selected as the primary model for this study due to its excellent performance in healthy prediction-related studies (<xref ref-type="bibr" rid="ref42">Zhang et al., 2021</xref>), and Optuna (<xref ref-type="bibr" rid="ref1">Akiba et al., 2019</xref>) was used for optimal parameter search. In detail, for Catboost, the <inline-formula>
<mml:math id="M13">
<mml:mi mathvariant="normal">loss</mml:mi>
<mml:mo>:</mml:mo>
<mml:mi mathvariant="normal">function</mml:mi>
</mml:math>
</inline-formula> is set to <inline-formula>
<mml:math id="M14">
<mml:mi mathvariant="normal">M</mml:mi>
<mml:mi mathvariant="normal">A</mml:mi>
<mml:mi mathvariant="normal">E</mml:mi>
</mml:math>
</inline-formula>, the <inline-formula>
<mml:math id="M15">
<mml:mi mathvariant="normal">l</mml:mi>
<mml:mn>2</mml:mn>
<mml:mo>:</mml:mo>
<mml:mi mathvariant="normal">leaf</mml:mi>
<mml:mo>:</mml:mo>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">g</mml:mi>
</mml:math>
</inline-formula> is fixed to 0.0034, the <inline-formula>
<mml:math id="M16">
<mml:mi mathvariant="normal">learning</mml:mi>
<mml:mo>:</mml:mo>
<mml:mi mathvariant="normal">rate</mml:mi>
</mml:math>
</inline-formula> is 0.0155, the <inline-formula>
<mml:math id="M17">
<mml:mi mathvariant="normal">n</mml:mi>
<mml:mo>:</mml:mo>
<mml:mi mathvariant="normal">estimators</mml:mi>
</mml:math>
</inline-formula> is 22,148, the <inline-formula>
<mml:math id="M18">
<mml:mi mathvariant="normal">depth</mml:mi>
</mml:math>
</inline-formula> is 7, the <inline-formula>
<mml:math id="M19">
<mml:mo>min</mml:mo>
<mml:mo>:</mml:mo>
<mml:mi mathvariant="normal">data</mml:mi>
<mml:mo>:</mml:mo>
<mml:mi mathvariant="normal">in</mml:mi>
<mml:mo>:</mml:mo>
<mml:mi mathvariant="normal">leaf</mml:mi>
</mml:math>
</inline-formula> is 13. All other model parameters use the default settings from Python libraries: scikit-learn (version 1.2.2), xgboost (version 1.7.5), and lightgbm (version 4.3).</p>
<p>In the regression task, we used four common metrics to evaluate the models&#x2019; performances, including mean absolute error (MAE), mean square error (MSE), median absolute error (MedAE), and R-squared. For MAE, MSE, and MedAE, the smaller error, the better the model. In addition, R<sup>2</sup> measures how well the model explains the total variance of the outcome variable; the closer R<sup>2</sup> is to 1, the better the model&#x2019;s fit. The calculations of these four metrics are shown in <xref ref-type="disp-formula" rid="EQ1">Equations 1</xref><xref ref-type="disp-formula" rid="EQ2"/><xref ref-type="disp-formula" rid="EQ3">&#x2013;</xref><xref ref-type="disp-formula" rid="EQ4">4</xref>.</p>
<disp-formula id="EQ1">
<label>(1)</label>
<mml:math id="M20">
<mml:mi mathvariant="normal">M</mml:mi>
<mml:mi mathvariant="normal">A</mml:mi>
<mml:mi mathvariant="normal">E</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>n</mml:mi>
</mml:mfrac>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:mfenced close="|" open="|">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo stretchy="true">^</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
<disp-formula id="EQ2">
<label>(2)</label>
<mml:math id="M21">
<mml:mi mathvariant="normal">M</mml:mi>
<mml:mi mathvariant="normal">S</mml:mi>
<mml:mi mathvariant="normal">E</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>n</mml:mi>
</mml:mfrac>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo stretchy="true">^</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:math>
</disp-formula>
<disp-formula id="EQ3">
<label>(3)</label>
<mml:math id="M22">
<mml:mi mathvariant="normal">MedAE</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="normal">median</mml:mi>
<mml:mfenced open="(" close=")" separators=",,">
<mml:mfenced close="|" open="|">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo stretchy="true">^</mml:mo>
</mml:mover>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2026;</mml:mo>
<mml:mfenced close="|" open="|">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo stretchy="true">^</mml:mo>
</mml:mover>
<mml:mi>n</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mfenced>
</mml:math>
</disp-formula>
<disp-formula id="EQ4">
<label>(4)</label>
<mml:math id="M23">
<mml:msup>
<mml:mi mathvariant="normal">R</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo stretchy="true">^</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo stretchy="true">&#x00AF;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfrac>
</mml:math>
</disp-formula>
</sec>
<sec id="sec10">
<label>2.4</label>
<title>SHAP and ALE methods</title>
<p>Based on the comparison of model prediction effectiveness, the optimal depression risk prediction model can be selected and used as the basis for SHAP and ALE analysis. We used SHAP to find the most critical influencing factors at the entire dataset level and then explained the individual heterogeneity through local sample analysis. Further, we adopted ALE to manipulate selected variables while controlling other covariates to analyze how different factors are nonlinear and associated with depression risk.</p>
<p>The core idea behind SHAP is to assign a value to each feature for a specific prediction, indicating its contribution to the outcome. This approach is based on game theory&#x2019;s Shapley values, which aim to fairly distribute the gains or losses among players in a cooperative game. Based on this, SHAP treats the machine learning model as a game where each feature is a player. For each prediction, the Shapley value is calculated for each feature, and the overall explanation is obtained by summing up the values for all features. Benefiting from the ability to explain both in individual samples and cumulatively, SHAP can obtain both global and local interpretability. If the SHAP value is positive, as in the case of the regression continuous outcome prediction task, it indicates that the feature increases the prediction and vice versa. A larger absolute value of SHAP means the feature is more critical to the model.</p>
<p>The ALE method aims to analyze how features influence the model&#x2019;s prediction on average (<xref ref-type="bibr" rid="ref29">Molnar, 2020</xref>). Consider clarifying how a change in one feature affects the prediction when other input variables are held constant. ALE first divides the input feature&#x2019;s values into a grid of bins, and for each bin, ALE gets the model&#x2019;s predictions using the corresponding input value. Based on this, ALE can calculate the difference between the predicted outcome and the overall mean prediction across all bins, namely, the local effect. Through accumulating the regional effects, ALE finally gets the total pattern of each feature. More details on the calculation of SHAP and ALE can be found in Python libraries: Shap library (version 0.41.0) and Alibi (version 0.9.2) (<xref ref-type="bibr" rid="ref9005">Klaise et al., 2021</xref>).</p>
</sec>
</sec>
<sec sec-type="results" id="sec11">
<label>3</label>
<title>Results</title>
<sec id="sec12">
<label>3.1</label>
<title>Model evaluation and comparison</title>
<p><xref ref-type="table" rid="tab1">Table 1</xref> demonstrates the performance of each model under the four regression prediction assessment metrics on the test dataset. The performance of the optimal model under each evaluation metric in the table is bolded, and the suboptimal is underlined. It can be seen that the CatBoost model achieved the best performance with MAE (3.2198), MSE (19.6281), MedAE (2.457), and <italic>R</italic><sup>2</sup> (0.3855). The sub-optimal modelsare LightGBM and RF. The values of MAE, MSE, MedAE, and <italic>R</italic><sup>2</sup> for the LightGBM model, were 3.3809, 19.8245, 2.6759, and 0.3793. The values of MAE, MSE, MedAE, and <italic>R</italic><sup>2</sup> for RF model were, respectively, 3.3231, 20.2088, 2.616, 0.3673. According to the evaluation index of model performance, most machine learning models are better than the traditional multiple linear regression model.</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption>
<p>Model performance for predicting depression.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Model</th>
<th align="center" valign="top">MAE</th>
<th align="center" valign="top">MSE</th>
<th align="center" valign="top">MedAE</th>
<th align="center" valign="top">
<italic>R</italic>
<sup>2</sup>
</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">XGBoost</td>
<td align="center" valign="middle">3.4413</td>
<td align="center" valign="middle">21.3028</td>
<td align="center" valign="middle">2.6485</td>
<td align="center" valign="middle">0.333</td>
</tr>
<tr>
<td align="left" valign="middle">GBDT</td>
<td align="center" valign="middle">3.4533</td>
<td align="center" valign="middle">20.1564</td>
<td align="center" valign="middle">2.7687</td>
<td align="center" valign="middle">0.3689</td>
</tr>
<tr>
<td align="left" valign="middle">RF</td>
<td align="center" valign="middle">3.3231</td>
<td align="center" valign="middle">20.2088</td>
<td align="center" valign="middle">2.616</td>
<td align="center" valign="middle">0.3673</td>
</tr>
<tr>
<td align="left" valign="middle">LightGBM</td>
<td align="center" valign="middle">3.3809</td>
<td align="center" valign="middle">19.8245</td>
<td align="center" valign="middle">2.6759</td>
<td align="center" valign="middle">0.3793</td>
</tr>
<tr>
<td align="left" valign="middle">CatBoost</td>
<td align="center" valign="middle">
<bold>3.2198</bold>
</td>
<td align="center" valign="middle">
<bold>19.6281</bold>
</td>
<td align="center" valign="middle">
<bold>2.457</bold>
</td>
<td align="center" valign="middle">
<bold>0.3855</bold>
</td>
</tr>
<tr>
<td align="left" valign="middle">LR</td>
<td align="center" valign="middle">3.5455</td>
<td align="center" valign="middle">20.8503</td>
<td align="center" valign="middle">2.8527</td>
<td align="center" valign="middle">0.3472</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Bold value means optimal performance in the corresponding metric.</p>
</table-wrap-foot>
</table-wrap>
<p>It is worth noting that, in the baseline experimental study, the outcome variable was also performed KNN-based missing value filling to retain more samples. In the Sect.3.4 robustness test, we further excluded samples with null values of the outcome variable to verify the reliability of the study.</p>
</sec>
<sec id="sec13">
<label>3.2</label>
<title>SHAP analysis results</title>
<p>We used SHAP to find the critical influencing factors at the whole dataset level and determine the importance of each variable. As shown in <xref ref-type="table" rid="tab2">Table 2</xref>, the top three significant impact characteristics are r4satlife, r4slfmem, and r4shlta. These three crucial factors were essential features of the depression prediction model, and the SHAP values were, respectively, 1.0871, 0.7614, and 0.6572. Based on the SHAP feature importance bar plot of the CatBoost model, we can more intuitively understand the variable dimensions that play an important role. As shown in <xref ref-type="fig" rid="fig2">Figure 2</xref>, the influence values of the top three crucial factors, such as r4satlife, r4slfmem, and r4shlta, were all above 0.6.</p>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption>
<p>Order of key influencing factors.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Variable</th>
<th align="center" valign="top">Feature importance value</th>
<th align="left" valign="top">Meanings</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="bottom">r4satlife</td>
<td align="center" valign="bottom">1.0871</td>
<td align="left" valign="bottom">Satisfied with life</td>
</tr>
<tr>
<td align="left" valign="bottom">r4slfmem</td>
<td align="center" valign="bottom">0.7614</td>
<td align="left" valign="bottom">Self-reported memory</td>
</tr>
<tr>
<td align="left" valign="bottom">r4shlta</td>
<td align="center" valign="bottom">0.6572</td>
<td align="left" valign="bottom">Self-reported health</td>
</tr>
<tr>
<td align="left" valign="bottom">r4mobilsev</td>
<td align="center" valign="bottom">0.3636</td>
<td align="left" valign="bottom">7 item mobility</td>
</tr>
<tr>
<td align="left" valign="bottom">ragender</td>
<td align="center" valign="bottom">0.3395</td>
<td align="left" valign="bottom">Gender</td>
</tr>
<tr>
<td align="left" valign="bottom">r4digeste</td>
<td align="center" valign="bottom">0.2665</td>
<td align="left" valign="bottom">Ever had stomach/digestive disease</td>
</tr>
<tr>
<td align="left" valign="bottom">r4agey</td>
<td align="center" valign="bottom">0.2600</td>
<td align="left" valign="bottom">Age in years</td>
</tr>
<tr>
<td align="left" valign="bottom">r4ipen</td>
<td align="center" valign="bottom">0.2152</td>
<td align="left" valign="bottom">Income: total pension income</td>
</tr>
<tr>
<td align="left" valign="bottom">h4rural</td>
<td align="center" valign="bottom">0.2007</td>
<td align="left" valign="bottom">Lives in rural or urban</td>
</tr>
<tr>
<td align="left" valign="bottom">r4lowermob</td>
<td align="center" valign="bottom">0.1974</td>
<td align="left" valign="bottom">Lower body mobility</td>
</tr>
<tr>
<td align="left" valign="bottom">raeduc_c</td>
<td align="center" valign="bottom">0.1809</td>
<td align="left" valign="bottom">Education</td>
</tr>
<tr>
<td align="left" valign="bottom">rafinacom</td>
<td align="center" valign="bottom">0.1728</td>
<td align="left" valign="bottom">Financial status compared to the average family in the same</td>
</tr>
<tr>
<td align="left" valign="bottom">r4arthre</td>
<td align="center" valign="bottom">0.1329</td>
<td align="left" valign="bottom">Ever had arthritis disease</td>
</tr>
<tr>
<td align="left" valign="bottom">r4housewka</td>
<td align="center" valign="bottom">0.1254</td>
<td align="left" valign="bottom">Some Diff-cleaning house</td>
</tr>
<tr>
<td align="left" valign="bottom">r4moneys</td>
<td align="center" valign="bottom">0.1221</td>
<td align="left" valign="bottom">Managing money</td>
</tr>
<tr>
<td align="left" valign="bottom">r4vgactx_c</td>
<td align="center" valign="bottom">0.1205</td>
<td align="left" valign="bottom">Days/wk. vigorous physical activity or exercise</td>
</tr>
<tr>
<td align="left" valign="bottom">rafeduc_c</td>
<td align="center" valign="bottom">0.1123</td>
<td align="left" valign="bottom">Father&#x2019;s education</td>
</tr>
<tr>
<td align="left" valign="bottom">r4adlab_c</td>
<td align="center" valign="bottom">0.1115</td>
<td align="left" valign="bottom">6-item summary</td>
</tr>
<tr>
<td align="left" valign="bottom">h4foamt</td>
<td align="center" valign="bottom">0.0968</td>
<td align="left" valign="bottom">The number of transfers from others</td>
</tr>
<tr>
<td align="left" valign="bottom">r4stoopa</td>
<td align="center" valign="bottom">0.0891</td>
<td align="left" valign="bottom">Stooping, kneeling, or crouching</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig position="float" id="fig2">
<label>Figure 2</label>
<caption>
<p>Bar plot of feature importance based on CatBoost-SHAP.</p>
</caption>
<graphic xlink:href="fpsyg-15-1392240-g002.tif"/>
</fig>
<p>SHAP beeswarm plot provides the degree of importance of the variable and the positive and negative impact on the CatBoost model&#x2019;s prediction results. In <xref ref-type="fig" rid="fig3">Figure 3</xref>, each row represents a variable, and the abscissa is the SHAP value, showing the importance ranking of each feature SHAP value from top to bottom. Each point represents a sample. The redder the color, the larger the feature itself, and the bluer the color, the smaller the feature itself. When the SHAP value is positive, it indicates increasing the model&#x2019;s prediction results; in contrast, when the SHAP value is negative, the negative impact is on the model&#x2019;s output. According to the three crucial factors in <xref ref-type="fig" rid="fig3">Figure 3</xref>, r4satlife had the most significant effect on outcomes, indicating that greater satisfaction values decreased the probability of depression and that satisfaction and depression were negatively correlated. The scores of r4slfmem and r4shlta ranged from 1 for excellent to 5 for very poor, meaning that higher respondent scores indicate worse self-rated memory and health. The results from <xref ref-type="fig" rid="fig3">Figure 3</xref> showed that higher scores of self-reported memory and self-reported health increased the risk of depression. Thus, self-reported memory and health were positively associated with depression.</p>
<fig position="float" id="fig3">
<label>Figure 3</label>
<caption>
<p>Beeswarm plot of effects of features based on CatBoost-SHAP.</p>
</caption>
<graphic xlink:href="fpsyg-15-1392240-g003.tif"/>
</fig>
<p>In this study, two samples were randomly selected to perform SHAP analysis to analyze the instance-level influencing factors and discover individual heterogeneity. As shown in <xref ref-type="fig" rid="fig4">Figures 4</xref>, <xref ref-type="fig" rid="fig5">5</xref>, the vertical axis represents different feature values; the horizontal axis represents SHAP values, and E (f(x)) represents the expectation of the predicted value of all samples. The red section indicates a positive effect on predicted depressive values, and the blue section suggests a negative impact. <xref ref-type="fig" rid="fig4">Figure 4</xref> demonstrates that the expected value of a sample f(x) is 14.592, much higher than the E (f(x)), indicating that the individual&#x2019;s depression is higher than the average level. According to the SHAP interpretation, the reasons why this sample was predicted as a higher level of depression were mainly influenced by variables such as r4satilfe, r4shlta, r4moblisev, r4toilta, and r4shopa. In <xref ref-type="fig" rid="fig5">Figure 5</xref>, the predicted value of sample f(x) is 5.136, indicating that the individual is lower than the average depression level. This lower level of depression is influenced by variables such as r4shlta, r4slfmem, r4ipen, r4agey, r4mobilsev, and r4rural. Therefore, under the SHAP individual case analysis, the crucial influencing factors of different samples are different, which reflects the individual heterogeneity well.</p>
<fig position="float" id="fig4">
<label>Figure 4</label>
<caption>
<p>Analysis of SHAP values for single-sample full features.</p>
</caption>
<graphic xlink:href="fpsyg-15-1392240-g004.tif"/>
</fig>
<fig position="float" id="fig5">
<label>Figure 5</label>
<caption>
<p>SHAP values analysis of single-sample full features.</p>
</caption>
<graphic xlink:href="fpsyg-15-1392240-g005.tif"/>
</fig>
</sec>
<sec id="sec14">
<label>3.3</label>
<title>ALE analysis results</title>
<p>We plotted the ALE of the depression prediction model based on the r4satlife, r4slfmem, and r4shlta. <xref ref-type="fig" rid="fig6">Figure 6</xref> shows that r4satlife has a strong negative impact on the prediction of depression, and the prediction value will decrease with the increase in life satisfaction. Since the scores on r4slfmem from 1 for excellent to 5 for very poor, <xref ref-type="fig" rid="fig7">Figure 7</xref> shows that higher scores of respondents&#x2019; self-reported memory had a strong positive effect on depression prediction, and the prediction value will increase with the higher self-reported memory. That is, the worse the respondents&#x2019; self-reported memory, the easier it is to predict the risk of depression. However, when the self-reported memory was greater than 4.7, the impact of the higher memory value on the increased predictive value of depression was weakened. Since the scores on r4shlta from 1 for excellent to 5 for very poor, <xref ref-type="fig" rid="fig8">Figure 8</xref> shows that higher scores of respondents&#x2019; self-reported health status of the individuals also had a strong positive impact on the prediction of depression, and the predicted value overall increased with the higher self-reported health value. That is, the worse the respondents&#x2019; self-reported health, the easier it is to predict the risk of depression.</p>
<fig position="float" id="fig6">
<label>Figure 6</label>
<caption>
<p>ALE plot of r4satlife effect on depression.</p>
</caption>
<graphic xlink:href="fpsyg-15-1392240-g006.tif"/>
</fig>
<fig position="float" id="fig7">
<label>Figure 7</label>
<caption>
<p>ALE plot of r4slfmem effects on depression.</p>
</caption>
<graphic xlink:href="fpsyg-15-1392240-g007.tif"/>
</fig>
<fig position="float" id="fig8">
<label>Figure 8</label>
<caption>
<p>ALE plot of r4shlta effects on depression.</p>
</caption>
<graphic xlink:href="fpsyg-15-1392240-g008.tif"/>
</fig>
</sec>
<sec id="sec15">
<label>3.4</label>
<title>Robustness test</title>
<p>To ensure the reliability of the findings of this study, we also conducted the robustness test. There are three main parts of the test. First, the samples with missing values of the dependent variable are excluded. In the baseline study, we used KNN to interpolate the missing values of the dependent variable to retain more samples. In the robustness test, we exclude these samples with missing values for the depression outcome variable and look at the performance of each machine learning model. The results are shown in <xref ref-type="supplementary-material" rid="SM1">Supplementary Table S2</xref>, where it can be seen that due to the reduced sample size, there are relatively fewer patterns of data to learn, and the error of each model has increased. However, Catboost is still the comparatively better model. The three most important variables remain <inline-formula>
<mml:math id="M24">
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mn>4</mml:mn>
<mml:mi mathvariant="normal">satlife</mml:mi>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math id="M25">
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mn>4</mml:mn>
<mml:mi mathvariant="normal">shlta</mml:mi>
</mml:math>
</inline-formula>, and <inline-formula>
<mml:math id="M26">
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mn>4</mml:mn>
<mml:mi mathvariant="normal">slfmem</mml:mi>
</mml:math>
</inline-formula>.</p>
<p>Second, due to the 93 input variables in the benchmark experiment, there is a more severe problem of multicollinearity. To deal with this concern, this study adopts the Recursive Feature Elimination (RFE) method (<xref ref-type="bibr" rid="ref12">Guyon et al., 2002</xref>), which sequentially removes the input features with the highest Variance Inflation Factor (VIF) values until the VIFs of all the features are less than 5. As shown in <xref ref-type="supplementary-material" rid="SM1">Supplementary Figure S2</xref>, during this process, the performance of the CatBoost model retraining and validation are recorded. It can be found that the multicollinearity problem is no longer serious after the input features are reduced to 58. CatBoost&#x2019;s performance decreases throughout the RFE process but still performs relatively best with the original 93 inputs. This indicates that the multicollinearity problem is not severe for CatBoost and is consistent with the characteristics of machine learning methods such as tree-based models (<xref ref-type="bibr" rid="ref45">Zhao et al., 2019</xref>; <xref ref-type="bibr" rid="ref5">Chan et al., 2022</xref>).</p>
<p>Third, due to the stochastic nature of machine learning, this study uses a 5-fold cross-validation (CV) approach to compare the comprehensive performance of each model in the baseline experiment. The results record the mean and standard deviation of the performance of all models in cross-validation, and it can be seen that the CatBoost model is still robust and has about 14% improvement over the LR model in the critical <italic>R</italic><sup>2</sup> measurement. The detailed results of the robustness test can be found in the <xref ref-type="supplementary-material" rid="SM1">Supplementary material</xref>.</p>
</sec>
</sec>
<sec sec-type="discussion" id="sec16">
<label>4</label>
<title>Discussion</title>
<p>In this study, based on data from the fourth wave of Harmonized CHARLS, we obtained demographic variables, health status, cognitive, income, family structure, stress, life satisfaction, and CESD in a total of 25,586 middle-aged and older people sample. Five machine learning and traditional models showed that the former was the best in predicting the risk of depression, consistent with earlier related studies (<xref ref-type="bibr" rid="ref2">Ay et al., 2019</xref>). Moreover, further comparison found that the CatBoost model more accurately predicted the risk of depression in older adults. Therefore, based on the CatBoost model, the SHAP method was used to analyze the crucial factors to obtain the global interpretation of the whole sample set and the feature importance ranking of the model. There were three significant impact characteristics: life satisfaction, self-reported memory, and self-reported health. Similarly, previous studies also found that there is considerable heterogeneity in depressive symptoms in older adults, but cognition and self-reported memory are considered to predict the essential characteristics of &#x2018;depressive symptoms events increased trajectory&#x2019; and &#x2018;chronic symptoms trajectory&#x2019; (<xref ref-type="bibr" rid="ref22">Lin et al., 2023</xref>), self-reported memory presented was also a vital influence feature obtained in this study.</p>
<p>The SHAP method analysis found that life satisfaction was one of the most critical key impact features that predicted depression levels. Prior studies have also shown that the higher the level of depression in middle-aged and older adults, the lower the life satisfaction, which is the strongest negative predictor of depression in older adults (<xref ref-type="bibr" rid="ref39">Yoo et al., 2016</xref>). Lee&#x2019;s study of older Koreans found that life satisfaction increased significantly over time, and the level of depression decreased (<xref ref-type="bibr" rid="ref21">Lee et al., 2020</xref>). These are consistent with the findings presented in the present study. We considered the following reasons: on the one hand, life satisfaction is an evaluation indicator of subjective well-being and is an important part of improving the quality of life of older adults (<xref ref-type="bibr" rid="ref4">Chachamovich et al., 2007</xref>). A study showed that subjective well-being and depression were negatively correlated with each other (<xref ref-type="bibr" rid="ref3">Bartels et al., 2013</xref>). On the other hand, life satisfaction was one of the important characteristics of individuals with chronic and highly stable depression trajectories, indicating that life satisfaction and the incidence of depression are highly related (<xref ref-type="bibr" rid="ref17">Kaup et al., 2016</xref>).</p>
<p>In this study, self-reported memory is confirmed to be a more critical influencing factor. Sun&#x2019;s study found that memory was significantly negatively associated with depression (<xref ref-type="bibr" rid="ref33">Sun et al., 2019</xref>). Zhou&#x2019;s study also found that participants with depressive symptoms all had poor cognitive function (<xref ref-type="bibr" rid="ref47">Zhou et al., 2021</xref>). We considered the following reasons: On the one hand, the deterioration of mental function is related to depression. Episodic memory deteriorated in answering structured questions (but not free recollection of past events) among depressed patients (<xref ref-type="bibr" rid="ref36">Wachowska et al., 2022</xref>).</p>
<p>On the other hand, cognitive impairment appears to be the core pathological symptom of depression, and cognitive impairment may appear before depression parameters (<xref ref-type="bibr" rid="ref27">Maramis et al., 2021</xref>). Some argue that cognitive symptoms should be viewed as a separate dimension and an important target for any treatment that has already started (<xref ref-type="bibr" rid="ref31">Planchez et al., 2019</xref>). Therefore, in this study, self-reported memory is considered to be a more important influencing factor in predicting depression.</p>
<p>Self-reported health is also confirmed to be a more important influencing factor. Kuchibhatla&#x2019;s study found that the reduction in depressive symptoms is strongly associated with health (<xref ref-type="bibr" rid="ref20">Kuchibhatla et al., 2012</xref>). Kaup&#x2019;s study found that good education, better health, fewer stressful events, and a more extensive social network would reduce the incidence of depressive symptoms (<xref ref-type="bibr" rid="ref17">Kaup et al., 2016</xref>). Chan&#x2019;s study found that a history of self-perceived health and perceived low cost in older men are important risk factors for depression (<xref ref-type="bibr" rid="ref6">Chan and Zeng, 2011</xref>). Wen&#x2019;s study found that individuals who perceived their health status as excellent had a 62% lower risk of depression compared with those who perceived their health as poor (<xref ref-type="bibr" rid="ref37">Wen et al., 2019</xref>). The reason is that self-rated health was associated with both objective health status and components of psychological perceptions. It has been found to be associated with depression factors that share the same psychological and biological mechanisms (<xref ref-type="bibr" rid="ref30">Peleg and Nudelman, 2021</xref>). Therefore, in this study, self-reported health is considered to be a more important influencing factor in predicting depression.</p>
<p>Based on the entire interpretable machine learning framework, this study identified the most critical factors affecting depression levels in a hypothesis-free manner, excluding possible interference from confounding factors, demonstrating the most crucial position of life satisfaction. Meanwhile, we randomly selected two respondents and explained the key factors affecting individual depression levels, namely, finding individual heterogeneity. In further research, using the ALE method, we found the depression prediction model diagram of some essential factors, so we can also change the level of depression and improve the quality of life by controlling for some related variables.</p>
<p>This study has three innovations but also has three limitations. First, we did not use the longitudinal data, only the fourth wave of the Harmonized CHARLS data. Second, the study did not include variables such as finance, housing, health care, and insurance. Third, we did not adopt more complex parameter adjustment methods, most of which used the default configuration, which has some limitations on the applicability of the challenging depression prediction task. In future research, we need to develop more advanced integration models.</p>
</sec>
<sec sec-type="data-availability" id="sec17">
<title>Data availability statement</title>
<p>This analysis uses data or information from the Harmonized CHARLS dataset and Codebook, Version D as of June 2021 developed by the Gateway to Global Aging Data. The development of the Harmonized CHARLS was funded by the National Institute on Aging (R01 AG030153, RC2 AG036619, R03 AG043052). For more information, please refer to <ext-link xlink:href="https://g2aging.org/" ext-link-type="uri">https://g2aging.org/</ext-link>. The CHARLS data can be accessed at: <ext-link xlink:href="https://charls.charlsdata.com/pages/Data/harmonized_charls/en.html" ext-link-type="uri">https://charls.charlsdata.com/pages/Data/harmonized_charls/en.html</ext-link>. For more details and code for this study, please contact the corresponding author.</p>
</sec>
<sec sec-type="author-contributions" id="sec18">
<title>Author contributions</title>
<p>RL: Conceptualized, Methodology, Writing &#x2013; original draft. XW: Conceptualized, Methodology, Writing &#x2013; review &#x0026; editing. LL: Conceptualized, Methodology, Writing &#x2013; review &#x0026; editing. YY: Conceptualized, Writing &#x2013; review &#x0026; editing.</p>
</sec>
</body>
<back>
<sec sec-type="funding-information" id="sec19">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. This research is supported by the Humanities and Social Science Fund of Ministry of Education of China (No. 23YJC630130), and Sichuan Science and Technology Program (No. 2023NSFSC1015).</p>
</sec>
<sec sec-type="COI-statement" id="sec20">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="sec21">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="sec22">
<title>Supplementary material</title>
<p>The Supplementary material for this article can be found online at: <ext-link xlink:href="https://www.frontiersin.org/articles/10.3389/fpsyg.2024.1392240/full#supplementary-material" ext-link-type="uri">https://www.frontiersin.org/articles/10.3389/fpsyg.2024.1392240/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Table_1.docx" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="ref1">
<citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Akiba</surname> <given-names>T.</given-names></name> <name><surname>Sano</surname> <given-names>S.</given-names></name> <name><surname>Yanase</surname> <given-names>T.</given-names></name> <name><surname>Ohta</surname> <given-names>T.</given-names></name> <name><surname>Koyama</surname> <given-names>M.</given-names></name></person-group> (<year>2019</year>). <article-title>Optuna: a next-generation Hyperparameter optimization framework</article-title>. <conf-name>Proceedings of the 25th ACM SIGKDD international conference on Knowledge Discovery &#x0026; Data Mining</conf-name>, <fpage>2623</fpage>&#x2013;<lpage>2631</lpage>.</citation>
</ref>
<ref id="ref9007">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Apley</surname> <given-names>D. W.</given-names></name> <name><surname>Zhu</surname> <given-names>J.</given-names></name></person-group> (<year>2020</year>). <article-title>Visualizing the effects of predictor variables in black box supervised learning models</article-title>. <source>J R Stat Soc Series B Stat Methodol.</source> <volume>82</volume>, <fpage>1059</fpage>&#x2013;<lpage>1086</lpage>.</citation>
</ref>
<ref id="ref2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ay</surname> <given-names>B.</given-names></name> <name><surname>Yildirim</surname> <given-names>O.</given-names></name> <name><surname>Talo</surname> <given-names>M.</given-names></name> <name><surname>Baloglu</surname> <given-names>U. B.</given-names></name> <name><surname>Aydin</surname> <given-names>G.</given-names></name> <name><surname>Puthankattil</surname> <given-names>S. D.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Automated depression detection using deep representation and sequence learning with EEG signals</article-title>. <source>J. Med. Syst.</source> <volume>43</volume>:<fpage>205</fpage>. doi: <pub-id pub-id-type="doi">10.1007/s10916-019-1345-y</pub-id>, PMID: <pub-id pub-id-type="pmid">31139932</pub-id></citation>
</ref>
<ref id="ref3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bartels</surname> <given-names>M.</given-names></name> <name><surname>Cacioppo</surname> <given-names>J. T.</given-names></name> <name><surname>Van Beijsterveldt</surname> <given-names>T. C. E. M.</given-names></name> <name><surname>Boomsma</surname> <given-names>D. I.</given-names></name></person-group> (<year>2013</year>). <article-title>Exploring the association between well-being and psychopathology in adolescents</article-title>. <source>Behav. Genet.</source> <volume>43</volume>, <fpage>177</fpage>&#x2013;<lpage>190</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s10519-013-9589-7</pub-id>, PMID: <pub-id pub-id-type="pmid">23471543</pub-id></citation>
</ref>
<ref id="ref9006">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Breiman</surname> <given-names>L.</given-names></name>
</person-group> (<year>2001</year>). <article-title>Random forests</article-title>. <source>Mach. Learn.</source> <volume>45</volume>, <fpage>5</fpage>&#x2013;<lpage>32</lpage>.</citation>
</ref>
<ref id="ref4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chachamovich</surname> <given-names>E.</given-names></name> <name><surname>Trentini</surname> <given-names>C.</given-names></name> <name><surname>Fleck</surname> <given-names>M. P.</given-names></name></person-group> (<year>2007</year>). <article-title>Assessment of the psychometric performance of the WHOQOL-BREF instrument in a sample of Brazilian older adults</article-title>. <source>Int. Psychogeriatr.</source> <volume>19</volume>, <fpage>635</fpage>&#x2013;<lpage>646</lpage>. doi: <pub-id pub-id-type="doi">10.1017/S1041610206003619</pub-id>, PMID: <pub-id pub-id-type="pmid">16870036</pub-id></citation>
</ref>
<ref id="ref5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chan</surname> <given-names>J. Y.-L.</given-names></name> <name><surname>Leow</surname> <given-names>S. M. H.</given-names></name> <name><surname>Bea</surname> <given-names>K. T.</given-names></name> <name><surname>Cheng</surname> <given-names>W. K.</given-names></name> <name><surname>Phoong</surname> <given-names>S. W.</given-names></name> <name><surname>Hong</surname> <given-names>Z.-W.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Mitigating the multicollinearity problem and its machine learning approach: a review</article-title>. <source>Mathematics</source> <volume>10</volume>:<fpage>1283</fpage>. doi: <pub-id pub-id-type="doi">10.3390/math10081283</pub-id></citation>
</ref>
<ref id="ref6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chan</surname> <given-names>M. F.</given-names></name> <name><surname>Zeng</surname> <given-names>W.</given-names></name></person-group> (<year>2011</year>). <article-title>Exploring risk factors for depression among older men residing in Macau</article-title>. <source>J. Clin. Nurs.</source> <volume>20</volume>, <fpage>2645</fpage>&#x2013;<lpage>2654</lpage>. doi: <pub-id pub-id-type="doi">10.1111/j.1365-2702.2010.03689.x</pub-id>, PMID: <pub-id pub-id-type="pmid">21627698</pub-id></citation>
</ref>
<ref id="ref7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>T.</given-names></name> <name><surname>Guestrin</surname> <given-names>C</given-names></name></person-group>. (<year>2016</year>). <article-title>XGBoost: A scalable tree boosting system</article-title>. <source>Proceedings of the 22nd ACM SIGKDD international conference on knowledge discovery and data mining</source>, <fpage>785</fpage>&#x2013;<lpage>794</lpage>. doi: <pub-id pub-id-type="doi">10.1145/2939672.2939785</pub-id></citation>
</ref>
<ref id="ref8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>X.</given-names></name> <name><surname>He</surname> <given-names>L.</given-names></name> <name><surname>Shi</surname> <given-names>K.</given-names></name> <name><surname>Wu</surname> <given-names>Y.</given-names></name> <name><surname>Lin</surname> <given-names>S.</given-names></name> <name><surname>Fang</surname> <given-names>Y.</given-names></name></person-group> (<year>2023</year>). <article-title>Interpretable machine learning for fall prediction among older adults in China</article-title>. <source>Am. J. Prev. Med.</source> <volume>65</volume>, <fpage>579</fpage>&#x2013;<lpage>586</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.amepre.2023.04.006</pub-id>, PMID: <pub-id pub-id-type="pmid">37087076</pub-id></citation>
</ref>
<ref id="ref9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chin</surname> <given-names>W. Y.</given-names></name> <name><surname>Choi</surname> <given-names>E. P. H.</given-names></name> <name><surname>Chan</surname> <given-names>K. T. Y.</given-names></name> <name><surname>Wong</surname> <given-names>C. K. H.</given-names></name></person-group> (<year>2015</year>). <article-title>The psychometric properties of the Center for Epidemiologic Studies Depression Scale in Chinese primary care patients: factor structure, construct validity, reliability, Sensitivity and Responsiveness</article-title>. <source>Plos One</source> <volume>10</volume>:<fpage>e0135131</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pone.0135131</pub-id>, PMID: <pub-id pub-id-type="pmid">26252739</pub-id></citation>
</ref>
<ref id="ref10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fawagreh</surname> <given-names>K.</given-names></name> <name><surname>Gaber</surname> <given-names>M. M.</given-names></name> <name><surname>Elyan</surname> <given-names>E.</given-names></name></person-group> (<year>2014</year>). <article-title>Random forests: from early developments to recent advancements</article-title>. <source>Syst. Sci. Cont. Eng.</source> <volume>2</volume>, <fpage>602</fpage>&#x2013;<lpage>609</lpage>. doi: <pub-id pub-id-type="doi">10.1080/21642583.2014.956265</pub-id></citation>
</ref>
<ref id="ref9003">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Friedman</surname> <given-names>J. H.</given-names></name>
</person-group> (<year>2001</year>). <article-title>Greedy function approximation: A gradient boosting machine</article-title>. <source>Ann. Stat.</source> <fpage>1189</fpage>&#x2013;<lpage>1232</lpage>.</citation>
</ref>
<ref id="ref11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Giannelis</surname> <given-names>A.</given-names></name> <name><surname>Palmos</surname> <given-names>A.</given-names></name> <name><surname>Hagenaars</surname> <given-names>S. P.</given-names></name> <name><surname>Breen</surname> <given-names>G.</given-names></name> <name><surname>Lewis</surname> <given-names>C. M.</given-names></name> <name><surname>Mutz</surname> <given-names>J.</given-names></name></person-group> (<year>2021</year>). <article-title>Examining the association between family status and depression in the UK biobank</article-title>. <source>J. Affect. Disord.</source> <volume>279</volume>, <fpage>585</fpage>&#x2013;<lpage>598</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jad.2020.10.017</pub-id>, PMID: <pub-id pub-id-type="pmid">33189065</pub-id></citation>
</ref>
<ref id="ref12">
<citation citation-type="other"><person-group person-group-type="author"><name><surname>Guyon</surname> <given-names>I.</given-names></name> <name><surname>Weston</surname> <given-names>J.</given-names></name> <name><surname>Barnhill</surname> <given-names>S.</given-names></name> <name><surname>Vapnik</surname> <given-names>V.</given-names></name></person-group> (<year>2002</year>). <article-title>Gene selection for Cancer classification using support vector machines</article-title>. <source>Mach. Learn.</source> <volume>46</volume>, <fpage>389</fpage>&#x2013;<lpage>422</lpage>.</citation>
</ref>
<ref id="ref13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hao</surname> <given-names>M.</given-names></name> <name><surname>Jiang</surname> <given-names>D.</given-names></name> <name><surname>Ding</surname> <given-names>F.</given-names></name> <name><surname>Fu</surname> <given-names>J.</given-names></name> <name><surname>Chen</surname> <given-names>S.</given-names></name></person-group> (<year>2019</year>). <article-title>Simulating Spatio-temporal patterns of terrorism incidents on the Indochina peninsula with GIS and the random Forest method</article-title>. <source>ISPRS Int. J. Geo Inf.</source> <volume>8</volume>:<fpage>133</fpage>. doi: <pub-id pub-id-type="doi">10.3390/ijgi8030133</pub-id></citation>
</ref>
<ref id="ref14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ibrahim</surname> <given-names>A. A.</given-names></name> <name><surname>Ridwan</surname> <given-names>R. L.</given-names></name> <name><surname>Muhammed</surname> <given-names>M. M.</given-names></name> <name><surname>Abdulaziz</surname> <given-names>R. O.</given-names></name> <name><surname>Saheed</surname> <given-names>G. A.</given-names></name></person-group> (<year>2020</year>). <article-title>Comparison of the CatBoost classifier with other machine learning methods</article-title>. <source>Int. J. Adv. Comput. Sci. Appl.</source> <volume>11</volume>:<fpage>11</fpage>. doi: <pub-id pub-id-type="doi">10.14569/IJACSA.2020.0111190</pub-id></citation>
</ref>
<ref id="ref15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jiang</surname> <given-names>C.</given-names></name> <name><surname>Zhu</surname> <given-names>F.</given-names></name> <name><surname>Qin</surname> <given-names>T.</given-names></name></person-group> (<year>2020</year>). <article-title>Relationships between chronic diseases and depression among middle-aged and elderly people in China: a prospective study from CHARLS</article-title>. <source>Curr. Med. Sci.</source> <volume>40</volume>, <fpage>858</fpage>&#x2013;<lpage>870</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s11596-020-2270-5</pub-id>, PMID: <pub-id pub-id-type="pmid">33123901</pub-id></citation>
</ref>
<ref id="ref16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jung</surname> <given-names>S. J.</given-names></name> <name><surname>Shin</surname> <given-names>A.</given-names></name> <name><surname>Kang</surname> <given-names>D.</given-names></name></person-group> (<year>2015</year>). <article-title>Menarche age, menopause age and other reproductive factors in association with post-menopausal onset depression: results from health examinees study (HEXA)</article-title>. <source>J. Affect. Disord.</source> <volume>187</volume>, <fpage>127</fpage>&#x2013;<lpage>135</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jad.2015.08.047</pub-id>, PMID: <pub-id pub-id-type="pmid">26339923</pub-id></citation>
</ref>
<ref id="ref17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kaup</surname> <given-names>A. R.</given-names></name> <name><surname>Byers</surname> <given-names>A. L.</given-names></name> <name><surname>Falvey</surname> <given-names>C.</given-names></name> <name><surname>Simonsick</surname> <given-names>E. M.</given-names></name> <name><surname>Satterfield</surname> <given-names>S.</given-names></name> <name><surname>Ayonayon</surname> <given-names>H. N.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>Trajectories of depressive symptoms in older adults and risk of dementia</article-title>. <source>JAMA Psychiatry</source> <volume>73</volume>, <fpage>525</fpage>&#x2013;<lpage>531</lpage>. doi: <pub-id pub-id-type="doi">10.1001/jamapsychiatry.2016.0004</pub-id>, PMID: <pub-id pub-id-type="pmid">26982217</pub-id></citation>
</ref>
<ref id="ref18">
<citation citation-type="other"><person-group person-group-type="author"><name><surname>Ke</surname> <given-names>G.</given-names></name> <name><surname>Meng</surname> <given-names>Q.</given-names></name> <name><surname>Finley</surname> <given-names>T.</given-names></name> <name><surname>Wang</surname> <given-names>T.</given-names></name> <name><surname>Chen</surname> <given-names>W.</given-names></name> <name><surname>Ma</surname> <given-names>W.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>Light GBM: A highly efficient gradient boosting decision tree</article-title>. <source>Adv. Neural Inf. Process Syst</source>. <volume>30</volume>.</citation>
</ref>
<ref id="ref19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kimmel</surname> <given-names>P. L.</given-names></name> <name><surname>Peterson</surname> <given-names>R. A.</given-names></name> <name><surname>Weihs</surname> <given-names>K. L.</given-names></name> <name><surname>Simmens</surname> <given-names>S. J.</given-names></name> <name><surname>Alleyne</surname> <given-names>S.</given-names></name> <name><surname>Cruz</surname> <given-names>I.</given-names></name> <etal/></person-group>. (<year>2000</year>). <article-title>Multiple measurements of depression predict mortality in a longitudinal study of chronic hemodialysis outpatients</article-title>. <source>Kidney Int.</source> <volume>57</volume>, <fpage>2093</fpage>&#x2013;<lpage>2098</lpage>. doi: <pub-id pub-id-type="doi">10.1046/j.1523-1755.2000.00059.x</pub-id></citation>
</ref>
<ref id="ref9005">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Klaise</surname> <given-names>J.</given-names></name> <name><surname>Van-Looveren</surname> <given-names>A.</given-names></name> <name><surname>Vacanti</surname> <given-names>G.</given-names></name> <name><surname>Coca</surname> <given-names>A.</given-names></name></person-group> (<year>2021</year>). <article-title>Alibi explain: Algorithms for explaining machine learning models</article-title>. <source>J. Mach. Learn. Res.</source> <volume>22</volume>, <fpage>1</fpage>&#x2013;<lpage>7</lpage>.</citation>
</ref>
<ref id="ref20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kuchibhatla</surname> <given-names>M. N.</given-names></name> <name><surname>Fillenbaum</surname> <given-names>G. G.</given-names></name> <name><surname>Hybels</surname> <given-names>C. F.</given-names></name> <name><surname>Blazer</surname> <given-names>D. G.</given-names></name></person-group> (<year>2012</year>). <article-title>Trajectory classes of depressive symptoms in a community sample of older adults</article-title>. <source>Acta Psychiatr. Scand.</source> <volume>125</volume>, <fpage>492</fpage>&#x2013;<lpage>501</lpage>. doi: <pub-id pub-id-type="doi">10.1111/j.1600-0447.2011.01801.x</pub-id>, PMID: <pub-id pub-id-type="pmid">22118370</pub-id></citation>
</ref>
<ref id="ref21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lee</surname> <given-names>S. W.</given-names></name> <name><surname>Yang</surname> <given-names>J. M.</given-names></name> <name><surname>Moon</surname> <given-names>S. Y.</given-names></name> <name><surname>Yoo</surname> <given-names>I. K.</given-names></name> <name><surname>Ha</surname> <given-names>E. K.</given-names></name> <name><surname>Kim</surname> <given-names>S. Y.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Association between mental illness and COVID-19 susceptibility and clinical outcomes in South Korea: a nationwide cohort study</article-title>. <source>Lancet Psychiatry</source> <volume>7</volume>, <fpage>1025</fpage>&#x2013;<lpage>1031</lpage>. doi: <pub-id pub-id-type="doi">10.1016/S2215-0366(20)30421-1</pub-id>, PMID: <pub-id pub-id-type="pmid">32950066</pub-id></citation>
</ref>
<ref id="ref22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lin</surname> <given-names>S.</given-names></name> <name><surname>Wu</surname> <given-names>Y.</given-names></name> <name><surname>He</surname> <given-names>L.</given-names></name> <name><surname>Fang</surname> <given-names>Y.</given-names></name></person-group> (<year>2023</year>). <article-title>Prediction of depressive symptoms onset and long-term trajectories in home-based older adults using machine learning techniques</article-title>. <source>Aging Ment. Health</source> <volume>27</volume>, <fpage>8</fpage>&#x2013;<lpage>17</lpage>. doi: <pub-id pub-id-type="doi">10.1080/13607863.2022.2031868</pub-id>, PMID: <pub-id pub-id-type="pmid">35118924</pub-id></citation>
</ref>
<ref id="ref23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lundberg</surname> <given-names>S. M.</given-names></name> <name><surname>Lee</surname> <given-names>S.-I.</given-names></name></person-group> (<year>2017</year>). <article-title>A unified approach to interpreting model predictions</article-title>.  <source>In Advances in Neural Information Processing Systems</source> <volume>30</volume> (pp. <fpage>4765</fpage>&#x2013;<lpage>4774</lpage>). Eds. <person-group person-group-type="editor"><name><surname>Guyon</surname> <given-names>I.</given-names></name> <name><surname>Luxburg</surname> <given-names>U. V.</given-names></name> <name><surname>Bengio</surname> <given-names>S.</given-names></name> <name><surname>Wallach</surname> <given-names>H.</given-names></name> <name><surname>Fergus</surname> <given-names>R.</given-names></name> <name><surname>Vishwanathan</surname> <given-names>S.</given-names></name></person-group>. <publisher-loc>Curran Associates, Inc</publisher-loc>. Available at: <ext-link xlink:href="http://papers.nips.cc/paper/7062-a-unified-approach-to-interpreting-model-predictions.pdf" ext-link-type="uri">http://papers.nips.cc/paper/7062-a-unified-approach-to-interpreting-model-predictions.pdf</ext-link></citation>
</ref>
<ref id="ref24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Luo</surname> <given-names>H.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>Q.</given-names></name> <name><surname>Cao</surname> <given-names>P.</given-names></name> <name><surname>Ren</surname> <given-names>X.</given-names></name> <name><surname>Fang</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Obesity and the onset of depressive symptoms among middle-aged and older adults in China: evidence from the CHARLS</article-title>. <source>BMC Public Health</source> <volume>18</volume>:<fpage>909</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s12889-018-5834-6</pub-id>, PMID: <pub-id pub-id-type="pmid">30041632</pub-id></citation>
</ref>
<ref id="ref25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Luo</surname> <given-names>L.</given-names></name> <name><surname>Qi</surname> <given-names>C.</given-names></name></person-group> (<year>2021</year>). <article-title>An analysis of the crucial indicators impacting the risk of terrorist attacks: a predictive perspective</article-title>. <source>Saf. Sci.</source> <volume>144</volume>:<fpage>105442</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.ssci.2021.105442</pub-id></citation>
</ref>
<ref id="ref26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Luo</surname> <given-names>Y.</given-names></name> <name><surname>Yang</surname> <given-names>P.</given-names></name> <name><surname>Wan</surname> <given-names>Z.</given-names></name> <name><surname>Kang</surname> <given-names>Y.</given-names></name> <name><surname>Dong</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Dietary diversity, physical activity and depressive symptoms among middle-aged women: a cross-sectional study of 48,637 women in China</article-title>. <source>J. Affect. Disord.</source> <volume>321</volume>, <fpage>147</fpage>&#x2013;<lpage>152</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jad.2022.10.038</pub-id>, PMID: <pub-id pub-id-type="pmid">36330900</pub-id></citation>
</ref>
<ref id="ref27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Maramis</surname> <given-names>M. M.</given-names></name> <name><surname>Mahajudin</surname> <given-names>M. S.</given-names></name> <name><surname>Khotib</surname> <given-names>J.</given-names></name></person-group> (<year>2021</year>). <article-title>Impaired cognitive flexibility and working memory precedes depression: a rat model to study depression</article-title>. <source>Neuropsychobiology</source> <volume>80</volume>, <fpage>225</fpage>&#x2013;<lpage>233</lpage>. doi: <pub-id pub-id-type="doi">10.1159/000508682</pub-id>, PMID: <pub-id pub-id-type="pmid">32712605</pub-id></citation>
</ref>
<ref id="ref28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>McMunn</surname> <given-names>A.</given-names></name> <name><surname>Lacey</surname> <given-names>R.</given-names></name> <name><surname>Worts</surname> <given-names>D.</given-names></name> <name><surname>Kuh</surname> <given-names>D.</given-names></name> <name><surname>McDonough</surname> <given-names>P.</given-names></name> <name><surname>Sacker</surname> <given-names>A.</given-names></name></person-group> (<year>2021</year>). <article-title>Work-family life courses and psychological distress: evidence from three British birth cohort studies</article-title>. <source>Adv. Life Course Res.</source> <volume>50</volume>:<fpage>100429</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.alcr.2021.100429</pub-id>, PMID: <pub-id pub-id-type="pmid">36661289</pub-id></citation>
</ref>
<ref id="ref29">
<citation citation-type="other"><person-group person-group-type="author"><name><surname>Molnar</surname> <given-names>C.</given-names></name>
</person-group> (<year>2020</year>). <article-title>Interpretable machine learning</article-title>. <source>Lulu. com.</source></citation>
</ref>
<ref id="ref30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Peleg</surname> <given-names>S.</given-names></name> <name><surname>Nudelman</surname> <given-names>G.</given-names></name></person-group> (<year>2021</year>). <article-title>Associations between self-rated health and depressive symptoms among older adults: does age matter?</article-title> <source>Soc. Sci. Med.</source> <volume>280</volume>:<fpage>114024</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.socscimed.2021.114024</pub-id>, PMID: <pub-id pub-id-type="pmid">34049050</pub-id></citation>
</ref>
<ref id="ref31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Planchez</surname> <given-names>B.</given-names></name> <name><surname>Surget</surname> <given-names>A.</given-names></name> <name><surname>Belzung</surname> <given-names>C.</given-names></name></person-group> (<year>2019</year>). <article-title>Animal models of major depression: drawbacks and challenges</article-title>. <source>J. Neural Transm.</source> <volume>126</volume>, <fpage>1383</fpage>&#x2013;<lpage>1408</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s00702-019-02084-y</pub-id>, PMID: <pub-id pub-id-type="pmid">31584111</pub-id></citation>
</ref>
<ref id="ref9002">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Prokhorenkova</surname> <given-names>L.</given-names></name> <name><surname>Gusev</surname> <given-names>G.</given-names></name> <name><surname>Vorobev</surname> <given-names>A.</given-names></name> <name><surname>Dorogush</surname> <given-names>A. V.</given-names></name> <name><surname>Gulin</surname> <given-names>A.</given-names></name></person-group> (<year>2018</year>). <source>CatBoost: Unbiased boosting with categorical features</source>. <publisher-name>Adv. Neural Inf</publisher-name>. <publisher-loc>Process Syst</publisher-loc>. <fpage>31</fpage>.</citation>
</ref>
<ref id="ref32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Qiu</surname> <given-names>Q.</given-names></name> <name><surname>Qian</surname> <given-names>S.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Jia</surname> <given-names>R.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Xu</surname> <given-names>Y.</given-names></name></person-group> (<year>2020</year>). <article-title>Risk factors for depressive symptoms among older Chinese adults: a meta-analysis</article-title>. <source>J. Affect. Disord.</source> <volume>277</volume>, <fpage>341</fpage>&#x2013;<lpage>346</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jad.2020.08.036</pub-id>, PMID: <pub-id pub-id-type="pmid">32861154</pub-id></citation>
</ref>
<ref id="ref33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sun</surname> <given-names>Z.</given-names></name> <name><surname>Wang</surname> <given-names>Z.</given-names></name> <name><surname>Xu</surname> <given-names>L.</given-names></name> <name><surname>Lv</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>Q.</given-names></name> <name><surname>Wang</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Characteristics of cognitive deficit in amnestic mild cognitive impairment with subthreshold depression</article-title>. <source>J. Geriatr. Psychiatry Neurol.</source> <volume>32</volume>, <fpage>344</fpage>&#x2013;<lpage>353</lpage>. doi: <pub-id pub-id-type="doi">10.1177/0891988719865943</pub-id>, PMID: <pub-id pub-id-type="pmid">31480987</pub-id></citation>
</ref>
<ref id="ref34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tjoa</surname> <given-names>E.</given-names></name> <name><surname>Guan</surname> <given-names>C.</given-names></name></person-group> (<year>2021</year>). <article-title>A survey on explainable artificial intelligence (XAI): toward medical XAI</article-title>. <source>IEEE Trans. Neural Netw. Learn. Syst.</source> <volume>32</volume>, <fpage>4793</fpage>&#x2013;<lpage>4813</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TNNLS.2020.3027314</pub-id>, PMID: <pub-id pub-id-type="pmid">33079674</pub-id></citation>
</ref>
<ref id="ref35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Troyanskaya</surname> <given-names>O.</given-names></name> <name><surname>Cantor</surname> <given-names>M.</given-names></name> <name><surname>Sherlock</surname> <given-names>G.</given-names></name> <name><surname>Brown</surname> <given-names>P.</given-names></name> <name><surname>Hastie</surname> <given-names>T.</given-names></name> <name><surname>Tibshirani</surname> <given-names>R.</given-names></name> <etal/></person-group>. (<year>2001</year>). <article-title>Missing value estimation methods for DNA microarrays</article-title>. <source>Bioinformatics</source> <volume>17</volume>, <fpage>520</fpage>&#x2013;<lpage>525</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/17.6.520</pub-id></citation>
</ref>
<ref id="ref36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wachowska</surname> <given-names>K.</given-names></name> <name><surname>Szemraj</surname> <given-names>J.</given-names></name> <name><surname>&#x015A;migielski</surname> <given-names>J.</given-names></name> <name><surname>Ga&#x0142;ecki</surname> <given-names>P.</given-names></name></person-group> (<year>2022</year>). <article-title>Inflammatory markers and episodic memory functioning in depressive disorders</article-title>. <source>J. Clin. Med.</source> <volume>11</volume>:<fpage>693</fpage>. doi: <pub-id pub-id-type="doi">10.3390/jcm11030693</pub-id>, PMID: <pub-id pub-id-type="pmid">35160143</pub-id></citation>
</ref>
<ref id="ref37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wen</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>C.</given-names></name> <name><surname>Liao</surname> <given-names>J.</given-names></name> <name><surname>Yin</surname> <given-names>Y.</given-names></name> <name><surname>Wu</surname> <given-names>D.</given-names></name></person-group> (<year>2019</year>). <article-title>Incidence and risk factors of depressive symptoms in 4 years of follow-up among mid-aged and elderly community-dwelling Chinese adults: findings from the China health and retirement longitudinal study</article-title>. <source>BMJ Open</source> <volume>9</volume>:<fpage>e029529</fpage>. doi: <pub-id pub-id-type="doi">10.1136/bmjopen-2019-029529</pub-id>, PMID: <pub-id pub-id-type="pmid">31501114</pub-id></citation>
</ref>
<ref id="ref9001">
<citation citation-type="journal"><person-group person-group-type="author">
<collab id="coll2001">World Health Organization</collab>
</person-group>. (<year>2017</year>). <source>Depression and other common mental disorders: global health estimates (No. WHO/MSD/MER/2017.2)</source>. Available at: <ext-link xlink:href="https://www.who.int/publications/i/item/depression-global-health-estimates" ext-link-type="uri">https://www.who.int/publications/i/item/depression-global-health-estimates</ext-link></citation>
</ref>
<ref id="ref38">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xiang</surname> <given-names>X.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name></person-group> (<year>2021</year>). <article-title>Childhood adversity and major depression in later life: a competing-risks regression analysis</article-title>. <source>Int. J. Geriatr. Psychiatry</source> <volume>36</volume>, <fpage>215</fpage>&#x2013;<lpage>223</lpage>. doi: <pub-id pub-id-type="doi">10.1002/gps.5417</pub-id>, PMID: <pub-id pub-id-type="pmid">32869351</pub-id></citation>
</ref>
<ref id="ref39">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yoo</surname> <given-names>J. S.</given-names></name> <name><surname>Chang</surname> <given-names>S. J.</given-names></name> <name><surname>Kim</surname> <given-names>H. S.</given-names></name></person-group> (<year>2016</year>). <article-title>Prevalence and predictive factors of depression in community-dwelling older adults in South Korea</article-title>. <source>Res. Theory Nurs. Pract.</source> <volume>30</volume>, <fpage>200</fpage>&#x2013;<lpage>211</lpage>. doi: <pub-id pub-id-type="doi">10.1891/1541-6577.30.3.200</pub-id>, PMID: <pub-id pub-id-type="pmid">28304266</pub-id></citation>
</ref>
<ref id="ref40">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yunming</surname> <given-names>L.</given-names></name> <name><surname>Changsheng</surname> <given-names>C.</given-names></name> <name><surname>Haibo</surname> <given-names>T.</given-names></name> <name><surname>Wenjun</surname> <given-names>C.</given-names></name> <name><surname>Shanhong</surname> <given-names>F.</given-names></name> <name><surname>Yan</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2012</year>). <article-title>Prevalence and risk factors for depression in older people in xi&#x2032;an China: a community-based study</article-title>. <source>Int. J. Geriatr. Psychiatry</source> <volume>27</volume>, <fpage>31</fpage>&#x2013;<lpage>39</lpage>. doi: <pub-id pub-id-type="doi">10.1002/gps.2685</pub-id>, PMID: <pub-id pub-id-type="pmid">21284042</pub-id></citation>
</ref>
<ref id="ref41">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zeng</surname> <given-names>Y.</given-names></name>
</person-group> (<year>2012</year>). <article-title>Toward deeper research and better policy for healthy aging &#x2013; using the unique data of Chinese longitudinal healthy longevity survey</article-title>. <source>China Econ. J.</source> <volume>5</volume>, <fpage>131</fpage>&#x2013;<lpage>149</lpage>. doi: <pub-id pub-id-type="doi">10.1080/17538963.2013.764677</pub-id>, PMID: <pub-id pub-id-type="pmid">24443653</pub-id></citation>
</ref>
<ref id="ref42">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>C.</given-names></name> <name><surname>Chen</surname> <given-names>X.</given-names></name> <name><surname>Wang</surname> <given-names>S.</given-names></name> <name><surname>Hu</surname> <given-names>J.</given-names></name> <name><surname>Wang</surname> <given-names>C.</given-names></name> <name><surname>Liu</surname> <given-names>X.</given-names></name></person-group> (<year>2021</year>). <article-title>Using CatBoost algorithm to identify middle-aged and elderly depression, national health and nutrition examination survey 2011&#x2013;2018</article-title>. <source>Psychiatry Res.</source> <volume>306</volume>:<fpage>114261</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.psychres.2021.114261</pub-id>, PMID: <pub-id pub-id-type="pmid">34781111</pub-id></citation>
</ref>
<ref id="ref43">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Jin</surname> <given-names>M.</given-names></name> <name><surname>Fu</surname> <given-names>J.</given-names></name> <name><surname>Hao</surname> <given-names>M.</given-names></name> <name><surname>Yu</surname> <given-names>C.</given-names></name> <name><surname>Xie</surname> <given-names>X.</given-names></name></person-group> (<year>2018</year>). <article-title>On the risk assessment of terrorist attacks coupled with multi-source factors</article-title>. <source>ISPRS Int. J. Geo Inf.</source> <volume>7</volume>:<fpage>354</fpage>. doi: <pub-id pub-id-type="doi">10.3390/ijgi7090354</pub-id></citation>
</ref>
<ref id="ref44">
<citation citation-type="other"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>Z.</given-names></name> <name><surname>Jung</surname> <given-names>C.</given-names></name></person-group> (<year>2019</year>). <article-title>GBDT-MO: Gradient boosted decision trees for multiple outputs (arXiv:1909.04373). arXiv</article-title>. Available at:<ext-link xlink:href="http://arxiv.org/abs/1909.04373" ext-link-type="uri">http://arxiv.org/abs/1909.04373</ext-link></citation>
</ref>
<ref id="ref45">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhao</surname> <given-names>X.</given-names></name> <name><surname>Yu</surname> <given-names>B.</given-names></name> <name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Chen</surname> <given-names>Z.</given-names></name> <name><surname>Li</surname> <given-names>Q.</given-names></name> <name><surname>Wang</surname> <given-names>C.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Estimation of poverty using random Forest regression with multi-source data: a case study in Bangladesh</article-title>. <source>Remote Sens.</source> <volume>11</volume>:<fpage>375</fpage>. doi: <pub-id pub-id-type="doi">10.3390/rs11040375</pub-id></citation>
</ref>
<ref id="ref46">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhao</surname> <given-names>Y.</given-names></name> <name><surname>Hu</surname> <given-names>Y.</given-names></name> <name><surname>Smith</surname> <given-names>J. P.</given-names></name> <name><surname>Strauss</surname> <given-names>J.</given-names></name> <name><surname>Yang</surname> <given-names>G.</given-names></name></person-group> (<year>2014</year>). <article-title>Cohort profile: the China health and retirement longitudinal study (CHARLS)</article-title>. <source>Int. J. Epidemiol.</source> <volume>43</volume>, <fpage>61</fpage>&#x2013;<lpage>68</lpage>. doi: <pub-id pub-id-type="doi">10.1093/ije/dys203</pub-id>, PMID: <pub-id pub-id-type="pmid">23243115</pub-id></citation>
</ref>
<ref id="ref47">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>L.</given-names></name> <name><surname>Ma</surname> <given-names>X.</given-names></name> <name><surname>Wang</surname> <given-names>W.</given-names></name></person-group> (<year>2021</year>). <article-title>Relationship between cognitive performance and depressive symptoms in Chinese older adults: the China health and retirement longitudinal study (CHARLS)</article-title>. <source>J. Affect. Disord.</source> <volume>281</volume>, <fpage>454</fpage>&#x2013;<lpage>458</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jad.2020.12.059</pub-id>, PMID: <pub-id pub-id-type="pmid">33360747</pub-id></citation>
</ref>
<ref id="ref48">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>P.</given-names></name> <name><surname>Wang</surname> <given-names>S.</given-names></name> <name><surname>Yan</surname> <given-names>Y.</given-names></name> <name><surname>Lu</surname> <given-names>Q.</given-names></name> <name><surname>Pei</surname> <given-names>J.</given-names></name> <name><surname>Guo</surname> <given-names>W.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Association between chronic diseases and depression in the middle-aged and older adult Chinese population&#x2014;a seven-year follow-up study based on CHARLS</article-title>. <source>Front. Public Health</source> <volume>11</volume>:<fpage>1176669</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fpubh.2023.1176669</pub-id>, PMID: <pub-id pub-id-type="pmid">37546300</pub-id></citation>
</ref>
<ref id="ref49">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>S.</given-names></name> <name><surname>Gao</surname> <given-names>L.</given-names></name> <name><surname>Liu</surname> <given-names>F.</given-names></name> <name><surname>Tian</surname> <given-names>W.</given-names></name> <name><surname>Jin</surname> <given-names>Y.</given-names></name> <name><surname>Zheng</surname> <given-names>Z.</given-names></name></person-group> (<year>2021</year>). <article-title>Socioeconomic status and depressive symptoms in older people with the mediation role of social support: a population-based longitudinal study</article-title>. <source>Int. J. Methods Psychiatr. Res.</source> <volume>30</volume>:<fpage>e1894</fpage>. doi: <pub-id pub-id-type="doi">10.1002/mpr.1894</pub-id>, PMID: <pub-id pub-id-type="pmid">34591341</pub-id></citation>
</ref>
</ref-list>
</back>
</article>