<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Physiol.</journal-id>
<journal-title>Frontiers in Physiology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Physiol.</abbrev-journal-title>
<issn pub-type="epub">1664-042X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1357123</article-id>
<article-id pub-id-type="doi">10.3389/fphys.2024.1357123</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Physiology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>SRPNet: stroke risk prediction based on two-level feature selection and deep fusion network</article-title>
<alt-title alt-title-type="left-running-head">Zhang et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fphys.2024.1357123">10.3389/fphys.2024.1357123</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Daoliang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yu</surname>
<given-names>Na</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yang</surname>
<given-names>Xiaodan</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>De Marinis</surname>
<given-names>Yang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/666834/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Liu</surname>
<given-names>Zhi-Ping</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/126999/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Gao</surname>
<given-names>Rui</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2606896/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>School of Control Science and Engineering</institution>, <institution>Shandong University</institution>, <addr-line>Jinan</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Rehabilitation Medicine</institution>, <institution>Affiliated Hospital of Jining Medical University</institution>, <addr-line>Jining</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Clinical Sciences</institution>, <institution>Lund University</institution>, <addr-line>Malm&#xf6;</addr-line>, <country>Sweden</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/20094/overview">Raimond L. Winslow</ext-link>, Northeastern University, United States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/847160/overview">Ming Huang</ext-link>, Nagoya City University, Japan</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/779009/overview">Eric S. Ho</ext-link>, Lafayette College, United States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Zhi-Ping Liu, <email>zpliu@sdu.edu.cn</email>; Rui Gao, <email>gaorui@sdu.edu.cn</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>11</day>
<month>11</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>15</volume>
<elocation-id>1357123</elocation-id>
<history>
<date date-type="received">
<day>20</day>
<month>12</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>23</day>
<month>10</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Zhang, Yu, Yang, De Marinis, Liu and Gao.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Zhang, Yu, Yang, De Marinis, Liu and Gao</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Background</title>
<p>Stroke is one of the major chronic non-communicable diseases (NCDs) with high morbidity, disability and mortality. The key to preventing stroke lies in controlling risk factors. However, screening risk factors and quantifying stroke risk levels remain challenging.</p>
</sec>
<sec>
<title>Methods</title>
<p>A novel prediction model for stroke risk based on two-level feature selection and deep fusion network (SRPNet) is proposed to solve the problem mentioned above. First, the two-level feature selection method is used to screen comprehensive features related to stroke risk, enabling accurate identification of significant risk factors while eliminating redundant information. Next, the deep fusion network integrating Transformer and fully connected neural network (FCN) is utilized to establish the risk prediction model SRPNet for stroke patients.</p>
</sec>
<sec>
<title>Results</title>
<p>We evaluate the performance of the SRPNet using screening data from the China Stroke Data Center (CSDC), and further validate its effectiveness with census data on stroke collected in affiliated hospital of Jining Medical University. The experimental results demonstrate that the SRPNet model selects features closely related to stroke and achieves superior risk prediction performance over benchmark methods.</p>
</sec>
<sec>
<title>Conclusions</title>
<p>SRPNet can rapidly identify high-quality stroke risk factors, improve the accuracy of stroke prediction, and provide a powerful tool for clinical diagnosis.</p>
</sec>
</abstract>
<kwd-group>
<kwd>stroke risk prediction</kwd>
<kwd>feature selection</kwd>
<kwd>deep fusion network</kwd>
<kwd>transformer</kwd>
<kwd>stroke risk factors</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Computational Physiology and Medicine</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Stroke is a global public health issue, ranking as the second leading cause of death and the third leading cause of disability worldwide (<xref ref-type="bibr" rid="B30">Owolabi et al., 2022</xref>). Moreover, the incidence of stroke is increasing in recent years, and the burden of stroke poses a huge challenge to low- and middle-income countries (<xref ref-type="bibr" rid="B31">Owolabi et al., 2021</xref>). However, the complexity, suddenness, and significant differences in clinical manifestations of stroke have brought great difficulties to treatment. It is widely acknowledged that stroke is preventable and controllable (<xref ref-type="bibr" rid="B16">Johnson et al., 2019</xref>). Therefore, active intervention on risk factors of stroke and accurate prediction of stroke risk through early screening can assist doctors and patients in implementing appropriate preventive and therapeutic measures, significantly reducing the harm caused by stroke.</p>
<p>So far, some studies employed traditional medical statistical methods to predict stroke risk (<xref ref-type="bibr" rid="B39">Wang et al., 2022</xref>; <xref ref-type="bibr" rid="B2">Abraham et al., 2021</xref>). These methods typically relied on a series of risk factors to construct mathematical models for calculating risk scores. However, these methods were time-consuming and labor-intensive, and ignored the complex nonlinear relationships and interactions among features, resulting in limited prediction performances (<xref ref-type="bibr" rid="B28">Obermeyer and Emanuel, 2016</xref>). With the rapid development of artificial intelligence, machine learning methods provide new solutions for stroke risk prediction. The machine learning methods can process complex screening data, and reveal patterns and associations hidden within large-scale data, thereby enhancing the accuracy of stroke risk prediction.</p>
<p>A better understanding of risk factors is critical for stroke diagnostic evaluation and treatment decision. In fact, controlling the risk factors (such as hypertension and diabetes) can reduce the risk of stroke. <xref ref-type="bibr" rid="B34">Qi et al. (2020)</xref> used multi-variable Cox regression analysis to obtain the features associated with the occurrence of stroke and its subtypes in China by introducing socioeconomic and other related factors. <xref ref-type="bibr" rid="B1">Abraham et al. (2019)</xref> employed elastic-net logistic regression to screen for genetic risk factors of stroke. <xref ref-type="bibr" rid="B15">Hunter and Kelleher (2023)</xref> used data from NHLBI Biologic Specimen and cardiac studies as risk factors, and studied the effect of age on stroke risk factors through a logistic regression algorithm. <xref ref-type="bibr" rid="B23">Maalouf et al. (2023)</xref> developed the regression model to find that negative emotions could increase stroke risk. Generally, stroke is a complex disease, and it is difficult to predict stroke risk via a single feature. However, having too many types of features may lead to redundant information and increase diagnostic costs. Furthermore, different risk factors contribute differently to stroke occurrence. More importantly, considering the association relationship among features is expected to be beneficial for the early stroke screening. Therefore, there is an urgent need to develop effective feature selection methods for predicting stroke risk.</p>
<p>Currently, numerous studies have been devoted to stroke risk prediction using machine learning techniques. For example, <xref ref-type="bibr" rid="B19">Li et al. (2019b)</xref> applied the Bayesian network model to estimate the incidence of stroke, revealing the relationship between combinations of multiple risk factors and stroke. <xref ref-type="bibr" rid="B27">Nwosu et al. (2019)</xref> analyzed the electronic health records of patients using neural networks, decision trees, and random forests to determine the impact of risk factors on stroke prediction. <xref ref-type="bibr" rid="B5">Arafa et al. (2022)</xref> developed a stroke risk prediction method for urban Japanese based on the Cox proportional hazards model, incorporating cardiovascular risk factors. <xref ref-type="bibr" rid="B11">Dritsas and Trigka (2022)</xref> designed an ensemble learning method for long-term stroke risk prediction. <xref ref-type="bibr" rid="B21">Liu et al. (2019)</xref> first adopted the random forest regression algorithm to impute missing data, and then used the deep neural network (DNN) to predict stroke on imbalanced physiological data. Although the above methods achieved promising results, the model structures they employed are relatively disconnected between features and algorithms, and the generalization ability of these models needs to be improved.</p>
<p>Here we propose a novel prediction model based on two-level feature selection and deep fusion network, termed SRPNet, for inferring stroke risk. In particular, two-level feature selection can comprehensively search for significant features related to stroke risk. We first apply multiple methods including Pearson correlation, chi-square test, Lasso and elastic net to select risk factors respectively, and combine the obtained risk factors as a candidate feature set. We then traverse all candidate risk factor combinations in the feature set by seven machine learning methods, such as support vector machine (SVM), k-nearest neighbor (KNN), decision tree (DT), gradient boosting decision tree (GBDT), random forest (RF), Gaussian Naive Bayes (GaussianNB) and AdaBoost, to identify the most important features associated with stroke. This enables evaluating the correlations between features and eliminating redundant information, providing reliable risk factors for stroke screening program. Next, the proposed deep fusion network integrates Transformer (<xref ref-type="bibr" rid="B38">Vaswani et al., 2017</xref>) and fully connected neural network (FCN) (<xref ref-type="bibr" rid="B22">Long et al., 2015</xref>) to establish a risk prediction model for stroke patients. This prediction model utilizes the attention mechanism of Transformer to explore hidden relationships among risk factors, and adopts FCN to better capture the nonlinear relationships among features. The experimental results indicate that SRPNet improves the accuracy and efficiency of stroke screening, and its performance is superior to existing benchmark methods. This work provides assistance for clinical diagnosis, and alleviates the burden of stroke.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>2 Materials and methods</title>
<sec id="s2-1">
<title>2.1 Datasets</title>
<p>The CSDC database covers 6 provinces, 41 hospitals and 12 population cohorts in China (<xref ref-type="bibr" rid="B41">Yu et al., 2016</xref>). The CSDC database facilitates stroke-related decision-making, research, and public health services through a comprehensive system. It collects and analyzes patient data, including risk factors, medical history, and sociodemographic information, ensuring that each subject has a unique record. A two-stage stratified cluster sampling method was employed during the data screening process (<xref ref-type="bibr" rid="B18">Li et al., 2019a</xref>). First, more than 200 screening areas were selected based on the local population size and the total number of counties. Then, urban communities and townships were used as the primary sampling units (PSUs) according to the geographical location and the recommendations from the local hospitals. In each PSU, all residents aged 40 and above were surveyed using cluster sampling during the initial screening period. Doctors assessed each patient&#x2019;s condition, categorizing them as low risk, medium risk, high risk, transient ischemic attack (TIA), or stroke. The CSDC dataset comprises 862,244 middle-aged residents. <xref ref-type="table" rid="T1">Table 1</xref> shows the detailed features of the CSDC dataset.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Summary of specific features in the CSDC dataset.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Risk factors</th>
<th align="center">Statistics</th>
<th align="center">Abbreviation</th>
<th align="center">Risk factors</th>
<th align="center">Statistics</th>
<th align="center">Abbreviation</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Age group</td>
<td align="center">54.48 &#xb1; 11.25</td>
<td align="center">AG</td>
<td align="center">Diabetes</td>
<td align="center">49,674/812,570</td>
<td align="center">Diabetes</td>
</tr>
<tr>
<td align="center">Gender (male/female)</td>
<td align="center">397,765/464,479</td>
<td align="center">Gender</td>
<td align="center">Lack of exercise</td>
<td align="center">169,500/692,744</td>
<td align="center">LE</td>
</tr>
<tr>
<td align="center">Ethnic groups (minorities/majority)</td>
<td align="center">2,716/859,528</td>
<td align="center">EG</td>
<td align="center">Overweight</td>
<td align="center">148,834/713,410</td>
<td align="center">Overweight</td>
</tr>
<tr>
<td align="center">Occupation (mental/manual)</td>
<td align="center">147,585/650,577</td>
<td align="center">Occupation</td>
<td align="center">Number of Marriages<xref ref-type="table-fn" rid="Tfn2">
<sup>c</sup>
</xref>
</td>
<td align="center">0.94 &#xb1; 0.23</td>
<td align="center">NM</td>
</tr>
<tr>
<td align="center">Education status<xref ref-type="table-fn" rid="Tfn3">
<sup>a</sup>
</xref>
</td>
<td align="center">1.79 &#xb1; 0.93</td>
<td align="center">ES</td>
<td align="center">Marital status</td>
<td align="center">783,723/78,521</td>
<td align="center">MS</td>
</tr>
<tr>
<td align="center">Family history of stroke/hypertension/coronary heart disease<xref ref-type="table-fn" rid="Tfn1">
<sup>b</sup>
</xref>
</td>
<td align="center">60,320/801,924</td>
<td align="center">FHS/HYP/CHD</td>
<td align="center">Marriage_other</td>
<td align="center">3,537/858,707</td>
<td align="center">MO</td>
</tr>
<tr>
<td align="center">History of stroke</td>
<td align="center">16,862/845,382</td>
<td align="center">HS</td>
<td align="center">Provincial GDP</td>
<td align="center">39.12 &#xb1; 15.88</td>
<td align="center">PGDP</td>
</tr>
<tr>
<td align="center">Hypertension</td>
<td align="center">182,800/679,444</td>
<td align="center">HYP</td>
<td align="center">Province longitude</td>
<td align="center">112.94 &#xb1; 6.61</td>
<td align="center">PLO</td>
</tr>
<tr>
<td align="center">Atrial fibrillation</td>
<td align="center">23,445/838,799</td>
<td align="center">AF</td>
<td align="center">Province latitude</td>
<td align="center">35.18 &#xb1; 2.96</td>
<td align="center">PLA</td>
</tr>
<tr>
<td align="center">Low-Density Lipoprotein Cholesterol</td>
<td align="center">270,313/591,931</td>
<td align="center">LDL-C</td>
<td align="center">Province precipitation</td>
<td align="center">721.27 &#xb1; 202.66</td>
<td align="center">PP</td>
</tr>
<tr>
<td align="center">Province&#x2019;s highest temperature</td>
<td align="center">26.83 &#xb1; 2.08</td>
<td align="center">PHT</td>
<td align="center">Province&#x2019;s highest humidity</td>
<td align="center">78.60 &#xb1; 4.67</td>
<td align="center">PHH</td>
</tr>
<tr>
<td align="center">Province&#x2019;s lowest temperature</td>
<td align="center">&#x2212;0.09 &#xb1; 3.32</td>
<td align="center">PLT</td>
<td align="center">Province&#x2019;s lowest humidity</td>
<td align="center">55.36 &#xb1; 12.80</td>
<td align="center">PLH</td>
</tr>
<tr>
<td align="center">Smoking</td>
<td align="center">155,982/706,262</td>
<td align="center">Smoking</td>
<td align="center">Category<xref ref-type="table-fn" rid="Tfn4">
<sup>d</sup>
</xref>
</td>
<td align="center">384,272/477,972</td>
<td align="center">Category</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="Tfn1">
<label>
<sup>a</sup>
</label>
<p>The education status is divided into five levels, where 0 indicates illiteracy, 1 represents primary education, 2 represents secondary education, 3 represents higher education, and 4 represents postgraduate education.</p>
</fn>
<fn id="Tfn2">
<label>
<sup>b</sup>
</label>
<p>The counts of subjects with and without family history of stroke/hypertension/coronary heart disease.</p>
</fn>
<fn id="Tfn3">
<label>
<sup>c</sup>
</label>
<p>The number of marriages represents the number of times a subject has been married, with 0 for single, 1 for once married, and 2 for remarried.</p>
</fn>
<fn id="Tfn4">
<label>
<sup>d</sup>
</label>
<p>The variable category represents the categories of random grouping.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The in-house data is sourced from the medical records of 49 patients at affiliated hospital of Jining Medical University in 2023. It includes 14 features such as gender, age group, ethnic groups, marital status, occupation, education level, hypertension, atrial fibrillation, smoking, hyperlipidemia, diabetes, overweight, and family history of stroke. Each patient has been diagnosed by a physician and classified as either having suffered a stroke or being in good health. The summary information for these two datasets is listed in <xref ref-type="table" rid="T2">Table 2</xref>.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>The detailed information of datasets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Datasets</th>
<th align="center">&#x23; of samples</th>
<th align="center">&#x23; of features</th>
<th align="center">Phenotypes of samples</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">CSDC</td>
<td align="center">862,244</td>
<td align="center">26</td>
<td align="center">Low risk (612,819), Medium risk (124,103), High risk (85,155), TIA (23,305), Stroke (16,862)</td>
</tr>
<tr>
<td align="center">In-house data</td>
<td align="center">49</td>
<td align="center">14</td>
<td align="center">Health (24), Stroke (25)</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2-2">
<title>2.2 Overview of SRPNet</title>
<p>The SRPNet model mainly consists of two modules: two-level feature selection, and deep fusion network. The overall framework is illustrated in <xref ref-type="fig" rid="F1">Figure 1</xref>. Since the dataset contains text information, the SRPNet firstly performs data preprocessing, which involves digitizing the textual information and normalizing the data. To eliminate low-correlation and redundant features, the two-level feature selection method is employed to identify comprehensive features associated with stroke. Finally, the deep fusion network, which adaptively fuse Transformer and FCN by attention mechanism, takes the obtained significant features as input to provide accurate stroke risk prediction results for stroke patients.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>The entire framework of SRPNet. <bold>(A)</bold> Preprocess the input data. <bold>(B)</bold> Select significant features using the two-level feature selection. <bold>(C)</bold> Predict stroke risk based on the deep fusion network. <bold>(D)</bold> Output the stroke prediction results.</p>
</caption>
<graphic xlink:href="fphys-15-1357123-g001.tif"/>
</fig>
</sec>
<sec id="s2-3">
<title>2.3 Data preprocessing</title>
<p>Based on the stroke risk researches (<xref ref-type="bibr" rid="B37">Tian et al., 2019</xref>; <xref ref-type="bibr" rid="B13">Guan et al., 2019</xref>), we used text information digitization to convert non-numeric features into numeric vectors suitable for machine learning or deep learning methods. Occupations are divided into mental workers and manual workers. For the marital status, we characterize it by whether the respondent is currently married and the number of marriage times. Based on the location information of the respondents, we convert it to the local climate, such as maximum temperature, minimum temperature, precipitation, humidity, etc. All of which are closely related to stroke. For the remaining features, we also use similar knowledge-based feature engineering for feature representation. Data normalization (<xref ref-type="bibr" rid="B33">Park et al., 2022</xref>) is used to scale data elements to the (0,1) interval, which helps improve the effectiveness and reliability of model training. The normalization formula is defined as follows <xref ref-type="disp-formula" rid="e1">Equation 1</xref>:<disp-formula id="e1">
<mml:math id="m1">
<mml:mrow>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>min</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>max</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>min</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>
</p>
</sec>
<sec id="s2-4">
<title>2.4 Two-level feature selection</title>
<p>In this section, the two-level feature selection method that contains two-step feature selection processes will be introduced. The first step of feature selection involves four distinct methods, which are Pearson correlation, chi-square test, Lasso, and elastic net. The union of selected features from each method forms a set of candidate features. In the second step, based on the seven machine learning models, such as SVM, KNN, GBDT, RF, DT, AdaBoost and GaussianNB, we evaluate all possible combinations of candidate features via grid search. Each combination is scored based on its performance in the given models. It allows us to determine the optimal combination of features that are most predictive of stroke risk.</p>
<p>The two-step approach provides a rigorous feature selection process by multiple machine learning methods. The first step reduces the number of features based on statistical tests of relevance. The second step further refines the features by evaluating prediction performance in representative machine learning models. This ensures that the most informative and generalizable features have been selected for predicting stroke risk.</p>
<sec id="s2-4-1">
<title>2.4.1 The first step of feature selection</title>
<p>We employ four feature selection methods, including chi-square test, Pearson correlation, Lasso and elastic net, to assess the correlation between features and disease risk from different perspectives. The chi-square test and Pearson correlation prefer to filter out features, which have the advantage of high computational efficiency while not being prone to overfitting. However, their over-reliance on filter thresholds may overlook many important features. On the other hand, Lasso and elastic net are embedded feature selection methods that select salient features while accounting for feature correlations by calculating feature weights. Therefore, we combined the filter and embedded methods to comprehensively screen for the important features related to stroke risk factors. For details, we provide brief introductions to the chi-square test, Pearson correlation, Lasso, elastic net.</p>
<p>
<bold>Chi-square test</bold> (<xref ref-type="bibr" rid="B36">Sharpe, 2015</xref>). The chi-square test is used to check the correlation of the independent variable with the dependent variable. We use the chi-square test to delete the features with small changes. The formula of chi-square test is described as <xref ref-type="disp-formula" rid="e2">Equation 2</xref>:<disp-formula id="e2">
<mml:math id="m2">
<mml:mrow>
<mml:msup>
<mml:mi>&#x3c7;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2211;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>where <inline-formula id="inf1">
<mml:math id="m3">
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the observed value of the feature, and <inline-formula id="inf2">
<mml:math id="m4">
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the expected value of the feature. The assumption of chi-square test is that features are independent. The larger result of the chi-square test means the higher correlation between features.</p>
<p>
<bold>Pearson correlation</bold> (<xref ref-type="bibr" rid="B8">Cohen et al., 2009</xref>). We use Pearson correlation coefficient to measure the linear correlation between features and disease risk. When all the features have been scaled to (0,1), the most important feature should have the highest coefficient, and the irrelevant feature should have a coefficient whose value is close to zero. The Pearson correlation coefficient can be determined by <xref ref-type="disp-formula" rid="e3">Equation 3</xref>:<disp-formula id="e3">
<mml:math id="m5">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c1;</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mtext>cov</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mi>&#x3c3;</mml:mi>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mi>E</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:msqrt>
<mml:msqrt>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mi>E</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>where <inline-formula id="inf3">
<mml:math id="m6">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf4">
<mml:math id="m7">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represent the different feature, respectively. <inline-formula id="inf5">
<mml:math id="m8">
<mml:mrow>
<mml:mtext>cov</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> denotes the covariance of <inline-formula id="inf6">
<mml:math id="m9">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf7">
<mml:math id="m10">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula id="inf8">
<mml:math id="m11">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the standard deviation of <inline-formula id="inf9">
<mml:math id="m12">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf10">
<mml:math id="m13">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the standard deviation of <inline-formula id="inf11">
<mml:math id="m14">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>
<bold>Lasso</bold> (<xref ref-type="bibr" rid="B26">Nusinovici et al., 2020</xref>). Lasso built upon logistic regression analysis techniques, serves to select the most crucial features while reducing model complexity through the shrinkage of feature weights. Specifically, lasso introduces <inline-formula id="inf12">
<mml:math id="m15">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> regularization into the loss function of a linear regression model, minimizing the mean squared error between predicted values and actual observations. The Lasso loss function is given by <xref ref-type="disp-formula" rid="e4">Equation 4</xref>:<disp-formula id="e4">
<mml:math id="m16">
<mml:mrow>
<mml:munder>
<mml:mi>min</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:munder>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>p</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>p</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>where <inline-formula id="inf13">
<mml:math id="m17">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>.</mml:mo>
<mml:mo>.</mml:mo>
<mml:mo>.</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> denote coefficients that we need to compute. <inline-formula id="inf14">
<mml:math id="m18">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the label that takes a value of 0 or 1. <inline-formula id="inf15">
<mml:math id="m19">
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is a positive tuning parameter used to balance the loss term and penalty term. <inline-formula id="inf16">
<mml:math id="m20">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the value of the <inline-formula id="inf17">
<mml:math id="m21">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-th feature of the <inline-formula id="inf18">
<mml:math id="m22">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-th sample.</p>
<p>
<bold>Elastic net</bold> (<xref ref-type="bibr" rid="B42">Zhang et al., 2017</xref>). Since Lasso regression sometimes performs poorly in inter-correlated features, the elastic net was proposed to overcome this limitation. Elastic net regularization combines <inline-formula id="inf19">
<mml:math id="m23">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> penalty with <inline-formula id="inf20">
<mml:math id="m24">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> penalty together to select better relevant features simultaneously. The elastic net is defined as <xref ref-type="disp-formula" rid="e5">Equation 5</xref>:<disp-formula id="e5">
<mml:math id="m25">
<mml:mrow>
<mml:munder>
<mml:mi>min</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:munder>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>p</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>p</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>p</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msubsup>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>j</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>where <inline-formula id="inf21">
<mml:math id="m26">
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> used to balance the <inline-formula id="inf22">
<mml:math id="m27">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> penalty and <inline-formula id="inf23">
<mml:math id="m28">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> penalty. The <inline-formula id="inf24">
<mml:math id="m29">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> penalty of regularization term is defined as <inline-formula id="inf25">
<mml:math id="m30">
<mml:mrow>
<mml:mi>&#x3c6;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>p</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msubsup>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>j</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, which is known as Ridge regression.</p>
</sec>
<sec id="s2-4-2">
<title>2.4.2 The second step of feature selection</title>
<p>Although we have selected the important risk factors at the first step feature selection, the filter and embedded methods have the shortcomings of excessive threshold reliance and simply correlation consideration. To capture the deep correlation between features, we use seven machine learning methods to conduct the second step feature selection, which traverse all candidate features combinations based on the result of first step feature selection.</p>
<p>The candidate feature combinations consist of all possible permutations of the features selected during the process of feature selection. Assume there are <inline-formula id="inf26">
<mml:math id="m31">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> features that are selected. Then totally there are <inline-formula id="inf27">
<mml:math id="m32">
<mml:mrow>
<mml:msup>
<mml:mn>2</mml:mn>
<mml:mi>n</mml:mi>
</mml:msup>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> candidate feature combinations. Next, we traverse all candidate combinations using seven machine learning methods and select the optimal feature combination based on the classification performance of these seven different classifiers. As we know, machine learning methods are based on specific theoretical assumptions. Therefore, employing different machine learning methods can increase the diversity of feature selection. Brief introductions of the seven machine learning methods are presented in <xref ref-type="table" rid="T3">Table 3</xref>. These algorithms have their own advantages and disadvantages, allowing us to thoroughly consider different scenarios in the feature selection process.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>The overview of seven machine learning methods.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Methods</th>
<th align="center">Theory</th>
<th align="center">Advantages</th>
<th align="center">Disadvantages</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">SVM (<xref ref-type="bibr" rid="B25">Noble, 2006</xref>)</td>
<td align="center">Find the optimal hyperplane</td>
<td align="center">Handling the interaction of nonlinear features</td>
<td align="center">Difficulty in selecting the kernel function</td>
</tr>
<tr>
<td align="center">KNN (<xref ref-type="bibr" rid="B9">Cunningham and Delany, 2021</xref>)</td>
<td align="center">Find the k nearest neighbors</td>
<td align="center">No assumptions, insensitive to outliers</td>
<td align="center">Cannot handle imbalanced data</td>
</tr>
<tr>
<td align="center">DT (<xref ref-type="bibr" rid="B4">Al Snousy et al., 2011</xref>)</td>
<td align="center">Divides feature subspaces</td>
<td align="center">Handle Boolean and numeric data simultaneously</td>
<td align="center">Prone to overfitting</td>
</tr>
<tr>
<td align="center">GBDT (<xref ref-type="bibr" rid="B43">Zhou et al., 2020</xref>)</td>
<td align="center">Iteratively train decision trees</td>
<td align="center">Strong interpretability</td>
<td align="center">Difficulty tuning parameters</td>
</tr>
<tr>
<td align="center">RF (<xref ref-type="bibr" rid="B10">Cutler et al., 2012</xref>)</td>
<td align="center">Integrate several DTs</td>
<td align="center">Strong generalization ability</td>
<td align="center">Poor interpretability</td>
</tr>
<tr>
<td align="center">AdaBoost (<xref ref-type="bibr" rid="B35">Schapire, 2013</xref>)</td>
<td align="center">Integrated learning strategy</td>
<td align="center">Prevent overfitting</td>
<td align="center">Sensitive to outlier</td>
</tr>
<tr>
<td align="center">GaussianNB (<xref ref-type="bibr" rid="B20">Liu et al., 2023</xref>)</td>
<td align="center">Based on independence assumption</td>
<td align="center">No need to tune parameters</td>
<td align="center">Not suitable for high-dimensional data</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s2-5">
<title>2.5 Deep fusion network</title>
<p>The complexity and diversity of stroke data require predicting stroke risk from multiple perspectives to enhance model robustness. Common predictive models, such as the Transformer, exhibit complex structures and excel at adapting to high-dimensional data, thus improving prediction performance. However, it often suffers from overfitting issues when dealing with small-scale datasets. In contrast, the FCN model has a simple structure and fast training speed, yielding exceptional performance on small-scale datasets. We utilize the advantages of both above predictive models and propose a deep fusion network method that can provide accurate the stroke risk prediction. As shown in <xref ref-type="fig" rid="F2">Figure 2</xref>, deep fusion network integrates the Transform and the FCN, in which the dependencies between stroke risk factors are captured by the attention mechanism of the Transformer, and the complex nonlinear relationship is fitted by deep network structure of the FCN.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>The structure of the proposed deep fusion network.</p>
</caption>
<graphic xlink:href="fphys-15-1357123-g002.tif"/>
</fig>
<p>
<bold>Transformer</bold> (<xref ref-type="bibr" rid="B38">Vaswani et al., 2017</xref>). Due to the powerful representation ability, Transformer can realize the outstanding performance in prediction tasks which is based on the self-attention mechanism. As observed in <xref ref-type="fig" rid="F2">Figure 2</xref>, given an input <inline-formula id="inf28">
<mml:math id="m33">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mo>&#x211d;</mml:mo>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf29">
<mml:math id="m34">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the number of patients (or patches) and <inline-formula id="inf30">
<mml:math id="m35">
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the embedded feature dimension for every patient. The self-attention mechanism can be defined as <xref ref-type="disp-formula" rid="e6">Equation 6</xref>:<disp-formula id="e6">
<mml:math id="m36">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>K</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>softmax</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:msup>
<mml:mi>K</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
</mml:mrow>
<mml:msqrt>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:msqrt>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>V</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>where <inline-formula id="inf31">
<mml:math id="m37">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the input matrix embedding dimension. The matrix <inline-formula id="inf32">
<mml:math id="m38">
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf33">
<mml:math id="m39">
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf34">
<mml:math id="m40">
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> can be computed by the input matrix and linear transformation matrix <inline-formula id="inf35">
<mml:math id="m41">
<mml:mrow>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mi>Q</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf36">
<mml:math id="m42">
<mml:mrow>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mi>K</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf37">
<mml:math id="m43">
<mml:mrow>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mi>V</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, respectively. Then, we can get the value of <inline-formula id="inf38">
<mml:math id="m44">
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf39">
<mml:math id="m45">
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf40">
<mml:math id="m46">
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> by computing <inline-formula id="inf41">
<mml:math id="m47">
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>X</mml:mi>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mi>Q</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf42">
<mml:math id="m48">
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>X</mml:mi>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mi>K</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf43">
<mml:math id="m49">
<mml:mrow>
<mml:mi>V</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>X</mml:mi>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mi>V</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf44">
<mml:math id="m50">
<mml:mrow>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mi>Q</mml:mi>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mo>&#x211d;</mml:mo>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>q</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf45">
<mml:math id="m51">
<mml:mrow>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mi>K</mml:mi>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mo>&#x211d;</mml:mo>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>q</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf46">
<mml:math id="m52">
<mml:mrow>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mi>V</mml:mi>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mo>&#x211d;</mml:mo>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>q</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf47">
<mml:math id="m53">
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes the linear mapping dimension.</p>
<p>
<bold>Fully connected neural network.</bold> The FCN, also known as a Multilayer Perceptron (MLP), is a widely used artificial neural network structure in medical data analysis. It offers the advantages of fast training speed and robust modeling capabilities as the network depth increases. The stroke risk prediction model we designed includes one input layer, one hidden layer and one output layer. The calculation formula of each layer of network is defined by <xref ref-type="disp-formula" rid="e7">Equation 7</xref>:<disp-formula id="e7">
<mml:math id="m54">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>W</mml:mi>
<mml:mi>x</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>where <inline-formula id="inf48">
<mml:math id="m55">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes the ReLU activation function. <inline-formula id="inf49">
<mml:math id="m56">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the input of the neuronal node, <inline-formula id="inf50">
<mml:math id="m57">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the output of the neuronal node. <inline-formula id="inf51">
<mml:math id="m58">
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf52">
<mml:math id="m59">
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denote weight and bias, respectively, which are learnable parameters.</p>
<p>
<bold>Attention mechanism.</bold> In the stroke risk prediction task, the Transformer and the FCN extract clinical features at different levels and make distinct contributions to the prediction. Therefore, we introduce an attention mechanism to adaptively learn the importance of latent embeddings. Specifically, for the feature <inline-formula id="inf53">
<mml:math id="m60">
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> extracted by Transformer, we apply a non-linear transformation and employ the shared attention vector <inline-formula id="inf54">
<mml:math id="m61">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> to obtain the attention coefficient <inline-formula id="inf55">
<mml:math id="m62">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, namely, <xref ref-type="disp-formula" rid="e8">Equation 8</xref>:<disp-formula id="e8">
<mml:math id="m63">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>softmax</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>W</mml:mi>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>where <inline-formula id="inf56">
<mml:math id="m64">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes the tanh activation, <inline-formula id="inf57">
<mml:math id="m65">
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes a trainable weight matrix, and <inline-formula id="inf58">
<mml:math id="m66">
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes a bias vector. Similarly, we can calculate attention coefficients <inline-formula id="inf59">
<mml:math id="m67">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> for the features <inline-formula id="inf60">
<mml:math id="m68">
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> extracted by the FCN. We combine these embeddings to obtain the final embedding <inline-formula id="inf61">
<mml:math id="m69">
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <xref ref-type="disp-formula" rid="e9">Equation 9</xref>:<disp-formula id="e9">
<mml:math id="m70">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x22c5;</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mo>&#x22c5;</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>where <inline-formula id="inf62">
<mml:math id="m71">
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes the single linear layer.</p>
</sec>
<sec id="s2-6">
<title>2.6 Evaluation metrics</title>
<p>Here we employ four evaluation metrics to assess the predictive performance of the model, including micro precision, micro F1-score, macro precision, and Cohen&#x2019;s Kappa coefficient (<xref ref-type="bibr" rid="B40">Younas et al., 2023</xref>). The definitions of these metrics are given as follows.</p>
<p>The micro average approach amalgamates performance measures across all samples. Specifically, for each class <inline-formula id="inf63">
<mml:math id="m72">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> within the set <inline-formula id="inf64">
<mml:math id="m73">
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>.</mml:mo>
<mml:mo>.</mml:mo>
<mml:mo>.</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf65">
<mml:math id="m74">
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes the total number of classes, a dedicated confusion matrix is constructed. In this context, the <inline-formula id="inf66">
<mml:math id="m75">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-th matrix designates the <inline-formula id="inf67">
<mml:math id="m76">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> class as the positive class, while considering the remaining classes <inline-formula id="inf68">
<mml:math id="m77">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> with <inline-formula id="inf69">
<mml:math id="m78">
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x2260;</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> as the negative classes. The micro precision and micro F1-score are computed by <xref ref-type="disp-formula" rid="e10">Equations 10</xref> and <xref ref-type="disp-formula" rid="e11">11</xref>:<disp-formula id="e10">
<mml:math id="m79">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mi>T</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mi>T</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(10)</label>
</disp-formula>
<disp-formula id="e11">
<mml:math id="m80">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:msub>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mi>T</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mi>T</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mi>F</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mi>F</mml:mi>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(11)</label>
</disp-formula>where TP represents the number of positive samples correctly predicted to be positive samples, FP represents the number of negative samples incorrectly predicted to be positive samples, FN represents the number of positive samples incorrectly predicted to be negative samples.</p>
<p>Micro average tends to provide misleading results in the case of imbalanced data, as it doesn&#x2019;t take the predictive performance of each specific class into account. In contrast, macro average computes averages through the individual performance of each class. The macro precision is defined as <xref ref-type="disp-formula" rid="e12">Equation 12</xref>:<disp-formula id="e12">
<mml:math id="m81">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>F</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(12)</label>
</disp-formula>
</p>
<p>Cohen&#x2019;s Kappa Coefficient is employed for assessing performance in situations of imbalanced class distribution, which is denoted by <xref ref-type="disp-formula" rid="e13">Equation 13</xref>:<disp-formula id="e13">
<mml:math id="m82">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>e</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>e</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(13)</label>
</disp-formula>where <inline-formula id="inf70">
<mml:math id="m83">
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the overall model accuracy, and <inline-formula id="inf71">
<mml:math id="m84">
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>e</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the agreement expected by chance between the model&#x2019;s predictions and the actual class values (<xref ref-type="bibr" rid="B24">McHugh, 2012</xref>).</p>
</sec>
<sec id="s2-7">
<title>2.7 Implementation details</title>
<p>The stroke risk prediction model was built and trained using the PyTorch. Experiments were conducted on a PC with Intel(R) Xeon(R) Gold 6258R CPU @ 2.70 GHz and NVIDIA QuADro GV100 GPU. We trained the model with the Adam optimizer (<xref ref-type="bibr" rid="B17">Kingma and Ba, 2014</xref>) with default parameters and a fixed learning rate of 0.001. And we randomly select 80% of the samples from whole dataset for training, and the remaining 20% for testing. The maximum number of epochs employed for training is 100. The datasets and source codes are publicly available on GitHub: <ext-link ext-link-type="uri" xlink:href="https://github.com/zhangdaoliang/SRPNet">https://github.com/zhangdaoliang/SRPNet</ext-link>.</p>
</sec>
</sec>
<sec sec-type="results|discussion" id="s3">
<title>3 Results and discussion</title>
<sec id="s3-1">
<title>3.1 Two-level feature selection results</title>
<p>We utilized a dataset from the CSDC database, consisting of 862,244 samples, with each sample originally having 26 distinct features. The proposed two-level feature selection method was used to screen out significant stroke features, which has a positive effect on improving the performance of the prediction model. In the first step of feature selection, we employed Lasso, elastic net, chi-square test and Pearson correlation methods for the initial screening of stroke-related factors. Here, we consider using <inline-formula id="inf72">
<mml:math id="m85">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.5</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> for the elastic net. <xref ref-type="fig" rid="F3">Figure 3</xref> shows that the impact of different parameters contained in these methods on the feature selection results. We can observe in <xref ref-type="fig" rid="F3">Figures 3A, B</xref> the paths of regression coefficient changes based on Lasso and elastic net, with each curve corresponding to one feature variable. <xref ref-type="fig" rid="F3">Figures 3C, D</xref> demonstrate the correlation of each feature with stroke. It is worth noting that we tend to select features with higher scores and <inline-formula id="inf73">
<mml:math id="m86">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>0.05</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> in the chi-square test (<xref ref-type="bibr" rid="B32">Pandis, 2016</xref>). According to <xref ref-type="fig" rid="F3">Figure 3</xref>, Lasso, elastic net, chi-square test and Pearson correlation methods select 13, 15, 8 and 12 features, respectively.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Feature selection results with different parameters in four feature selection methods. <bold>(A)</bold> Lasso. <bold>(B)</bold> Elastic net. <bold>(C)</bold> Chi-square test. <bold>(D)</bold> Pearson correlation.</p>
</caption>
<graphic xlink:href="fphys-15-1357123-g003.tif"/>
</fig>
<p>The specific feature selection results of each method are shown in <xref ref-type="table" rid="T4">Table 4</xref>. Subsequently, we took the union of features selected by the four methods as the robust candidate feature set, which includes 16 features, i.e., AG, Gender, Smoking, MS, Occupation, ES, HS, HYP, AF, LDL-C, Diabetes, LE, Overweight, FHS/HYP/CHD, PLT, and PLH. The receiver operating characteristic (ROC) curves (<xref ref-type="bibr" rid="B12">Fan et al., 2006</xref>) corresponding to different feature sets are shown in <xref ref-type="fig" rid="F4">Figure 4</xref>. We find that using the candidate feature set achieves better prediction results than features selected by individual methods. It illustrates that the first step of feature selection is of great significance for stroke risk diagnosis.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Feature selection results of four methods.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Methods</th>
<th align="center">Features</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Lasso</td>
<td align="center">AG, Gender, Occupation, HS, HYP, AF, Smoking, LDL-C, Diabetes, LE, Overweight, FHS/HYP/CHD</td>
</tr>
<tr>
<td align="center">Elastic net</td>
<td align="center">AG, Gender, Occupation, ES, HS, HYP, AF, Smoking, LDL-C, Diabetes, LE, Overweight, FHS/HYP/CHD, PLT, PLH</td>
</tr>
<tr>
<td align="center">Chi-square Test</td>
<td align="center">HS, HYP, AF, LDL-C, Diabetes, LE, Overweight, FHS/HYP/CHD</td>
</tr>
<tr>
<td align="center">Pearson correlation</td>
<td align="center">AG, ES, HS, HYP, AF, Smoking, LDL-C, Diabetes, LE, Overweight, FHS/HYP/CHD, MS</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>ROC curves corresponding to the feature sets selected by the four methods and the candidate feature set. <bold>(A)</bold> Lasso. <bold>(B)</bold> Elastic net. <bold>(C)</bold> Chi-square test. <bold>(D)</bold> Pearson correlation. <bold>(E)</bold> The robust candidate feature set.</p>
</caption>
<graphic xlink:href="fphys-15-1357123-g004.tif"/>
</fig>
<p>In the second step of feature selection, we eliminate risk factors with strong correlation between features. Based on the results of the first step of feature selection, we iterate through all candidate feature combinations. All feature combinations are evaluated under different machine learning methods as classifiers. The optimal feature combinations for different number of features are determined with respect to the evaluation results. <xref ref-type="fig" rid="F5">Figure 5</xref> shows the performance of the method with different numbers of feature variables. We see that as the number of features increases, the micro precision of most machine learning methods gradually improves and tends to stabilize. However, the performance of the DT and AdaBoost methods decreases significantly when the number of features is 9 and 15 respectively. When the number of features reaches 12, all seven machine learning methods overall achieve the best performance. Finally, we obtained risk factors that are highly relevant to stroke patients and have no redundant information among features, including Smoking, Occupation, ES, HS, HYP, AF, LDL-C, Diabetes, LE, Overweight, FHS/HYP/CHD, and PLT. It is worth noting that traditional methods consider age and gender to be strongly correlated with stroke risk (<xref ref-type="bibr" rid="B14">Howard et al., 2023</xref>; <xref ref-type="bibr" rid="B29">Ospel et al., 2023</xref>). However, two-level feature selection has removed them due to their redundancy with occupation and other risk factors. In contrast, the PLT features reflecting the climate of the patient&#x2019;s location are preserved, and it has been confirmed that low temperatures are associated with an increased risk of stroke (<xref ref-type="bibr" rid="B7">Chen et al., 2013</xref>). This indicates that SRPNet could provide new insights for future risk screening.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>The prediction performance micro precision in seven machine learning methods with different number of features.</p>
</caption>
<graphic xlink:href="fphys-15-1357123-g005.tif"/>
</fig>
</sec>
<sec id="s3-2">
<title>3.2 Stroke risk prediction results</title>
<p>In this section, we validate the effectiveness of the SRPNet model on the CSDC dataset. Decision tree C5.0 (C5.0) (<xref ref-type="bibr" rid="B3">Ahmadi et al., 2018</xref>), random forests (RF) (<xref ref-type="bibr" rid="B6">Breiman, 2001</xref>),FCN, one-dimensional convolutional neural network (CNN), long short-term memory network (LSTM) and Transformer are used as comparison methods to predict stroke risk. <xref ref-type="table" rid="T5">Table 5</xref> shows the prediction performance of the seven methods on the original CSDC data (all features) and the data after two-level feature selection (selected features). We can find that SRPNet model obtains the best prediction results in terms of the four evaluation metrics. The performance of all predictors after two-level feature selection is significantly better than their performance when using all features. This demonstrates that the two-level feature selection can effectively filter weak and redundant information, thus improving the results of all predictors. On the selected feature data, SRPNet outperformes FCN and Transformer by approximately 1.4%, 1.4%, 12% and 3.2% on metrics micro F1-score, micro precision, macro precision, Cohen&#x2019;s Kappa coefficient. This reflects that deep fusion network can better explore potential relationships between risk factors. In summary, the proposed SRPNet model is reasonable and effective for predicting stroke risk.</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Comparison of stroke risk prediction results for the seven methods.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="center">Methods</th>
<th colspan="4" align="center">Selected features</th>
<th colspan="4" align="center">All features</th>
</tr>
<tr>
<th align="center">Micro F1-score</th>
<th align="center">Micro precision</th>
<th align="center">Macro precision</th>
<th align="center">Cohen&#x2019;s Kappa coefficient</th>
<th align="center">Micro F1-score</th>
<th align="center">Micro precision</th>
<th align="center">Macro precision</th>
<th align="center">Cohen&#x2019;s Kappa coefficient</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">C5.0</td>
<td align="center">0.9470</td>
<td align="center">0.9470</td>
<td align="center">0.7369</td>
<td align="center">0.8828</td>
<td align="center">0.9149</td>
<td align="center">0.9149</td>
<td align="center">0.7288</td>
<td align="center">0.8068</td>
</tr>
<tr>
<td align="center">RF</td>
<td align="center">0.9478</td>
<td align="center">0.9478</td>
<td align="center">0.7672</td>
<td align="center">0.8853</td>
<td align="center">0.9167</td>
<td align="center">0.9167</td>
<td align="center">0.7170</td>
<td align="center">0.8172</td>
</tr>
<tr>
<td align="center">FCN</td>
<td align="center">0.9257</td>
<td align="center">0.9257</td>
<td align="center">0.7119</td>
<td align="center">0.8335</td>
<td align="center">0.8906</td>
<td align="center">0.8906</td>
<td align="center">0.6811</td>
<td align="center">0.7449</td>
</tr>
<tr>
<td align="center">CNN</td>
<td align="center">0.9421</td>
<td align="center">0.9422</td>
<td align="center">0.7316</td>
<td align="center">0.8716</td>
<td align="center">0.9371</td>
<td align="center">0.9371</td>
<td align="center">0.7310</td>
<td align="center">0.8591</td>
</tr>
<tr>
<td align="center">LSTM</td>
<td align="center">0.9424</td>
<td align="center">0.9424</td>
<td align="center">0.8144</td>
<td align="center">0.8723</td>
<td align="center">0.9399</td>
<td align="center">0.9399</td>
<td align="center">0.7328</td>
<td align="center">0.8654</td>
</tr>
<tr>
<td align="center">Transformer</td>
<td align="center">0.9480</td>
<td align="center">0.9480</td>
<td align="center">0.7449</td>
<td align="center">0.8846</td>
<td align="center">0.9198</td>
<td align="center">0.9198</td>
<td align="center">0.7396</td>
<td align="center">0.8176</td>
</tr>
<tr>
<td align="center">SRPNet</td>
<td align="center">
<bold>0.9618</bold>
</td>
<td align="center">
<bold>0.9618</bold>
</td>
<td align="center">
<bold>0.8642</bold>
</td>
<td align="center">
<bold>0.9165</bold>
</td>
<td align="center">
<bold>0.9511</bold>
</td>
<td align="center">
<bold>0.9511</bold>
</td>
<td align="center">
<bold>0.8126</bold>
</td>
<td align="center">
<bold>0.8920</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Note: The best experimental results are highlighted in bold.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Furthermore, to make the results more convincing, we evaluated six predictors on in-house data from affiliated hospital of Jining Medical University. The experimental results are recorded in <xref ref-type="table" rid="T6">Table 6</xref>. We can draw the similar conclusion that the proposed SRPNet model is an ideal and effective prediction tool of stroke risk. To explore the features that play a dominant role in precise classification, we removed each feature and obtained the prediction results for stroke risk. We found that after removing the hypertension (HYP) feature resulted in micro F1-score, micro precision, macro precision, and Cohen&#x2019;s Kappa coefficient of 0.7, 0.7, 0.83, and 0.28 respectively, which had the greatest impact on stroke prediction performance. Secondly, gender and age also significantly influenced stroke classification, while they are identified as redundant features in the CSDC dataset. The reason is that the analysis conducted on the CSDC dataset involves complex stroke risk prediction, focusing on differences between multiple risk levels, whereas the in-house dataset only focuses on whether someone has a stroke, conducting a simple stroke prediction analysis. Understanding these risk factors can assist doctors in making quick and accurate stroke diagnoses.</p>
<table-wrap id="T6" position="float">
<label>TABLE 6</label>
<caption>
<p>Prediction results based on our in-house dataset.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Methods</th>
<th align="center">Micro F1-score</th>
<th align="center">Micro precision</th>
<th align="center">Macro precision</th>
<th align="center">Cohen&#x2019;s Kappa coefficient</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">C5.0</td>
<td align="center">0.9000</td>
<td align="center">0.9000</td>
<td align="center">0.8750</td>
<td align="center">0.7826</td>
</tr>
<tr>
<td align="center">RF</td>
<td align="center">0.9000</td>
<td align="center">0.9000</td>
<td align="center">0.9285</td>
<td align="center">0.7826</td>
</tr>
<tr>
<td align="center">FCN</td>
<td align="center">0.9953</td>
<td align="center">0.9953</td>
<td align="center">0.9933</td>
<td align="center">0.9917</td>
</tr>
<tr>
<td align="center">CNN</td>
<td align="center">0.8000</td>
<td align="center">0.8000</td>
<td align="center">0.8571</td>
<td align="center">0.6000</td>
</tr>
<tr>
<td align="center">LSTM</td>
<td align="center">0.8000</td>
<td align="center">0.8000</td>
<td align="center">0.8000</td>
<td align="center">0.6000</td>
</tr>
<tr>
<td align="center">Transformer</td>
<td align="center">0.9000</td>
<td align="center">0.9000</td>
<td align="center">0.9283</td>
<td align="center">0.7826</td>
</tr>
<tr>
<td align="center">SRPNet</td>
<td align="center">
<bold>0.9978</bold>
</td>
<td align="center">
<bold>0.9978</bold>
</td>
<td align="center">
<bold>0.9954</bold>
</td>
<td align="center">
<bold>0.9929</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Note: The best experimental results are highlighted in bold.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>To further evaluate the superiority of SRPNet, we visualize the confusion matrices obtained by the six methods on the CSDC dataset and the in-house dataset in <xref ref-type="fig" rid="F6">Figures 6</xref>, <xref ref-type="fig" rid="F7">7</xref>, where the columns and rows are the predicted labels and true labels, respectively. It shows that compared to other methods, The SRPNet method wins in all categories in terms of prediction accuracy. Additionally, we discover that the history of stroke (HS) feature and the hypertension (HYP) feature significantly enhance the ability of almost all algorithms in <xref ref-type="fig" rid="F6">Figure 6</xref> to detect stroke effectively.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>The confusion matrices for the six methods on the CSDC dataset. <bold>(A)</bold> C5.0. <bold>(B)</bold> FCN. <bold>(C)</bold> CNN. <bold>(D)</bold> LSTM. <bold>(E)</bold> Transformer. <bold>(F)</bold> SRPNet.</p>
</caption>
<graphic xlink:href="fphys-15-1357123-g006.tif"/>
</fig>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>The confusion matrices for the six methods on the in-house dataset. <bold>(A)</bold> C5.0. <bold>(B)</bold> FCN. <bold>(C)</bold> CNN. <bold>(D)</bold> LSTM. <bold>(E)</bold> Transformer. <bold>(F)</bold> SRPNet.</p>
</caption>
<graphic xlink:href="fphys-15-1357123-g007.tif"/>
</fig>
</sec>
</sec>
<sec sec-type="conclusion" id="s4">
<title>4 Conclusion</title>
<p>In this paper, a novel prediction model based on two-level feature selection and deep fusion network is proposed for stroke risk prediction. Compared with traditional feature selection methods, the proposed two-level feature selection method not only focuses on the importance of individual these features, but also eliminates redundant information among important features. Furthermore, the proposed deep fusion network harnesses Transformer and fully connected networks to capture feature dependencies and model the non-linear relationships among features, respectively. Experimental results on the CSDC database and in-house dataset demonstrate that our proposed prediction model outperforms other representative methods. This prediction model can rapidly identify high-quality stroke risk factors and improve the accuracy of stroke prediction for patients, thereby effectively assisting doctors in formulating rational diagnosis and treatment plans.</p>
<p>The features included in the CSDC database and in-house dataset are limited. In the future, we will collect more clinical indicator features related to stroke for model training and testing. And we will also work on applying the proposed model to predict other diseases, demonstrating its generalizability. It&#x27;s worth noting that researchers have the flexibility to substitute the feature selection method used in SRPNet with other methods that are frequently applied in the context of medical information, tailored to their specific requirements.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>The datasets presented in this study can be found in online repositories. The names of the repository/repositories and accession number(s) can be found below: <ext-link ext-link-type="uri" xlink:href="https://github.com/zhangdaoliang/SRPNet">https://github.com/zhangdaoliang/SRPNet</ext-link>.</p>
</sec>
<sec sec-type="author-contributions" id="s6">
<title>Author contributions</title>
<p>DZ: Conceptualization, Data curation, Methodology, Software, Validation, Writing&#x2013;original draft, Writing&#x2013;review and editing. NY: Conceptualization, Investigation, Validation, Writing&#x2013;review and editing. XY: Conceptualization, Data curation, Validation, Writing&#x2013;review and editing. YD: Funding acquisition, Validation, Writing&#x2013;review and editing. Z-PL: Conceptualization, Funding acquisition, Supervision, Validation, Writing&#x2013;review and editing. RG: Conceptualization, Funding acquisition, Project administration, Supervision, Validation, Writing&#x2013;review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s7">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. This research was supported by the National Natural Science Foundation of China (NSFC) (Grant Nos U1806202, 62373216), the Fundamental Research Funds for the Central Universities (2022JC008), and the Program of Qilu Young Scholars of Shandong University.</p>
</sec>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
<p>The author(s) declared that they were an editorial board member of Frontiers, at the time of submission. This had no impact on the peer review process and the final decision.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Abraham</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Malik</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Yonova-Doing</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Salim</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Danesh</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Genomic risk score offers predictive performance comparable to clinical risk factors for ischaemic stroke</article-title>. <source>Nat. Commun.</source> <volume>10</volume>, <fpage>5819</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-019-13848-1</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Abraham</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Rutten-Jacobs</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Inouye</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Risk prediction using polygenic risk scores for prevention of stroke and other cardiovascular diseases</article-title>. <source>Stroke</source> <volume>52</volume>, <fpage>2983</fpage>&#x2013;<lpage>2991</lpage>. <pub-id pub-id-type="doi">10.1161/STROKEAHA.120.032619</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ahmadi</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Weckman</surname>
<given-names>G. R.</given-names>
</name>
<name>
<surname>Masel</surname>
<given-names>D. T.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Decision making model to predict presence of coronary artery disease using neural network and C5. 0 decision tree</article-title>. <source>J. Ambient Intell. Humaniz. Comput.</source> <volume>9</volume>, <fpage>999</fpage>&#x2013;<lpage>1011</lpage>. <pub-id pub-id-type="doi">10.1007/s12652-017-0499-z</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Al Snousy</surname>
<given-names>M. B.</given-names>
</name>
<name>
<surname>El-Deeb</surname>
<given-names>H. M.</given-names>
</name>
<name>
<surname>Badran</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Al Khlil</surname>
<given-names>I. A.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Suite of decision tree-based classification algorithms on cancer gene expression data</article-title>. <source>Egypt. Inf. J.</source> <volume>12</volume>, <fpage>73</fpage>&#x2013;<lpage>82</lpage>. <pub-id pub-id-type="doi">10.1016/j.eij.2011.04.003</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Arafa</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kokubo</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sheerah</surname>
<given-names>H. A.</given-names>
</name>
<name>
<surname>Sakai</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Watanabe</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Developing a stroke risk prediction model using cardiovascular risk factors: the Suita Study</article-title>. <source>Cerebrovasc. Dis.</source> <volume>51</volume>, <fpage>323</fpage>&#x2013;<lpage>330</lpage>. <pub-id pub-id-type="doi">10.1159/000520100</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Breiman</surname>
<given-names>L. J. M. L.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Random forests</article-title>. <source>Mach. Learn.</source> <volume>45</volume>, <fpage>5</fpage>&#x2013;<lpage>32</lpage>. <pub-id pub-id-type="doi">10.1023/a:1010933404324</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Meng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Thach</surname>
<given-names>T. Q.</given-names>
</name>
<name>
<surname>Wong</surname>
<given-names>C.-M.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Both low and high temperature may increase the risk of stroke mortality</article-title>. <source>Neurology</source> <volume>81</volume>, <fpage>1064</fpage>&#x2013;<lpage>1070</lpage>. <pub-id pub-id-type="doi">10.1212/WNL.0b013e3182a4a43c</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cohen</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Benesty</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Benesty</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2009</year>). <article-title>Pearson correlation coefficient</article-title>. <source>Noise Reduct. speech Process.</source>, <fpage>1</fpage>&#x2013;<lpage>4</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-642-00296-0_5</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cunningham</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Delany</surname>
<given-names>S. J.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>K-nearest neighbour classifiers-a tutorial</article-title>. <source>ACM Comput. Surv. (CSUR)</source> <volume>54</volume>, <fpage>1</fpage>&#x2013;<lpage>25</lpage>. <pub-id pub-id-type="doi">10.1145/3459665</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Cutler</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Cutler</surname>
<given-names>D. R.</given-names>
</name>
<name>
<surname>Stevens</surname>
<given-names>J. R.</given-names>
</name>
</person-group> (<year>2012</year>). <source>Random forests</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Zhang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>Y. Q.</given-names>
</name>
</person-group> (<publisher-loc>Springer, New York</publisher-loc>: <publisher-name>Ensemble Machine Learning</publisher-name>), <fpage>157</fpage>&#x2013;<lpage>175</lpage>. <pub-id pub-id-type="doi">10.1007/978-1-4419-9326-7_5</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dritsas</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Trigka</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Stroke risk prediction with machine learning techniques</article-title>. <source>Sensors</source> <volume>22</volume>, <fpage>4670</fpage>. <pub-id pub-id-type="doi">10.3390/s22134670</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Upadhye</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Worster</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Understanding receiver operating characteristic (ROC) curves</article-title>. <source>Can. J. Emerg. Med.</source> <volume>8</volume>, <fpage>19</fpage>&#x2013;<lpage>20</lpage>. <pub-id pub-id-type="doi">10.1017/s1481803500013336</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guan</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Clay</surname>
<given-names>S. J.</given-names>
</name>
<name>
<surname>Sloan</surname>
<given-names>G. J.</given-names>
</name>
<name>
<surname>Pretlow</surname>
<given-names>L. G.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Effects of barometric pressure and temperature on acute ischemic stroke hospitalization in Augusta, GA</article-title>. <source>Transl. Stroke Res.</source> <volume>10</volume>, <fpage>259</fpage>&#x2013;<lpage>264</lpage>. <pub-id pub-id-type="doi">10.1007/s12975-018-0640-0</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Howard</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Banach</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kissela</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Cushman</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Muntner</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Judd</surname>
<given-names>S. E.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Age-related differences in the role of risk factors for ischemic stroke</article-title>. <source>Neurology</source> <volume>100</volume>, <fpage>e1444</fpage>-<lpage>e1453</lpage>. <pub-id pub-id-type="doi">10.1212/WNL.0000000000206837</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hunter</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Kelleher</surname>
<given-names>J. D.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Determining the proportionality of ischemic stroke risk factors to age</article-title>. <source>J. Cardiovasc. Dev. Dis.</source> <volume>10</volume>, <fpage>42</fpage>. <pub-id pub-id-type="doi">10.3390/jcdd10020042</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Johnson</surname>
<given-names>C. O.</given-names>
</name>
<name>
<surname>Nguyen</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Roth</surname>
<given-names>G. A.</given-names>
</name>
<name>
<surname>Nichols</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Alam</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Abate</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Global, regional, and national burden of stroke, 1990&#x2013;2016: a systematic analysis for the Global Burden of Disease Study 2016</article-title>. <source>Lancet Neurology</source> <volume>18</volume>, <fpage>439</fpage>&#x2013;<lpage>458</lpage>. <pub-id pub-id-type="doi">10.1016/S1474-4422(19)30034-1</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kingma</surname>
<given-names>D. P.</given-names>
</name>
<name>
<surname>Ba</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Adam: a method for stochastic optimization</article-title>. <source>arXiv Prepr. arXiv:1412.6980</source>.</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Bian</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2019a</year>). <article-title>Using machine learning models to improve stroke risk level classification methods of China national stroke screening</article-title>. <source>BMC Med. Inf. Decis. Mak.</source> <volume>19</volume>, <fpage>261</fpage>&#x2013;<lpage>267</lpage>. <pub-id pub-id-type="doi">10.1186/s12911-019-0998-2</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Pang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2019b</year>). <article-title>Discover high-risk factor combinations using Bayesian network from cohort data of National Stoke Screening in China</article-title>. <source>BMC Med. Inf. Decis. Mak.</source> <volume>19</volume>, <fpage>67</fpage>&#x2013;<lpage>68</lpage>. <pub-id pub-id-type="doi">10.1186/s12911-019-0753-8</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>NeuroCNN_GNB: an ensemble model to predict neuropeptides based on a convolution neural network and Gaussian naive Bayes</article-title>. <source>Front. Genet.</source> <volume>14</volume>, <fpage>1226905</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2023.1226905</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>A hybrid machine learning approach to cerebral stroke prediction based on imbalanced medical dataset</article-title>. <source>Artif. Intell. Med.</source> <volume>101</volume>, <fpage>101723</fpage>. <pub-id pub-id-type="doi">10.1016/j.artmed.2019.101723</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Long</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Shelhamer</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Darrell</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>Fully convolutional networks for semantic segmentation</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE conference on computer vision and pattern recognition</conf-name>, <fpage>3431</fpage>&#x2013;<lpage>3440</lpage>.</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Maalouf</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Hallit</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Salameh</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Hosseini</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Depression, anxiety, insomnia, stress, and the way of coping emotions as risk factors for ischemic stroke and their influence on stroke severity: a case&#x2013;control study in Lebanon</article-title>. <source>Front. psychiatry</source> <volume>14</volume>, <fpage>1097873</fpage>. <pub-id pub-id-type="doi">10.3389/fpsyt.2023.1097873</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mchugh</surname>
<given-names>M. L.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Interrater reliability: the kappa statistic</article-title>. <source>Biochem. medica</source> <volume>22</volume>, <fpage>276</fpage>&#x2013;<lpage>282</lpage>. <pub-id pub-id-type="doi">10.11613/bm.2012.031</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Noble</surname>
<given-names>W. S.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>What is a support vector machine?</article-title> <source>Nat. Biotechnol.</source> <volume>24</volume>, <fpage>1565</fpage>&#x2013;<lpage>1567</lpage>. <pub-id pub-id-type="doi">10.1038/nbt1206-1565</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nusinovici</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Tham</surname>
<given-names>Y. C.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>M. Y. C.</given-names>
</name>
<name>
<surname>Ting</surname>
<given-names>D. S. W.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Sabanayagam</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Logistic regression was as good as machine learning for predicting major chronic diseases</article-title>. <source>J. Clin. Epidemiol.</source> <volume>122</volume>, <fpage>56</fpage>&#x2013;<lpage>69</lpage>. <pub-id pub-id-type="doi">10.1016/j.jclinepi.2020.03.002</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Nwosu</surname>
<given-names>C. S.</given-names>
</name>
<name>
<surname>Dev</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Bhardwaj</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Veeravalli</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>John</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Predicting stroke from electronic health records</article-title>,&#x201d; in <conf-name>2019 41st Annual International Conference of the IEEE Engineering in Medicine and Biology Society (EMBC)</conf-name> (<publisher-name>IEEE</publisher-name>), <fpage>5704</fpage>&#x2013;<lpage>5707</lpage>.</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Obermeyer</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Emanuel</surname>
<given-names>E. J.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Predicting the future&#x2014;big data, machine learning, and clinical medicine</article-title>. <source>N. Engl. J. Med.</source> <volume>375</volume>, <fpage>1216</fpage>&#x2013;<lpage>1219</lpage>. <pub-id pub-id-type="doi">10.1056/NEJMp1606181</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ospel</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Ganesh</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Goyal</surname>
<given-names>M. J. J. O. S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Sex and gender differences in stroke and their practical implications in acute care</article-title>. <source>J. Stroke</source> <volume>25</volume>, <fpage>16</fpage>&#x2013;<lpage>25</lpage>. <pub-id pub-id-type="doi">10.5853/jos.2022.04077</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Owolabi</surname>
<given-names>M. O.</given-names>
</name>
<name>
<surname>Thrift</surname>
<given-names>A. G.</given-names>
</name>
<name>
<surname>Mahal</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ishida</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Martins</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Johnson</surname>
<given-names>W. D.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Primary stroke prevention worldwide: translating evidence into action</article-title>. <source>Lancet Public Health</source> <volume>7</volume>, <fpage>e74</fpage>&#x2013;<lpage>e85</lpage>. <pub-id pub-id-type="doi">10.1016/S2468-2667(21)00230-9</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Owolabi</surname>
<given-names>M. O.</given-names>
</name>
<name>
<surname>Thrift</surname>
<given-names>A. G.</given-names>
</name>
<name>
<surname>Martins</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Johnson</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Pandian</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Abd-Allah</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>The state of stroke services across the globe: report of world stroke organization&#x2013;world health organization surveys</article-title>. <source>Int. J. Stroke</source> <volume>16</volume>, <fpage>889</fpage>&#x2013;<lpage>901</lpage>. <pub-id pub-id-type="doi">10.1177/17474930211019568</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pandis</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>The chi-square test</article-title>. <source>Am. J. Of Orthod. And Dentofac. Orthop.</source> <volume>150</volume>, <fpage>898</fpage>&#x2013;<lpage>899</lpage>. <pub-id pub-id-type="doi">10.1016/j.ajodo.2016.08.009</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Park</surname>
<given-names>H. W.</given-names>
</name>
<name>
<surname>Pitti</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Madhavan</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Jeon</surname>
<given-names>Y.-J.</given-names>
</name>
<name>
<surname>Manavalan</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>MLACP 2.0: an updated machine learning tool for anticancer peptide prediction</article-title>. <source>Comput. Struct. Biotechnol. J.</source> <volume>20</volume>, <fpage>4473</fpage>&#x2013;<lpage>4480</lpage>. <pub-id pub-id-type="doi">10.1016/j.csbj.2022.07.043</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qi</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Guan</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Abu&#x2010;Hanna</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Schut</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Risk factors for incident stroke and its subtypes in China: a prospective study</article-title>. <source>J. Am. Heart Assoc.</source> <volume>9</volume>, <fpage>e016352</fpage>. <pub-id pub-id-type="doi">10.1161/JAHA.120.016352</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Schapire</surname>
<given-names>R. E.</given-names>
</name>
</person-group> (<year>2013</year>). &#x201c;<article-title>Explaining adaboost</article-title>,&#x201d; in <source>Empirical inference: festschrift in honor of vladimir N. Vapnik</source> (<publisher-name>Springer Berlin Heidelberg</publisher-name>), <fpage>37</fpage>&#x2013;<lpage>52</lpage>.</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sharpe</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Chi-square test is statistically significant: now what?</article-title> <source>Pract. Assess. Res. Eval.</source> <volume>20</volume>, <fpage>8</fpage>. <pub-id pub-id-type="doi">10.7275/tbfa-x148</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tian</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Si</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Association between temperature variability and daily hospital admissions for cause-specific cardiovascular disease in urban China: a national time-series study</article-title>. <source>PLoS Med.</source> <volume>16</volume>, <fpage>e1002738</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pmed.1002738</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vaswani</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Shazeer</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Parmar</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Uszkoreit</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jones</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Gomez</surname>
<given-names>A. N.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Attention is all you need</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>30</volume>.</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Development of stroke predictive model in community-dwelling population: a longitudinal cohort study in Southeast China</article-title>. <source>Front. Aging Neurosci.</source> <volume>14</volume>, <fpage>1036215</fpage>. <pub-id pub-id-type="doi">10.3389/fnagi.2022.1036215</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Younas</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Usman</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>W. Q.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A deep ensemble learning method for colorectal polyp classification with optimized network parameters</article-title>. <source>Appl. Intell.</source> <volume>53</volume>, <fpage>2410</fpage>&#x2013;<lpage>2433</lpage>. <pub-id pub-id-type="doi">10.1007/s10489-022-03689-9</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Mao</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ye</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>CSDC&#x2014;a nationwide screening platform for stroke control and prevention in China</article-title>,&#x201d; in <conf-name>2016 38th Annual International Conference of the IEEE Engineering in Medicine and Biology Society (EMBC)</conf-name> (<publisher-name>IEEE</publisher-name>), <fpage>2974</fpage>&#x2013;<lpage>2977</lpage>.</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Lai</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Shao</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>G.-S.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Discriminative elastic-net regularized linear regression</article-title>. <source>IEEE Trans. Image Process.</source> <volume>26</volume>, <fpage>1466</fpage>&#x2013;<lpage>1481</lpage>. <pub-id pub-id-type="doi">10.1109/TIP.2017.2651396</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Azim</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Predicting potential miRNA-disease associations by combining gradient boosting decision tree with logistic regression</article-title>. <source>Comput. Biol. Chem.</source> <volume>85</volume>, <fpage>107200</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiolchem.2020.107200</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>