<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Neurol.</journal-id>
<journal-title>Frontiers in Neurology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Neurol.</abbrev-journal-title>
<issn pub-type="epub">1664-2295</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fneur.2024.1379916</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Neurology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Enhancing dementia risk screening with GAN-synthesized periodontal examination and general blood test data</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes"><name><surname>Oyama</surname> <given-names>Katsunori</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref><xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/887063/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author"><name><surname>Isogai</surname> <given-names>Toshiki</given-names></name><xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author"><name><surname>Nakayama</surname> <given-names>Yohei</given-names></name><xref ref-type="aff" rid="aff3"><sup>3</sup></xref><xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author"><name><surname>Kobayashi</surname> <given-names>Ryoki</given-names></name><xref ref-type="aff" rid="aff3"><sup>3</sup></xref><xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author"><name><surname>Kitano</surname> <given-names>Daisuke</given-names></name><xref ref-type="aff" rid="aff6"><sup>6</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author"><name><surname>Karako</surname> <given-names>Kenji</given-names></name><xref ref-type="aff" rid="aff7">
<sup>7</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2422027/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author"><name><surname>Sakatani</surname> <given-names>Kaoru</given-names></name><xref ref-type="aff" rid="aff7">
<sup>7</sup></xref><xref ref-type="aff" rid="aff8">
<sup>8</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/876342/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Department of Computer Science, College of Engineering, Nihon University</institution>, <addr-line>Koriyama</addr-line>, <country>Japan</country></aff>
<aff id="aff2"><sup>2</sup><institution>Graduate School of Computer Science, Nihon University</institution>, <addr-line>Koriyama</addr-line>, <country>Japan</country></aff>
<aff id="aff3"><sup>3</sup><institution>Research Institute of Oral Science, Nihon University School of Dentistry at Matsudo</institution>, <addr-line>Matsudo</addr-line>, <country>Japan</country></aff>
<aff id="aff4"><sup>4</sup><institution>Department of Periodontology, Nihon University School of Dentistry at Matsudo</institution>, <addr-line>Matsudo</addr-line>, <country>Japan</country></aff>
<aff id="aff5"><sup>5</sup><institution>Department of Infection and Immunology, Nihon University School of Dentistry at Matsudo</institution>, <addr-line>Matsudo</addr-line>, <country>Japan</country></aff>
<aff id="aff6"><sup>6</sup><institution>Division of Cardiology, Department of Medicine, Nihon University School of Medicine</institution>, <addr-line>Itabashi</addr-line>, <country>Japan</country></aff>
<aff id="aff7"><sup>7</sup><institution>Department of Human and Engineered Environmental Studies, Graduate School of Frontier Sciences, The University of Tokyo</institution>, <addr-line>Kashiwa</addr-line>, <country>Japan</country></aff>
<aff id="aff8"><sup>8</sup><institution>Institute of Gerontology, The University of Tokyo</institution>, <addr-line>Bunkyo</addr-line>, <country>Japan</country></aff>
<author-notes>
<fn fn-type="edited-by" id="fn0001">
<p>Edited by: Shang-Ming Zhou, University of Plymouth, United Kingdom</p>
</fn>
<fn fn-type="edited-by" id="fn0002">
<p>Reviewed by: Johann Faouzi, National School of Statistics and Information Analysis, France</p>
<p>Atsuhiro Tsubaki, Niigata University of Health and Welfare, Japan</p>
</fn>
<corresp id="c001">&#x002A;Correspondence: Katsunori Oyama, <email>oyama.katsunori@nihon-u.ac.jp</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>14</day>
<month>08</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>15</volume>
<elocation-id>1379916</elocation-id>
<history>
<date date-type="received">
<day>31</day>
<month>01</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>22</day>
<month>07</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2024 Oyama, Isogai, Nakayama, Kobayashi, Kitano, Karako and Sakatani.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Oyama, Isogai, Nakayama, Kobayashi, Kitano, Karako and Sakatani</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec id="sec1001">
<title>Introduction</title>
<p>This study aimed to investigate the effectiveness of data augmentation to improve dementia risk prediction using machine learning models. Recent studies have shown that basic blood tests are cost-effective in predicting cognitive function. However, developing models that address various conditions poses challenges due to constraints associated with blood test results and cognitive assessments, including high costs, limited sample sizes, and missing data from tests not performed in certain facilities. Despite being often limited by small sample sizes, periodontal examination data have also emerged as a cost-effective screening tool.</p>
</sec>
<sec id="sec2001">
<title>Methods</title>
<p>To address these challenges, this study explored the effectiveness of data augmentation using the Synthetic Minority Over-sampling Technique for Regression with Gaussian noise (SMOGN), a Generative Adversarial Network (GAN), and a Conditional Tabular GAN (CTGAN) on periodontal examination and blood test data. The datasets included parameters such as cognitive assessment results from the Mini-Mental State Examination (MMSE), demographic characteristics, periodontal examination data, and blood test results. Linear regression models, random forests, and deep neural networks were used to evaluate the effectiveness of the synthesized data.</p>
</sec>
<sec id="sec3001">
<title>Results</title>
<p>This study used measured data from 108 participants and the synthesized data generated from the measured data. External validity was evaluated using a different dataset of 41 participants with missing items. The results suggested that normal GANs have the advantage of investigating models in data diversity, whereas CTGANs preserve the data structure and linear relationships in tabular data from the measured data, which drastically improves linear regression models.</p>
</sec>
<sec id="sec4001">
<title>Discussion</title>
<p>Importantly, by interpolating sparse areas in the distribution, such as age, the synthesized models maintained prediction accuracy for test data with extreme inputs. These findings suggest that GAN-synthesized data can effectively address regression problems and improve dementia risk prediction.</p>
</sec>
</abstract>
<kwd-group>
<kwd>blood test</kwd>
<kwd>periodontal examination</kwd>
<kwd>deep learning</kwd>
<kwd>generative adversarial networks</kwd>
<kwd>cognitive function</kwd>
</kwd-group>
<contract-num rid="cn1">JP23K25233</contract-num>
<contract-sponsor id="cn1">JSPS<named-content content-type="fundref-id">10.13039/501100001691</named-content></contract-sponsor>
<counts>
<fig-count count="3"/>
<table-count count="6"/>
<equation-count count="0"/>
<ref-count count="14"/>
<page-count count="10"/>
<word-count count="5244"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Dementia and Neurodegenerative Diseases</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec1">
<label>1</label>
<title>Introduction</title>
<p>The rapidly increasing older population has led to a rise in the prevalence of dementia, including Alzheimer&#x2019;s disease (AD). The number of people living with dementia across the world is expected to increase from 55 million in 2019 to 139 million in 2050 (<xref ref-type="bibr" rid="ref1">1</xref>). Accurate diagnosis remains complex because of the subtleties of mental status assessment and the similarity of AD to other types of dementia. Mild cognitive impairment (MCI), often a precursor to AD, is widely recognized as crucial for early detection and intervention.</p>
<p>Blood tests, which have recently been correlated with Mini-Mental State Examination (MMSE) scores (<xref ref-type="bibr" rid="ref2">2</xref>, <xref ref-type="bibr" rid="ref3">3</xref>), have emerged as a promising tool for cognitive screening. However, because of a variety of factors, machine learning models trained on blood test data often face limitations in accurately predicting dementia risks. One significant challenge is heterogeneity in patient data, including a wide range of biomarkers and cognitive scores. This variability can lead to inconsistencies in model performance, especially when dealing with multifaceted diseases such as dementia. Additionally, standard analytical models often struggle with the sparse and imbalanced nature of medical datasets, which can result in overfitting or the underrepresentation of certain patient groups. To address these issues, there is a growing need for innovative approaches capable of effectively handling diverse and incomplete data while maintaining predictive accuracy and reliability.</p>
<p>In addition to blood tests, recent research has highlighted the potential role of periodontal examination data in dementia risk assessment. Some studies have suggested a close relationship between cognitive function, oral health, and systemic metabolic function in older adults, with the number of healthy teeth being a significant predictor (<xref ref-type="bibr" rid="ref4">4</xref>). However, to the best of our knowledge, the combination of periodontal examination and blood test results have never been investigated for cost-effective and rapid screening of dementia risk. It is because the integration of periodontal examination data with blood tests still faces the obstacles, including the associated high costs, limited sample sizes, and missing data from unperformed tests, while periodontal health is increasingly recognized for its potential links with cognitive function, offering a promising avenue for early dementia detection.</p>
<p>Recent studies applying generative adversarial networks (GANs) for clinical applications, including diagnosis, prediction, and anomaly detection, can mostly be found in the field of medical imaging (<xref ref-type="bibr" rid="ref5">5</xref>, <xref ref-type="bibr" rid="ref6">6</xref>). This is the first study to integrate periodontal examination and blood test data and to apply synthesized models from tabular data for dementia risk prediction, which aim to address the challenges of data scarcity and heterogeneity in medical datasets that often impede accurate dementia risk prediction. This approach represents a step forward for cost-effective and rapid screening methods in early-stage dementia risk assessment.</p>
</sec>
<sec sec-type="methods" id="sec2">
<label>2</label>
<title>Methods</title>
<p>Data augmentation techniques such as SMOGN, GAN, and CTGAN were applied to generate synthesized datasets after preprocessing the measured data to handle missing values, as shown in <xref ref-type="fig" rid="fig1">Figure 1</xref>. The synthesized datasets were then used for training three basic machine learning models: linear regression (LR), random forest (RF), and deep neural network (DNN). Prediction errors and the robustness of the synthesized models were evaluated through 10 repeated hold-out validation process with different random seeds to ensure the reliability and generalizability of the synthesized data and models.</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption>
<p>Flowchart of the overall methodology, including data collection, preprocessing, augmentation, hold-out validation, and result analysis.</p>
</caption>
<graphic xlink:href="fneur-15-1379916-g001.tif"/>
</fig>
<sec id="sec3">
<label>2.1</label>
<title>Participants and measurements</title>
<p>We evaluated 108 individuals who were appointed for oral health assessments at Nihon University Itabashi Hospital (mean age&#x2009;&#x00B1;&#x2009;standard deviation [SD], 69.4&#x2009;&#x00B1;&#x2009;9.7&#x2009;years; age range, 33&#x2013;85&#x2009;years). Written informed consent was obtained from all participants who agreed to additionally take MMSE and blood tests after receiving ethical approval from the Institutional Review Board (approval No.: RK-191210-3). Periodontal examinations, performed by a dentist in the research project, included assessments such as the number of remaining healthy teeth.</p>
<p>Regarding the external validity of the synthesized models, we used a dataset of 41 participants (28 males, 13 females, mean age&#x2009;&#x00B1;&#x2009;SD, 69.7&#x2009;&#x00B1;&#x2009;5.6&#x2009;years) from a public health center in Koriyama, Fukushima, Japan, published in a preliminary study (<xref ref-type="bibr" rid="ref4">4</xref>). The mean MMSE score was 26.7&#x2009;&#x00B1;&#x2009;2.1. Compared with the measured data shown in <xref ref-type="table" rid="tab1">Table 1</xref>, the external validation test dataset lacked thirteen of 27 items; however, the remaining 14 items were commonly available as predictor variables between the measured data and the external validation test dataset: age, sex, white blood cell count (WBC), hemoglobin (Hb), platelet count (Plt), aspartate aminotransferase (AST), alanine aminotransferase (ALT), total cholesterol (T-Cho), triglyceride (TG), blood urea nitrogen (BUN), creatinine (Cr), uric acid (UA), total protein (TP), and number of remaining teeth.</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption>
<p>Measured data (<italic>N</italic>&#x2009;=&#x2009;108) with the parameters of Pearson&#x2019;s correlation with MMSE scores and statistical difference between two MMSE groups (&#x003C;28, &#x2265;28).</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top" colspan="2">Item</th>
<th align="center" valign="top">All (<italic>N</italic> =&#x2009;108)</th>
<th align="center" valign="top">Pearson&#x2019;s correlation</th>
<th align="center" valign="top">MMSE &#x003C;28 (<italic>n</italic> =&#x2009;45)</th>
<th align="center" valign="top">MMSE &#x2265;28 (<italic>n</italic> =&#x2009;63)</th>
<th align="center" valign="top"><italic>p</italic>-value</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">Response</td>
<td align="left" valign="middle">MMSE</td>
<td align="center" valign="middle">27.5&#x2009;&#x00B1;&#x2009;2.4</td>
<td/>
<td align="center" valign="middle">25.2&#x2009;&#x00B1;&#x2009;1.9</td>
<td align="center" valign="middle">29.2&#x2009;&#x00B1;&#x2009;0.8</td>
<td align="center" valign="middle">&#x002A;&#x002A;</td>
</tr>
<tr>
<td align="left" valign="middle" rowspan="7">Demographic data</td>
<td align="left" valign="middle">Age (years)</td>
<td align="center" valign="middle">69.4&#x2009;&#x00B1;&#x2009;9.7</td>
<td align="center" valign="middle">&#x2212;0.32&#x002A;</td>
<td align="center" valign="middle">72.3&#x2009;&#x00B1;&#x2009;8.4</td>
<td align="center" valign="middle">67.3&#x2009;&#x00B1;&#x2009;10</td>
<td align="center" valign="middle">&#x002A;&#x002A;</td>
</tr>
<tr>
<td align="left" valign="middle">Sex</td>
<td align="center" valign="middle">M:88/F:20</td>
<td/>
<td align="center" valign="middle">M:51/F:12</td>
<td align="center" valign="middle">M:37/F:8</td>
<td/>
</tr>
<tr>
<td align="left" valign="middle">Height (cm)</td>
<td align="center" valign="middle">164.3&#x2009;&#x00B1;&#x2009;0.1</td>
<td align="center" valign="middle">0.19</td>
<td align="center" valign="middle">1.63&#x2009;&#x00B1;&#x2009;0.08</td>
<td align="center" valign="middle">1.65&#x2009;&#x00B1;&#x2009;0.08</td>
<td align="center" valign="middle">&#x002A;</td>
</tr>
<tr>
<td align="left" valign="middle">Weight (kg)</td>
<td align="center" valign="middle">65.2&#x2009;&#x00B1;&#x2009;11.2</td>
<td align="center" valign="middle">0.11</td>
<td align="center" valign="middle">63.6&#x2009;&#x00B1;&#x2009;10.4</td>
<td align="center" valign="middle">66.3&#x2009;&#x00B1;&#x2009;11.6</td>
<td/>
</tr>
<tr>
<td align="left" valign="middle">sBP</td>
<td align="center" valign="middle">128.7&#x2009;&#x00B1;&#x2009;17.5</td>
<td align="center" valign="middle">&#x2212;0.08</td>
<td align="center" valign="middle">129.1&#x2009;&#x00B1;&#x2009;18.8</td>
<td align="center" valign="middle">128.4&#x2009;&#x00B1;&#x2009;16.6</td>
<td/>
</tr>
<tr>
<td align="left" valign="middle">dBP</td>
<td align="center" valign="middle">75.2&#x2009;&#x00B1;&#x2009;10.6</td>
<td align="center" valign="middle">0</td>
<td align="center" valign="middle">74.2&#x2009;&#x00B1;&#x2009;10.5</td>
<td align="center" valign="middle">75.9&#x2009;&#x00B1;&#x2009;10.7</td>
<td/>
</tr>
<tr>
<td align="left" valign="middle">HR</td>
<td align="center" valign="middle">73.8&#x2009;&#x00B1;&#x2009;12.8</td>
<td align="center" valign="middle">&#x2212;0.07</td>
<td align="center" valign="middle">74.8&#x2009;&#x00B1;&#x2009;12.4</td>
<td align="center" valign="middle">73.0&#x2009;&#x00B1;&#x2009;13.0</td>
<td/>
</tr>
<tr>
<td align="left" valign="middle" rowspan="15">General blood test</td>
<td align="left" valign="middle">WBC</td>
<td align="center" valign="middle">5623.8&#x2009;&#x00B1;&#x2009;1458.9</td>
<td align="center" valign="middle">&#x2212;0.04</td>
<td align="center" valign="middle">5,650&#x2009;&#x00B1;&#x2009;1464.4</td>
<td align="center" valign="middle">5606.3&#x2009;&#x00B1;&#x2009;1466.7</td>
<td/>
</tr>
<tr>
<td align="left" valign="middle">Hb</td>
<td align="center" valign="middle">13.8&#x2009;&#x00B1;&#x2009;1.4</td>
<td align="center" valign="middle">0.08</td>
<td align="center" valign="middle">13.6&#x2009;&#x00B1;&#x2009;1.6</td>
<td align="center" valign="middle">14.0&#x2009;&#x00B1;&#x2009;1.2</td>
<td/>
</tr>
<tr>
<td align="left" valign="middle">Plt</td>
<td align="center" valign="middle">21.1&#x2009;&#x00B1;&#x2009;6.4</td>
<td align="center" valign="middle">&#x2212;0.04</td>
<td align="center" valign="middle">20.8&#x2009;&#x00B1;&#x2009;6.3</td>
<td align="center" valign="middle">21.4&#x2009;&#x00B1;&#x2009;6.5</td>
<td/>
</tr>
<tr>
<td align="left" valign="middle">AST</td>
<td align="center" valign="middle">24.8&#x2009;&#x00B1;&#x2009;8.9</td>
<td align="center" valign="middle">&#x2212;0.10</td>
<td align="center" valign="middle">24.7&#x2009;&#x00B1;&#x2009;9.2</td>
<td align="center" valign="middle">24.8&#x2009;&#x00B1;&#x2009;8.7</td>
<td/>
</tr>
<tr>
<td align="left" valign="middle">ALT</td>
<td align="center" valign="middle">23.1&#x2009;&#x00B1;&#x2009;14.1</td>
<td align="center" valign="middle">&#x2212;0.01</td>
<td align="center" valign="middle">21.9&#x2009;&#x00B1;&#x2009;13.6</td>
<td align="center" valign="middle">24.0&#x2009;&#x00B1;&#x2009;14.4</td>
<td/>
</tr>
<tr>
<td align="left" valign="middle">LDH</td>
<td align="center" valign="middle">179.3&#x2009;&#x00B1;&#x2009;30.7</td>
<td align="center" valign="middle">&#x2212;0.17</td>
<td align="center" valign="middle">184.2&#x2009;&#x00B1;&#x2009;37</td>
<td align="center" valign="middle">175.9&#x2009;&#x00B1;&#x2009;25.1</td>
<td/>
</tr>
<tr>
<td align="left" valign="middle">T-Cho</td>
<td align="center" valign="middle">171.0&#x2009;&#x00B1;&#x2009;33.8</td>
<td align="center" valign="middle">0.07</td>
<td align="center" valign="middle">171.3&#x2009;&#x00B1;&#x2009;31.3</td>
<td align="center" valign="middle">170.9&#x2009;&#x00B1;&#x2009;35.5</td>
<td/>
</tr>
<tr>
<td align="left" valign="middle">HDL cholesterol</td>
<td align="center" valign="middle">56.3&#x2009;&#x00B1;&#x2009;15.2</td>
<td align="center" valign="middle">&#x2212;0.01</td>
<td align="center" valign="middle">57.9&#x2009;&#x00B1;&#x2009;16.8</td>
<td align="center" valign="middle">55.3&#x2009;&#x00B1;&#x2009;14.2</td>
<td/>
</tr>
<tr>
<td align="left" valign="middle">LDL cholesterol</td>
<td align="center" valign="middle">85.5&#x2009;&#x00B1;&#x2009;27.1</td>
<td align="center" valign="middle">0.14</td>
<td align="center" valign="middle">82.8&#x2009;&#x00B1;&#x2009;24.1</td>
<td align="center" valign="middle">87.2&#x2009;&#x00B1;&#x2009;28.9</td>
<td/>
</tr>
<tr>
<td align="left" valign="middle">TG</td>
<td align="center" valign="middle">144.6&#x2009;&#x00B1;&#x2009;114.7</td>
<td align="center" valign="middle">&#x2212;0.04</td>
<td align="center" valign="middle">149.5&#x2009;&#x00B1;&#x2009;115.8</td>
<td align="center" valign="middle">141.4&#x2009;&#x00B1;&#x2009;114.9</td>
<td/>
</tr>
<tr>
<td align="left" valign="middle">BUN</td>
<td align="center" valign="middle">17.5&#x2009;&#x00B1;&#x2009;5.3</td>
<td align="center" valign="middle">&#x2212;0.22&#x002A;</td>
<td align="center" valign="middle">19.0&#x2009;&#x00B1;&#x2009;6.3</td>
<td align="center" valign="middle">16.4&#x2009;&#x00B1;&#x2009;4.1</td>
<td align="center" valign="middle">&#x002A;&#x002A;</td>
</tr>
<tr>
<td align="left" valign="middle">Cr</td>
<td align="center" valign="middle">0.9&#x2009;&#x00B1;&#x2009;0.3</td>
<td align="center" valign="middle">&#x2212;0.04</td>
<td align="center" valign="middle">1.0&#x2009;&#x00B1;&#x2009;0.4</td>
<td align="center" valign="middle">0.9&#x2009;&#x00B1;&#x2009;0.3</td>
<td/>
</tr>
<tr>
<td align="left" valign="middle">UA</td>
<td align="center" valign="middle">5.5&#x2009;&#x00B1;&#x2009;1.3</td>
<td align="center" valign="middle">0.11</td>
<td align="center" valign="middle">5.3&#x2009;&#x00B1;&#x2009;1.3</td>
<td align="center" valign="middle">5.6&#x2009;&#x00B1;&#x2009;1.3</td>
<td/>
</tr>
<tr>
<td align="left" valign="middle">TP</td>
<td align="center" valign="middle">7.1&#x2009;&#x00B1;&#x2009;0.5</td>
<td align="center" valign="middle">0.04</td>
<td align="center" valign="middle">7.0&#x2009;&#x00B1;&#x2009;0.6</td>
<td align="center" valign="middle">7.1&#x2009;&#x00B1;&#x2009;0.4</td>
<td/>
</tr>
<tr>
<td align="left" valign="middle">HbA1c</td>
<td align="center" valign="middle">6.1&#x2009;&#x00B1;&#x2009;0.7</td>
<td align="center" valign="middle">&#x2212;0.05</td>
<td align="center" valign="middle">6.1&#x2009;&#x00B1;&#x2009;0.9</td>
<td align="center" valign="middle">6.1&#x2009;&#x00B1;&#x2009;0.6</td>
<td/>
</tr>
<tr>
<td align="left" valign="middle" rowspan="5">Periodontal examination</td>
<td align="left" valign="middle">No. of remaining teeth</td>
<td align="center" valign="middle">22.3&#x2009;&#x00B1;&#x2009;6.8</td>
<td align="center" valign="middle">0.18</td>
<td align="center" valign="middle">21.1&#x2009;&#x00B1;&#x2009;7.6</td>
<td align="center" valign="middle">23.0&#x2009;&#x00B1;&#x2009;6.2</td>
<td/>
</tr>
<tr>
<td align="left" valign="middle">Ave. PD</td>
<td align="center" valign="middle">2.7&#x2009;&#x00B1;&#x2009;0.5</td>
<td align="center" valign="middle">&#x2212;0.08</td>
<td align="center" valign="middle">2.7&#x2009;&#x00B1;&#x2009;0.4</td>
<td align="center" valign="middle">2.7&#x2009;&#x00B1;&#x2009;0.6</td>
<td/>
</tr>
<tr>
<td align="left" valign="middle">Ave. CAL</td>
<td align="center" valign="middle">3.9&#x2009;&#x00B1;&#x2009;1.1</td>
<td align="center" valign="middle">&#x2212;0.12</td>
<td align="center" valign="middle">3.9&#x2009;&#x00B1;&#x2009;1.1</td>
<td align="center" valign="middle">3.8&#x2009;&#x00B1;&#x2009;1.1</td>
<td/>
</tr>
<tr>
<td align="left" valign="middle">PISA</td>
<td align="center" valign="middle">221.2&#x2009;&#x00B1;&#x2009;215.8</td>
<td align="center" valign="middle">0.05</td>
<td align="center" valign="middle">197.8&#x2009;&#x00B1;&#x2009;203.1</td>
<td align="center" valign="middle">236.7&#x2009;&#x00B1;&#x2009;224.1</td>
<td/>
</tr>
<tr>
<td align="left" valign="middle">PESA</td>
<td align="center" valign="middle">1108.1&#x2009;&#x00B1;&#x2009;372.6</td>
<td align="center" valign="middle">0.17</td>
<td align="center" valign="middle">1036.4&#x2009;&#x00B1;&#x2009;377.5</td>
<td align="center" valign="middle">1155.4&#x2009;&#x00B1;&#x2009;364.6</td>
<td align="center" valign="middle">&#x002A;</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>MMSE, Mini-Mental State Examination; WBC, white blood cell count; Hb, hemoglobin; Plt, platelet count; AST, aspartate aminotransferase; T-Cho, total cholesterol; HDL, high-density lipoprotein; LDL, low-density lipoprotein; TG, triglycerides; BUN, blood urea nitrogen; UA, uric acid; Ave. PD, average probing depth; Ave. CAL, average clinical attachment level; PISA, periodontal inflammatory surface area; PESA, periodontal epithelial surface area.&#x002A;<italic>p</italic> &#x003C; 0.05, &#x002A;&#x002A;<italic>p</italic> &#x003C; 0.01.</p>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="sec4">
<label>2.2</label>
<title>Data augmentation</title>
<p>GANs were used to generate synthesized data from our dataset with combinations of sample sizes (100 and 500) and learning epochs (300, 1,000, 3,000, 5,000), while SMOGN served as the base model, effectively balancing the data distribution for oversampling in regression problems. For more details, the learning parameters of SMOGN, GAN, and CTGAN are listed in <xref ref-type="supplementary-material" rid="SM1">Supplementary Table S1</xref>.</p>
<sec id="sec5">
<label>2.2.1</label>
<title>Synthesizer 1&#x2014;synthetic minority over-sampling technique for regression with Gaussian noise</title>
<p>To address data imbalances, we applied SMOGN, which enhances minority data representation by generating synthetic samples with a Gaussian noise model (<xref ref-type="bibr" rid="ref7">7</xref>). This technique ensures that the dataset remains representative of the original distribution for training unbiased models.</p>
</sec>
<sec id="sec6">
<label>2.2.2</label>
<title>Synthesizer 2&#x2014;GAN</title>
<p>Our standard GAN were implemented using the TensorFlow 2 library. GANs operate using a pair of neural networks&#x2014;the Generator and the Discriminator&#x2014;which are trained concurrently through a competitive process. The Generator creates data from random noise, learning to make it indistinguishable from measured data. Conversely, the Discriminator learns to distinguish accurately whether the data presented is generated or real (<xref ref-type="bibr" rid="ref8">8</xref>). The standard GAN in this study features a Generator and Discriminator, each with three hidden layers. These layers were configured with 256, 128, and 64&#x2009;units for the Generator, and 128, 64, and 32&#x2009;units for the Discriminator. We opted for the leaky rectified linear unit activation function to maintain gradient flow during training as a standard model of GAN, with each layer followed by batch normalization. The model was trained using a batch size of 16 and a learning rate of 0.0001 with the ADAM optimizer, which was selected based on preliminary experiments to optimize convergence.</p>
</sec>
<sec id="sec7">
<label>2.2.3</label>
<title>Synthesizer 3&#x2014;conditional tabular GAN</title>
<p>In addition to the standard GAN, a CTGAN (<xref ref-type="bibr" rid="ref9">9</xref>) was utilized because of its proficiency in synthesizing tabular data while preserving conditional distributions, which is particularly effective if a medical dataset needs to preserve specific statistical characteristics. The CTGAN open source library in the Synthetic Data Vault project is adept at capturing complex relationships between variables in a table format, making it particularly suitable for medical datasets, which often contain a mix of categorical and continuous features and require the statistical properties of the original data to be retained. In our application, CTGAN was applied to generate synthetic yet realistic and representative patient data, thereby enhancing the training process by providing a richer and more diverse set of samples for improving the generalization capabilities of our dementia risk prediction models. The model was trained with 5,000 epochs, a batch size of 500, and a learning rate of 0.0002 to achieve optimized convergence.</p>
</sec>
</sec>
<sec id="sec8">
<label>2.3</label>
<title>Machine learning models for data analysis</title>
<p>We employed LR, RF, and DNN models to estimate MMSE scores using the measured and synthesized data. The LR served as a baseline for comparison because of its interpretability and previous applications in AD research. RF, which is known for its ability to handle nonlinear data and robustness to noise, was included to assess its performance in our context. The DNNs, which were constructed using Tensor Flow 2, consisted of four hidden layers designed to capture intricate relationships. For more details, the learning parameters of RF and DNN are listed as <xref ref-type="supplementary-material" rid="SM1">Supplementary Table S2</xref>.</p>
<sec id="sec9">
<label>2.3.1</label>
<title>LR</title>
<p>We implemented ordinary least-squares regression as our baseline model because of its interpretability and established use in AD research. This method provides a clear and straightforward way to analyze the linear relationship between predictor variables (e.g., blood test results, periodontal examination data) and MMSE scores.</p>
</sec>
<sec id="sec10">
<label>2.3.2</label>
<title>RF</title>
<p>Recognized for its ability to handle nonlinear relationships and robustness to noise, the RF model was constructed with 300 trees and a maximum depth of 10 using the scikit-learn library. This approach enhances predictive performance and helps control overfitting. The suitability of RF for AD research is supported by its success in similar applications (<xref ref-type="bibr" rid="ref10">10</xref>).</p>
</sec>
<sec id="sec11">
<label>2.3.3</label>
<title>DNN</title>
<p>The neural network architecture included 19 input neurons, representing demographic, blood test, and periodontal examination data. Four hidden layers with descending neuron counts (256, 128, and 64) were used to process effectively the inputs as reported. We employed the scaled exponential linear unit activation function, which normalizes input signals to improve the training efficiency and stability. Batch normalization was applied after each layer. The network was trained using a batch size of 8 and a learning rate of 0.001 with the ADAM optimizer. These hyperparameters were optimized using a grid search to ensure stable model convergence.</p>
</sec>
</sec>
</sec>
<sec sec-type="results" id="sec12">
<label>3</label>
<title>Results</title>
<sec id="sec13">
<label>3.1</label>
<title>Statistical analysis of measured data</title>
<p>We systematically assessed cognitive function using the MMSE, complemented by demographic and blood test data, to explore correlations with periodontal health indicators, as shown in <xref ref-type="table" rid="tab1">Table 1</xref>. Given the exploratory nature of our study, we reported unadjusted <italic>p</italic>-values to highlight potential associations. With an average MMSE score of 27.5&#x2009;&#x00B1;&#x2009;2.4, the participants were categorized for further analysis into two groups based on MMSE scores: those with scores &#x003C;28, which are indicative of possible MCI, and those with scores &#x2265;28, which are considered within the normal cognitive range. The MMSE cut-off score of 27/28 is frequently employed in studies focusing on early detection of cognitive decline among older adults (<xref ref-type="bibr" rid="ref11">11</xref>). Independent <italic>t</italic>-tests showed differences in the mean values for age, height, BUN, and PESA between these groups, suggesting their potential influence on cognitive status as determined by the MMSE (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.01 for age and BUN; <italic>p</italic>&#x2009;&#x003C;&#x2009;0.05 for height and PESA).</p>
<p>The blood test items selected in <xref ref-type="table" rid="tab1">Table 1</xref> are commonly measured parameters in practice, and the periodontal examination items include PISA, PESA, and related computational metrics. For the subsequent machine learning analyses, a total of 27 variables were selected. These were chosen by having &#x003C;20% missing values and ensuring a correlation coefficient&#x2009;&#x003C;&#x2009;0.9 between items to check for multicollinearity.</p>
</sec>
<sec id="sec14">
<label>3.2</label>
<title>Results of data augmentation</title>
<p>For conducting the repeated hold-out validations, we employed SMOGN, GAN, and CTGAN to generate 10 sets of synthesized data from the measured dataset (<xref ref-type="table" rid="tab2">Table 2</xref>). Both GAN and CTGAN were utilized for addition to the measured data for a comparison between them. The Kolmogorov&#x2013;Smirnov (KS) complement scores indicated that SMOGN and CTGAN replicated the original data distribution more closely than GAN, especially for the MMSE score distribution in which CTGAN reached the highest score of 0.88&#x2009;&#x00B1;&#x2009;0.04. It is noteworthy that GAN extensively extrapolates both in MMSE score and Age distributions around the minimum score (Min.) of the measured data, whereas SMOGN and CTGAN maintain the measured data distribution.</p>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption>
<p>Augmentation results: 10 sets of synthesized data for repeated hold-out validation.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th colspan="2"></th>
<th align="center" valign="top">Measured data (no augmentation)</th>
<th align="center" valign="top">SMOGN</th>
<th align="center" valign="top">GAN</th>
<th align="center" valign="top">CTGAN</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle" colspan="2">Sample size of training data</td>
<td align="center" valign="middle" colspan="4">86.4&#x2009;&#x00B1;&#x2009;0.5</td>
</tr>
<tr>
<td align="left" valign="middle" colspan="2">Size of output synthesized data</td>
<td align="center" valign="middle">&#x2013;</td>
<td align="center" valign="middle">108.3&#x2009;&#x00B1;&#x2009;8.7</td>
<td align="center" valign="middle">100 or 500</td>
<td align="center" valign="middle">100 or 500</td>
</tr>
<tr>
<td align="left" valign="middle" rowspan="3">MMSE score</td>
<td align="left" valign="middle">Min.</td>
<td align="center" valign="middle">21&#x2009;&#x00B1;&#x2009;0</td>
<td align="center" valign="middle">20.9&#x2009;&#x00B1;&#x2009;0.0</td>
<td align="center" valign="middle">14.4&#x2009;&#x00B1;&#x2009;2.4</td>
<td align="center" valign="middle">21.0&#x2009;&#x00B1;&#x2009;0.0</td>
</tr>
<tr>
<td align="left" valign="middle">Mean</td>
<td align="center" valign="middle">27.4&#x2009;&#x00B1;&#x2009;0.1</td>
<td align="center" valign="middle">27.1&#x2009;&#x00B1;&#x2009;0.2</td>
<td align="center" valign="middle">27.2&#x2009;&#x00B1;&#x2009;0.3</td>
<td align="center" valign="middle">26.7&#x2009;&#x00B1;&#x2009;0.4</td>
</tr>
<tr>
<td align="left" valign="middle">KS complement score</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">0.83&#x2009;&#x00B1;&#x2009;0.01</td>
<td align="center" valign="middle">0.75&#x2009;&#x00B1;&#x2009;0.07</td>
<td align="center" valign="middle">0.88&#x2009;&#x00B1;&#x2009;0.04</td>
</tr>
<tr>
<td align="left" valign="middle" rowspan="5">Age</td>
<td align="left" valign="middle">Min.</td>
<td align="center" valign="middle">33.0&#x2009;&#x00B1;&#x2009;8.0</td>
<td align="center" valign="middle">33.0&#x2009;&#x00B1;&#x2009;8.0</td>
<td align="center" valign="middle">0.0&#x2009;&#x00B1;&#x2009;16.7</td>
<td align="center" valign="middle">33.0&#x2009;&#x00B1;&#x2009;8.0</td>
</tr>
<tr>
<td align="left" valign="middle">Mean</td>
<td align="center" valign="middle">68.5&#x2009;&#x00B1;&#x2009;0.6</td>
<td align="center" valign="middle">69.1&#x2009;&#x00B1;&#x2009;0.6</td>
<td align="center" valign="middle">66.5&#x2009;&#x00B1;&#x2009;1.5</td>
<td align="center" valign="middle">65.8&#x2009;&#x00B1;&#x2009;2.2</td>
</tr>
<tr>
<td align="left" valign="middle">Max.</td>
<td align="center" valign="middle">85.0&#x2009;&#x00B1;&#x2009;0.0</td>
<td align="center" valign="middle">85.0&#x2009;&#x00B1;&#x2009;0.1</td>
<td align="center" valign="middle">85.0&#x2009;&#x00B1;&#x2009;8.7</td>
<td align="center" valign="middle">85.0&#x2009;&#x00B1;&#x2009;0.0</td>
</tr>
<tr>
<td align="left" valign="middle">KS complement score</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">0.89&#x2009;&#x00B1;&#x2009;0.02</td>
<td align="center" valign="middle">0.83&#x2009;&#x00B1;&#x2009;0.06</td>
<td align="center" valign="middle">0.86&#x2009;&#x00B1;&#x2009;0.03</td>
</tr>
<tr>
<td align="left" valign="middle">Similarity of Pearson&#x2019;s correlation coefficients</td>
<td align="center" valign="middle">1</td>
<td align="center" valign="middle">0.91&#x2009;&#x00B1;&#x2009;0.02</td>
<td align="center" valign="middle">0.79&#x2009;&#x00B1;&#x2009;0.06</td>
<td align="center" valign="middle">0.80&#x2009;&#x00B1;&#x2009;0.07</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>SMOGN, synthetic minority over-sampling technique for regression with Gaussian noise; GAN, generative adversarial network; CTGAN, conditional tabular GAN.</p>
</table-wrap-foot>
</table-wrap>
<p>To understand how CTGAN interpolates the measured data, distributions of MMSE score and Age are compared as shown in <xref ref-type="fig" rid="fig2">Figure 2</xref>. In terms of the age distribution, CTGAN-synthesized data are closer to the measured data with a correlation of &#x2212;0.27&#x2009;&#x00B1;&#x2009;0.01, and the addition of CTGAN-synthesized data to the measured data only changes the correlation to MMSE scores within 0.05. Age distribution of the synthesized data also is closer to that of the measured data.</p>
<fig position="float" id="fig2">
<label>Figure 2</label>
<caption>
<p>Distributions of MMSE scores and age. <bold>(A)</bold> Distribution of MMSE scores: measured data (blue) and synthesized data (red). <bold>(B)</bold> Distribution of age: measured data (blue) and synthesized data (red).</p>
</caption>
<graphic xlink:href="fneur-15-1379916-g002.tif"/>
</fig>
</sec>
<sec id="sec15">
<label>3.3</label>
<title>Results of hold-out validation</title>
<p>The LR, RF, and DNN models trained on measured data without periodontal examination items exhibited mean absolute errors (MAEs) of 2.10&#x2009;&#x00B1;&#x2009;0.21, 1.95&#x2009;&#x00B1;&#x2009;0.22, and 2.30&#x2009;&#x00B1;&#x2009;0.27, respectively, in repeated hold-out validation (<xref ref-type="table" rid="tab3">Table 3</xref>). After including periodontal examination items in the models, LR showed an increase in prediction error due to a larger number of variables, while RF and DNN exhibited improvements with mean MAEs of 2.28&#x2009;&#x00B1;&#x2009;0.33, 1.94&#x2009;&#x00B1;&#x2009;0.21, and 2.18&#x2009;&#x00B1;&#x2009;0.23, respectively.</p>
<table-wrap position="float" id="tab3">
<label>Table 3</label>
<caption>
<p>Prediction errors in repeated hold-out validation using measured and synthesized data with and without periodontal examination items: most improved MAEs are highlighted.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top" rowspan="2">Training (measured&#x2009;+&#x2009;synthesized data)</th>
<th align="center" valign="top" rowspan="2">Learning epochs for GAN</th>
<th align="center" valign="top" colspan="3">Without periodontal examination items</th>
<th align="center" valign="top" colspan="3">With all available inputs</th>
</tr>
<tr>
<th align="center" valign="top">LR</th>
<th align="center" valign="top">RF</th>
<th align="center" valign="top">DNN</th>
<th align="center" valign="top">LR</th>
<th align="center" valign="top">RF</th>
<th align="center" valign="top">DNN</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">Measured data (<italic>N</italic>&#x2009;=&#x2009;86) only</td>
<td/>
<td align="center" valign="middle">2.10&#x2009;&#x00B1;&#x2009;0.21</td>
<td align="center" valign="middle">1.95&#x2009;&#x00B1;&#x2009;0.22</td>
<td align="center" valign="middle">2.30&#x2009;&#x00B1;&#x2009;0.27</td>
<td align="center" valign="middle">2.28&#x2009;&#x00B1;&#x2009;0.33</td>
<td align="center" valign="middle">1.94&#x2009;&#x00B1;&#x2009;0.21</td>
<td align="center" valign="middle">2.18&#x2009;&#x00B1;&#x2009;0.23</td>
</tr>
<tr>
<td align="left" valign="middle">+ SMOGN (<italic>N</italic>&#x2009;=&#x2009;75.4&#x2009;&#x00B1;&#x2009;4.7)</td>
<td/>
<td align="center" valign="middle">2.19&#x2009;&#x00B1;&#x2009;0.24</td>
<td align="center" valign="middle">1.95&#x2009;&#x00B1;&#x2009;0.17</td>
<td align="center" valign="middle">2.27&#x2009;&#x00B1;&#x2009;0.30</td>
<td align="center" valign="middle">2.37&#x2009;&#x00B1;&#x2009;0.31</td>
<td align="center" valign="middle">1.94&#x2009;&#x00B1;&#x2009;0.16</td>
<td align="center" valign="middle">2.24&#x2009;&#x00B1;&#x2009;0.27</td>
</tr>
<tr>
<td align="left" valign="middle" rowspan="4">+ GAN (<italic>N</italic>&#x2009;=&#x2009;100)</td>
<td align="center" valign="middle">300</td>
<td align="center" valign="middle">2.08&#x2009;&#x00B1;&#x2009;0.36</td>
<td align="center" valign="middle">1.97&#x2009;&#x00B1;&#x2009;0.28</td>
<td align="center" valign="middle">2.27&#x2009;&#x00B1;&#x2009;0.28</td>
<td align="center" valign="middle">2.15&#x2009;&#x00B1;&#x2009;0.37</td>
<td align="center" valign="middle">1.96&#x2009;&#x00B1;&#x2009;0.30</td>
<td align="center" valign="middle">2.25&#x2009;&#x00B1;&#x2009;0.34</td>
</tr>
<tr>
<td align="center" valign="middle">1,000</td>
<td align="center" valign="middle">2.09&#x2009;&#x00B1;&#x2009;0.23</td>
<td align="center" valign="middle">1.98&#x2009;&#x00B1;&#x2009;0.23</td>
<td align="center" valign="middle">2.18&#x2009;&#x00B1;&#x2009;0.26</td>
<td align="center" valign="middle">2.18&#x2009;&#x00B1;&#x2009;0.31</td>
<td align="center" valign="middle">1.97&#x2009;&#x00B1;&#x2009;0.22</td>
<td align="center" valign="middle">2.02&#x2009;&#x00B1;&#x2009;0.23</td>
</tr>
<tr>
<td align="center" valign="middle">3,000</td>
<td align="center" valign="middle">2.12&#x2009;&#x00B1;&#x2009;0.22</td>
<td align="center" valign="middle">2.02&#x2009;&#x00B1;&#x2009;0.28</td>
<td align="center" valign="middle">2.26&#x2009;&#x00B1;&#x2009;0.19</td>
<td align="center" valign="middle">2.16&#x2009;&#x00B1;&#x2009;0.24</td>
<td align="center" valign="middle">1.98&#x2009;&#x00B1;&#x2009;0.25</td>
<td align="center" valign="middle">2.11&#x2009;&#x00B1;&#x2009;0.25</td>
</tr>
<tr>
<td align="center" valign="middle">5,000</td>
<td align="center" valign="middle">2.13&#x2009;&#x00B1;&#x2009;0.13</td>
<td align="center" valign="middle">2.04&#x2009;&#x00B1;&#x2009;0.22</td>
<td align="center" valign="middle">2.19&#x2009;&#x00B1;&#x2009;0.27</td>
<td align="center" valign="middle">2.23&#x2009;&#x00B1;&#x2009;0.24</td>
<td align="center" valign="middle">2.02&#x2009;&#x00B1;&#x2009;0.24</td>
<td align="center" valign="middle">2.19&#x2009;&#x00B1;&#x2009;0.21</td>
</tr>
<tr>
<td align="left" valign="middle" rowspan="4">+ GAN (<italic>N</italic>&#x2009;=&#x2009;500)</td>
<td align="center" valign="middle">300</td>
<td align="center" valign="middle">2.19&#x2009;&#x00B1;&#x2009;0.43</td>
<td align="center" valign="middle">1.96&#x2009;&#x00B1;&#x2009;0.35</td>
<td align="center" valign="middle">2.15&#x2009;&#x00B1;&#x2009;0.43</td>
<td align="center" valign="middle">2.25&#x2009;&#x00B1;&#x2009;0.43</td>
<td align="center" valign="middle">1.96&#x2009;&#x00B1;&#x2009;0.35</td>
<td align="center" valign="middle">2.09&#x2009;&#x00B1;&#x2009;0.32</td>
</tr>
<tr>
<td align="center" valign="middle">1,000</td>
<td align="center" valign="middle">2.18&#x2009;&#x00B1;&#x2009;0.31</td>
<td align="center" valign="middle">2.01&#x2009;&#x00B1;&#x2009;0.26</td>
<td align="center" valign="middle">2.25&#x2009;&#x00B1;&#x2009;0.26</td>
<td align="center" valign="middle">2.21&#x2009;&#x00B1;&#x2009;0.38</td>
<td align="center" valign="middle">1.98&#x2009;&#x00B1;&#x2009;0.23</td>
<td align="center" valign="middle">2.04&#x2009;&#x00B1;&#x2009;0.27</td>
</tr>
<tr>
<td align="center" valign="middle">3,000</td>
<td align="center" valign="middle">2.09&#x2009;&#x00B1;&#x2009;0.24</td>
<td align="center" valign="middle">2.06&#x2009;&#x00B1;&#x2009;0.32</td>
<td align="center" valign="middle">2.32&#x2009;&#x00B1;&#x2009;0.28</td>
<td align="center" valign="middle">2.08&#x2009;&#x00B1;&#x2009;0.22</td>
<td align="center" valign="middle">2.04&#x2009;&#x00B1;&#x2009;0.28</td>
<td align="center" valign="middle">2.20&#x2009;&#x00B1;&#x2009;0.25</td>
</tr>
<tr>
<td align="center" valign="middle">5,000</td>
<td align="center" valign="middle">2.19&#x2009;&#x00B1;&#x2009;0.14</td>
<td align="center" valign="middle">2.10&#x2009;&#x00B1;&#x2009;0.24</td>
<td align="center" valign="middle">2.17&#x2009;&#x00B1;&#x2009;0.33</td>
<td align="center" valign="middle">2.22&#x2009;&#x00B1;&#x2009;0.23</td>
<td align="center" valign="middle">2.07&#x2009;&#x00B1;&#x2009;0.24</td>
<td align="center" valign="middle">2.18&#x2009;&#x00B1;&#x2009;0.28</td>
</tr>
<tr>
<td align="left" valign="middle" rowspan="4">+ CTGAN (<italic>N</italic>&#x2009;=&#x2009;100)</td>
<td align="center" valign="middle">300</td>
<td align="center" valign="middle">2.05&#x2009;&#x00B1;&#x2009;0.24</td>
<td align="center" valign="middle">2.10&#x2009;&#x00B1;&#x2009;0.30</td>
<td align="center" valign="middle">2.20&#x2009;&#x00B1;&#x2009;0.23</td>
<td align="center" valign="middle">2.08&#x2009;&#x00B1;&#x2009;0.27</td>
<td align="center" valign="middle">2.09&#x2009;&#x00B1;&#x2009;0.33</td>
<td align="center" valign="middle">2.19&#x2009;&#x00B1;&#x2009;0.20</td>
</tr>
<tr>
<td align="center" valign="middle">1,000</td>
<td align="center" valign="middle">2.09&#x2009;&#x00B1;&#x2009;0.16</td>
<td align="center" valign="middle">2.09&#x2009;&#x00B1;&#x2009;0.20</td>
<td align="center" valign="middle">2.35&#x2009;&#x00B1;&#x2009;0.16</td>
<td align="center" valign="middle">2.10&#x2009;&#x00B1;&#x2009;0.22</td>
<td align="center" valign="middle">2.03&#x2009;&#x00B1;&#x2009;0.21</td>
<td align="center" valign="middle">2.28&#x2009;&#x00B1;&#x2009;0.32</td>
</tr>
<tr>
<td align="center" valign="middle">3,000</td>
<td align="center" valign="middle">1.99&#x2009;&#x00B1;&#x2009;0.23</td>
<td align="center" valign="middle">1.96&#x2009;&#x00B1;&#x2009;0.21</td>
<td align="center" valign="middle">2.33&#x2009;&#x00B1;&#x2009;0.15</td>
<td align="center" valign="middle">1.97&#x2009;&#x00B1;&#x2009;0.25</td>
<td align="center" valign="middle">1.95&#x2009;&#x00B1;&#x2009;0.24</td>
<td align="center" valign="middle">2.12&#x2009;&#x00B1;&#x2009;0.26</td>
</tr>
<tr>
<td align="center" valign="middle">5,000</td>
<td align="center" valign="middle">2.02&#x2009;&#x00B1;&#x2009;0.24</td>
<td align="center" valign="middle">2.02&#x2009;&#x00B1;&#x2009;0.30</td>
<td align="center" valign="middle">2.35&#x2009;&#x00B1;&#x2009;0.38</td>
<td align="center" valign="middle">2.06&#x2009;&#x00B1;&#x2009;0.29</td>
<td align="center" valign="middle">1.99&#x2009;&#x00B1;&#x2009;0.29</td>
<td align="center" valign="middle">2.40&#x2009;&#x00B1;&#x2009;0.29</td>
</tr>
<tr>
<td align="left" valign="middle" rowspan="4">+ CTGAN (<italic>N</italic>&#x2009;=&#x2009;500)</td>
<td align="center" valign="middle">300</td>
<td align="center" valign="middle">2.13&#x2009;&#x00B1;&#x2009;0.30</td>
<td align="center" valign="middle">2.25&#x2009;&#x00B1;&#x2009;0.37</td>
<td align="center" valign="middle">2.30&#x2009;&#x00B1;&#x2009;0.24</td>
<td align="center" valign="middle">2.13&#x2009;&#x00B1;&#x2009;0.29</td>
<td align="center" valign="middle">2.28&#x2009;&#x00B1;&#x2009;0.38</td>
<td align="center" valign="middle">2.20&#x2009;&#x00B1;&#x2009;0.23</td>
</tr>
<tr>
<td align="center" valign="middle">1,000</td>
<td align="center" valign="middle">2.14&#x2009;&#x00B1;&#x2009;0.17</td>
<td align="center" valign="middle">2.16&#x2009;&#x00B1;&#x2009;0.27</td>
<td align="center" valign="middle">2.31&#x2009;&#x00B1;&#x2009;0.24</td>
<td align="center" valign="middle">2.12&#x2009;&#x00B1;&#x2009;0.17</td>
<td align="center" valign="middle">2.14&#x2009;&#x00B1;&#x2009;0.27</td>
<td align="center" valign="middle">2.22&#x2009;&#x00B1;&#x2009;0.25</td>
</tr>
<tr>
<td align="center" valign="middle">3,000</td>
<td align="center" valign="middle">1.95&#x2009;&#x00B1;&#x2009;0.27</td>
<td align="center" valign="middle">2.05&#x2009;&#x00B1;&#x2009;0.27</td>
<td align="center" valign="middle">2.21&#x2009;&#x00B1;&#x2009;0.26</td>
<td align="center" valign="middle">1.96&#x2009;&#x00B1;&#x2009;0.34</td>
<td align="center" valign="middle">2.03&#x2009;&#x00B1;&#x2009;0.28</td>
<td align="center" valign="middle">2.12&#x2009;&#x00B1;&#x2009;0.30</td>
</tr>
<tr>
<td align="center" valign="middle">5,000</td>
<td align="center" valign="middle">
<bold>1.97&#x2009;&#x00B1;&#x2009;0.24</bold>
</td>
<td align="center" valign="middle">2.02&#x2009;&#x00B1;&#x2009;0.26</td>
<td align="center" valign="middle">2.19&#x2009;&#x00B1;&#x2009;0.32</td>
<td align="center" valign="middle">
<bold>1.96&#x2009;&#x00B1;&#x2009;0.27</bold>
</td>
<td align="center" valign="middle">2.00&#x2009;&#x00B1;&#x2009;0.26</td>
<td align="center" valign="middle">2.15&#x2009;&#x00B1;&#x2009;0.30</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>LR, Linear Regression model; RF, Random Forest model; DNN, Deep Neural Network model.</p>
</table-wrap-foot>
</table-wrap>
<p>Interestingly, LR achieved the most significant improvement in prediction errors compared to RF and DNN when CTGAN-synthesized data (<italic>N</italic>&#x2009;=&#x2009;500) were added to the measured data, as highlighted in <xref ref-type="table" rid="tab3">Table 3</xref>. This result suggests that LR effectively utilizes CTGAN-synthesized data for dementia risk predictions, which outperforms RF and DNN models under these conditions. <xref ref-type="fig" rid="fig3">Figure 3</xref> illustrate that, compared with the results of the worst models, CTGAN with the certain seed number drastically improved all LR, RF, and DNN models by showing proportional measured and predicted MMSE scores, while the models with a poor augmentation result can diffuse predicted MMSE scores.</p>
<fig position="float" id="fig3">
<label>Figure 3</label>
<caption>
<p>Measured and predicted scores in the worst and best models using CTGAN-synthesized data: <bold>(A)</bold> LR, <bold>(B)</bold> RF, and <bold>(C)</bold> DNN.</p>
</caption>
<graphic xlink:href="fneur-15-1379916-g003.tif"/>
</fig>
<p>Next, the mean&#x2009;&#x00B1;&#x2009;SD of standard coefficients in the LR models were assessed to understand the variable importance of periodontal examination items (<xref ref-type="table" rid="tab4">Table 4</xref>). While PESA originally showed higher importance with the measured data, its importance diminished when CTGAN-synthesized data were included. Instead, the average PD level became relatively important while the number of remaining teeth is known to be a good biomarker. To assess the dependency on age, we further analyzed the models by excluding this variable as listed in <xref ref-type="table" rid="tab4">Table 4</xref>. BUN, Height, Weight, and PESA especially emerged as more influential variables in the LR models without Age.</p>
<table-wrap position="float" id="tab4">
<label>Table 4</label>
<caption>
<p>Standard coefficients in the LR models for comparison between measured data only (<italic>N</italic>&#x2009;=&#x2009;108) and measured data + CTGAN-synthesized data (<italic>N</italic>&#x2009;=&#x2009;108&#x2009;+&#x2009;500) in repeated hold-out validation: periodontal examination items are highlighted.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Variable (top 15)</th>
<th align="center" valign="top">LR using measured data (no augmentation)</th>
<th align="center" valign="top">Variable (top 15)</th>
<th align="center" valign="top">LR using measured data<break/>+&#x2009;CTGAN data</th>
<th align="center" valign="top">Variable (top 15)</th>
<th align="center" valign="top">LR using measured data&#x2009;+&#x2009;CTGAN data without age</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">
<bold>PESA</bold>
</td>
<td align="center" valign="middle">1.33&#x2009;&#x00B1;&#x2009;0.97</td>
<td align="center" valign="middle">Age</td>
<td align="center" valign="middle">&#x2212;0.42&#x2009;&#x00B1;&#x2009;0.15</td>
<td align="center" valign="middle">Sex</td>
<td align="center" valign="middle">0.35&#x2009;&#x00B1;&#x2009;0.20</td>
</tr>
<tr>
<td align="left" valign="middle">
<bold>RemainingTooth</bold>
</td>
<td align="center" valign="middle">&#x2212;0.91&#x2009;&#x00B1;&#x2009;0.78</td>
<td align="center" valign="middle">Sex</td>
<td align="center" valign="middle">0.33&#x2009;&#x00B1;&#x2009;0.19</td>
<td align="center" valign="middle">BUN</td>
<td align="center" valign="middle">&#x2212;0.31&#x2009;&#x00B1;&#x2009;0.10</td>
</tr>
<tr>
<td align="left" valign="middle">BUN</td>
<td align="center" valign="middle">&#x2212;0.73&#x2009;&#x00B1;&#x2009;0.19</td>
<td align="center" valign="middle">BUN</td>
<td align="center" valign="middle">&#x2212;0.24&#x2009;&#x00B1;&#x2009;0.09</td>
<td align="center" valign="middle">Weight</td>
<td align="center" valign="middle">0.22&#x2009;&#x00B1;&#x2009;0.10</td>
</tr>
<tr>
<td align="left" valign="middle">
<bold>AvePD</bold>
</td>
<td align="center" valign="middle">&#x2212;0.72&#x2009;&#x00B1;&#x2009;0.47</td>
<td align="center" valign="middle">
<bold>AvePD</bold>
</td>
<td align="center" valign="middle">&#x2212;0.18&#x2009;&#x00B1;&#x2009;0.13</td>
<td align="center" valign="middle">Height</td>
<td align="center" valign="middle">0.22&#x2009;&#x00B1;&#x2009;0.16</td>
</tr>
<tr>
<td align="left" valign="middle">Sex</td>
<td align="center" valign="middle">0.68&#x2009;&#x00B1;&#x2009;0.19</td>
<td align="center" valign="middle">Weight</td>
<td align="center" valign="middle">0.16&#x2009;&#x00B1;&#x2009;0.10</td>
<td align="center" valign="middle">Plt</td>
<td align="center" valign="middle">&#x2212;0.15&#x2009;&#x00B1;&#x2009;0.16</td>
</tr>
<tr>
<td align="left" valign="middle">Age</td>
<td align="center" valign="middle">&#x2212;0.60&#x2009;&#x00B1;&#x2009;0.12</td>
<td align="center" valign="middle">Height</td>
<td align="center" valign="middle">0.16&#x2009;&#x00B1;&#x2009;0.15</td>
<td align="center" valign="middle">
<bold>AvePD</bold>
</td>
<td align="center" valign="middle">&#x2212;0.15&#x2009;&#x00B1;&#x2009;0.14</td>
</tr>
<tr>
<td align="left" valign="middle">dBP</td>
<td align="center" valign="middle">&#x2212;0.38&#x2009;&#x00B1;&#x2009;0.07</td>
<td align="center" valign="middle">Plt</td>
<td align="center" valign="middle">&#x2212;0.16&#x2009;&#x00B1;&#x2009;0.16</td>
<td align="center" valign="middle">LDH</td>
<td align="center" valign="middle">&#x2212;0.14&#x2009;&#x00B1;&#x2009;0.11</td>
</tr>
<tr>
<td align="left" valign="middle">
<bold>PISA</bold>
</td>
<td align="center" valign="middle">&#x2212;0.37&#x2009;&#x00B1;&#x2009;0.38</td>
<td align="center" valign="middle">LDH</td>
<td align="center" valign="middle">&#x2212;0.10&#x2009;&#x00B1;&#x2009;0.10</td>
<td align="center" valign="middle">LDL</td>
<td align="center" valign="middle">0.12&#x2009;&#x00B1;&#x2009;0.15</td>
</tr>
<tr>
<td align="left" valign="middle">Cr</td>
<td align="center" valign="middle">0.37&#x2009;&#x00B1;&#x2009;0.25</td>
<td align="center" valign="middle">HR</td>
<td align="center" valign="middle">&#x2212;0.10&#x2009;&#x00B1;&#x2009;0.11</td>
<td align="center" valign="middle">HR</td>
<td align="center" valign="middle">&#x2212;0.11&#x2009;&#x00B1;&#x2009;0.10</td>
</tr>
<tr>
<td align="left" valign="middle">UA</td>
<td align="center" valign="middle">0.36&#x2009;&#x00B1;&#x2009;0.16</td>
<td align="center" valign="middle">LDL</td>
<td align="center" valign="middle">0.10&#x2009;&#x00B1;&#x2009;0.14</td>
<td align="center" valign="middle">
<bold>PESA</bold>
</td>
<td align="center" valign="middle">0.10&#x2009;&#x00B1;&#x2009;0.06</td>
</tr>
<tr>
<td align="left" valign="middle">Height</td>
<td align="center" valign="middle">0.34&#x2009;&#x00B1;&#x2009;0.10</td>
<td align="center" valign="middle">
<bold>RemainingTooth</bold>
</td>
<td align="center" valign="middle">0.09&#x2009;&#x00B1;&#x2009;0.09</td>
<td align="center" valign="middle">
<bold>RemainingTooth</bold>
</td>
<td align="center" valign="middle">0.10&#x2009;&#x00B1;&#x2009;0.09</td>
</tr>
<tr>
<td align="left" valign="middle">AST</td>
<td align="center" valign="middle">&#x2212;0.33&#x2009;&#x00B1;&#x2009;0.22</td>
<td align="center" valign="middle">ALT</td>
<td align="center" valign="middle">&#x2212;0.08&#x2009;&#x00B1;&#x2009;0.10</td>
<td align="center" valign="middle">Hb</td>
<td align="center" valign="middle">0.10&#x2009;&#x00B1;&#x2009;0.13</td>
</tr>
<tr>
<td align="left" valign="middle">HDL</td>
<td align="center" valign="middle">&#x2212;0.29&#x2009;&#x00B1;&#x2009;0.29</td>
<td align="center" valign="middle">UA</td>
<td align="center" valign="middle">0.08&#x2009;&#x00B1;&#x2009;0.09</td>
<td align="center" valign="middle">UA</td>
<td align="center" valign="middle">0.08&#x2009;&#x00B1;&#x2009;0.08</td>
</tr>
<tr>
<td align="left" valign="middle">WBC</td>
<td align="center" valign="middle">&#x2212;0.23&#x2009;&#x00B1;&#x2009;0.08</td>
<td align="center" valign="middle">T-Cho</td>
<td align="center" valign="middle">0.08&#x2009;&#x00B1;&#x2009;0.11</td>
<td align="center" valign="middle">
<bold>AveCAL</bold>
</td>
<td align="center" valign="middle">&#x2212;0.07&#x2009;&#x00B1;&#x2009;0.16</td>
</tr>
<tr>
<td align="left" valign="middle">TG</td>
<td align="center" valign="middle">&#x2212;0.22&#x2009;&#x00B1;&#x2009;0.31</td>
<td align="center" valign="middle">AST</td>
<td align="center" valign="middle">&#x2212;0.08&#x2009;&#x00B1;&#x2009;0.11</td>
<td align="center" valign="middle">T-Cho</td>
<td align="center" valign="middle">0.07&#x2009;&#x00B1;&#x2009;0.13</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="sec16">
<label>3.4</label>
<title>Robustness experiments</title>
<p>To evaluate model robustness against the insertion of anomalous data, validation tests were conducted with the input Age set to &#x201C;0,&#x201D; where Age is the important variable both for both original and synthesized models. In these cases, the models trained with measured data exhibited a significant increase in prediction errors. Specifically, LR, as a linear model, showed heightened sensitivity to anomalous values in both the measured and synthesized datasets. By contrast, the GAN-and CTGAN-synthesized models demonstrated stable MAEs as detailed in <xref ref-type="table" rid="tab5">Table 5</xref>. These results suggest that handling diverseness in the data distribution during the augmentation process may be key to addressing challenges prevalent in medical datasets, such as imbalanced data and missing values.</p>
<table-wrap position="float" id="tab5">
<label>Table 5</label>
<caption>
<p>Prediction errors in repeated hold-out validation with the anomalous inputs by replacing test data: age&#x2009;=&#x2009;0.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Training (measured + synthesized data with the anomalous inputs: <bold>age&#x2009;=&#x2009;0</bold>)</th>
<th align="center" valign="top">LR</th>
<th align="center" valign="top">RF</th>
<th align="center" valign="top">DNN</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">Measured data (<italic>N</italic>&#x2009;=&#x2009;86) only</td>
<td align="center" valign="middle">4.66&#x2009;&#x00B1;&#x2009;1.10</td>
<td align="center" valign="middle">1.98&#x2009;&#x00B1;&#x2009;0.24</td>
<td align="center" valign="middle">2.58&#x2009;&#x00B1;&#x2009;0.55</td>
</tr>
<tr>
<td align="left" valign="middle">+ SMOGN (<italic>N</italic>&#x2009;=&#x2009;75.4&#x2009;&#x00B1;&#x2009;4.7)</td>
<td align="center" valign="middle">5.05&#x2009;&#x00B1;&#x2009;1.28</td>
<td align="center" valign="middle">2.02&#x2009;&#x00B1;&#x2009;0.19</td>
<td align="center" valign="middle">2.33&#x2009;&#x00B1;&#x2009;0.39</td>
</tr>
<tr>
<td align="left" valign="middle">+ GAN (<italic>N</italic>&#x2009;=&#x2009;500)</td>
<td align="center" valign="middle">3.30&#x2009;&#x00B1;&#x2009;1.77</td>
<td align="center" valign="middle">2.00&#x2009;&#x00B1;&#x2009;0.25</td>
<td align="center" valign="middle">2.24&#x2009;&#x00B1;&#x2009;0.44</td>
</tr>
<tr>
<td align="left" valign="middle">+ CTGAN (<italic>N</italic>&#x2009;=&#x2009;500)</td>
<td align="center" valign="middle">3.40&#x2009;&#x00B1;&#x2009;0.89</td>
<td align="center" valign="middle">2.02&#x2009;&#x00B1;&#x2009;0.28</td>
<td align="center" valign="middle">2.10&#x2009;&#x00B1;&#x2009;0.46</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="sec17">
<label>3.5</label>
<title>External validity</title>
<p>We also evaluated the prediction errors using the external validation test dataset (<xref ref-type="table" rid="tab6">Table 6</xref>). CTGAN-synthesized LR exhibited a significant decrease in MAEs, to 1.55&#x2009;&#x00B1;&#x2009;0.27 (18.8% improvements). Despite of the limited 14 common items, we confirmed that reasonable prediction error results are obtained using CTGAN-synthesized data.</p>
<table-wrap position="float" id="tab6">
<label>Table 6</label>
<caption>
<p>Prediction errors for the external validation test data (<italic>N</italic>&#x2009;=&#x2009;41) using the 10 sets of measured and synthesized data: most improved MAEs are highlighted.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Training (measured + synthesized data)</th>
<th align="center" valign="top">LR</th>
<th align="center" valign="top">RF</th>
<th align="center" valign="top">DNN</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">Measured data (<italic>N</italic>&#x2009;=&#x2009;86) only</td>
<td align="center" valign="middle">1.91&#x2009;&#x00B1;&#x2009;0.41</td>
<td align="center" valign="middle">1.89&#x2009;&#x00B1;&#x2009;0.10</td>
<td align="center" valign="middle">1.87&#x2009;&#x00B1;&#x2009;0.34</td>
</tr>
<tr>
<td align="left" valign="middle">+ SMOGN (<italic>N</italic>&#x2009;=&#x2009;75.4&#x2009;&#x00B1;&#x2009;4.7)</td>
<td align="center" valign="middle">2.13&#x2009;&#x00B1;&#x2009;0.39</td>
<td align="center" valign="middle">1.92&#x2009;&#x00B1;&#x2009;0.22</td>
<td align="center" valign="middle">2.03&#x2009;&#x00B1;&#x2009;0.53</td>
</tr>
<tr>
<td align="left" valign="middle">+ GAN (<italic>N</italic>&#x2009;=&#x2009;500)</td>
<td align="center" valign="middle">1.99&#x2009;&#x00B1;&#x2009;0.83</td>
<td align="center" valign="middle">1.74&#x2009;&#x00B1;&#x2009;0.16</td>
<td align="center" valign="middle">2.01&#x2009;&#x00B1;&#x2009;0.50</td>
</tr>
<tr>
<td align="left" valign="middle">+ CTGAN (<italic>N</italic>&#x2009;=&#x2009;500)</td>
<td align="center" valign="middle">
<bold>1.55&#x2009;&#x00B1;&#x2009;0.27</bold>
</td>
<td align="center" valign="middle">1.95&#x2009;&#x00B1;&#x2009;0.28</td>
<td align="center" valign="middle">1.90&#x2009;&#x00B1;&#x2009;0.45</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec sec-type="discussion" id="sec18">
<label>4</label>
<title>Discussion</title>
<sec id="sec19">
<label>4.1</label>
<title>Validity of periodontal examination and blood test data for dementia risk screening</title>
<p>The integration of periodontal examination and blood test data showed slight improvements. However, the average probing depth (PD) and the number of remaining teeth were identified as valuable biomarkers. This confirms the importance of tooth counts as a modifiable dementia risk factor (<xref ref-type="bibr" rid="ref12">12</xref>) and the potential role of PD levels in dementia risk screening.</p>
<p>PESA can be useful for dementia risk screening; however, it is highly dependent on Age. PESA is calculated by summing the product of the remaining teeth and the probing depth (PD) level for each tooth, meaning the number of remaining teeth significantly affects the results. This suggests that further stratified analysis based on Age and the number of teeth may contribute to a better understanding of the importance of periodontal examination items if PESA is considered.</p>
<p>While the relationship between periodontitis and dementia is well established, other blood test items related to oxygen transport and nutrition did not show the expected variations with MMSE scores in our dataset. This could be attributed to the demographic characteristics of our participants, who were primarily patients undergoing oral health assessments and were less likely to exhibit symptoms typically associated with dementia, such as anemia, metabolic syndrome, and chronic inflammation.</p>
<p>Further research is needed to explore the interrelations between blood test results and periodontal disease. Although salivary levels of BUN and AST are known to be useful biomarkers for screening periodontal disease (<xref ref-type="bibr" rid="ref13">13</xref>), whether blood test results are significantly influenced by periodontal disease remains unclear.</p>
</sec>
<sec id="sec20">
<label>4.2</label>
<title>Effectiveness of synthesized medical tabular data for model performance</title>
<p>The synthesized data generated using GAN and CTGAN showed promising results, especially for improving the performance of the LR models. Interestingly, the inclusion of CTGAN-synthesized data (<italic>N</italic>&#x2009;=&#x2009;500) significantly enhanced the predictive accuracy of the LR models more than RF and DNN models. This suggests that LR models are better suited for application of synthesized data in predicting dementia risk.</p>
<p>The findings also indicate that synthesized data can help handle the issue of limited sample sizes in medical research. Not only addition of CTGAN-synthesized data preserves the statistical properties of the original dataset, synthesized data are effective to improve the robustness and generalizability of machine learning models. This will be particularly important in the context of dementia risk prediction, where obtaining large and diverse datasets can be challenging.</p>
<p>We conclude that the potential of combining blood test results, periodontal examination data with synthesized data to improve dementia risk screening, and boot-strapping of synthesized data will be the key for successful models of dementia risk screening where data collection may be limited or expensive, on the other hand, it may always take some time to find the best result and some degree of automation is necessary for practical use.</p>
</sec>
<sec id="sec21">
<label>4.3</label>
<title>Implications for dementia risk screening and limitations</title>
<p>The application of artificial intelligence (AI) and machine learning in health care, particularly in dementia research, represents a significant technological advancement. However, this comes with ethical responsibilities, including the protection of data privacy, informed patient consent, and the reliability of predictions. Collaborative efforts among data scientists, clinicians, and health-care professionals are more vital to ensure the responsible use of AI in medical practice than ever.</p>
<p>This study did have some limitations, including the initial dataset size and its representativeness. Previous studies have identified other crucial blood test components, such as Plt, glycated hemoglobin, albumin, and electrolytes (<xref ref-type="bibr" rid="ref3">3</xref>, <xref ref-type="bibr" rid="ref14">14</xref>); however, we could not examine these components in the present study. Further research considering these elements could provide a more comprehensive understanding of dementia risk factors. Moreover, while the correlation between periodontitis and dementia is more evident, the causal relationships remain to be fully explained. In this sense, longitudinal data analysis to track changes over time and identify potential causal pathways can give good solutions.</p>
<p>This study also acknowledges that the training process for both GAN and CTGAN can be time-intensive and highly sensitive to learning parameters. Recent advancements in the field of GAN training may result in methodologies that can mitigate these challenges. The success of GANs in this study warrants their exploration in other health-care domains where data limitations are a significant barrier to innovation.</p>
</sec>
</sec>
<sec sec-type="conclusions" id="sec22">
<label>5</label>
<title>Conclusion</title>
<p>The results of this study confirm the potential of GANs in enhancing dementia risk prediction, particularly in settings with data limitations. The GAN- and CTGAN-synthesized models were able to maintain robustness against anomalies and outperform models trained with limited data. Future advancements in GAN methodologies could further revolutionize health-care technology and patient care in dementia and beyond.</p>
</sec>
<sec sec-type="data-availability" id="sec23">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec sec-type="ethics-statement" id="sec24">
<title>Ethics statement</title>
<p>The studies involving humans were approved by Nihon University Hospitals&#x2019; Joint Institutional Review Board. The studies were conducted in accordance with the local legislation and institutional requirements. The participants provided their written informed consent to participate in this study.</p>
</sec>
<sec sec-type="author-contributions" id="sec25">
<title>Author contributions</title>
<p>KO: Conceptualization, Formal analysis, Funding acquisition, Investigation, Methodology, Project administration, Resources, Software, Supervision, Validation, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. TI: Data curation, Investigation, Writing &#x2013; original draft. YN: Data curation, Funding acquisition, Investigation, Project administration, Writing &#x2013; review &#x0026; editing. RK: Data curation, Investigation, Writing &#x2013; review &#x0026; editing. DK: Data curation, Writing &#x2013; review &#x0026; editing. KK: Formal analysis, Writing &#x2013; review &#x0026; editing. KS: Supervision, Writing &#x2013; review &#x0026; editing.</p>
</sec>
</body>
<back>
<sec sec-type="funding-information" id="sec26">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This work was supported in part by a JSPS Grant-in-Aid for Scientific Research (B) grant number JP23K25233 and Nihon University Research Grants for 2020.</p>
</sec>
<sec sec-type="COI-statement" id="sec27">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="sec28">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="sec29">
<title>Supplementary material</title>
<p>The Supplementary material for this article can be found online at: <ext-link xlink:href="https://www.frontiersin.org/articles/10.3389/fneur.2024.1379916/full#supplementary-material" ext-link-type="uri">https://www.frontiersin.org/articles/10.3389/fneur.2024.1379916/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Data_Sheet_1.pdf" id="SM1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="ref1">
<label>1.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Long</surname> <given-names>S</given-names></name> <name><surname>Benoist</surname> <given-names>C</given-names></name> <name><surname>Weidner</surname> <given-names>W</given-names></name></person-group>. <article-title>World Alzheimer report 2023: reducing dementia risk: never too early, never too late</article-title>. <source>Alzheimer&#x2019;s Dis Int</source>. (<year>2023</year>)</citation>
</ref>
<ref id="ref2">
<label>2.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Murayama</surname> <given-names>Y</given-names></name> <name><surname>Sato</surname> <given-names>Y</given-names></name> <name><surname>Hu</surname> <given-names>L</given-names></name> <name><surname>Brugnera</surname> <given-names>A</given-names></name> <name><surname>Compare</surname> <given-names>A</given-names></name> <name><surname>Sakatani</surname> <given-names>K</given-names></name></person-group>. <article-title>Relation between cognitive function and baseline concentrations of hemoglobin in prefrontal cortex of elderly people measured by time-resolved near-infrared spectroscopy</article-title>. <source>Adv Exp Med Biol</source>. (<year>2017</year>) <volume>977</volume>:<fpage>269</fpage>&#x2013;<lpage>76</lpage>. doi: <pub-id pub-id-type="doi">10.1007/978-3-319-55231-6_37</pub-id>, PMID: <pub-id pub-id-type="pmid">28685456</pub-id></citation>
</ref>
<ref id="ref3">
<label>3.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sakatani</surname> <given-names>K</given-names></name> <name><surname>Oyama</surname> <given-names>K</given-names></name> <name><surname>Hu</surname> <given-names>L</given-names></name></person-group>. <article-title>Deep learning-based screening test for cognitive impairment using basic blood test data for health examination</article-title>. <source>Front Neurol</source>. (<year>2020</year>) <volume>11</volume>:<fpage>588140</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fneur.2020.588140</pub-id>, PMID: <pub-id pub-id-type="pmid">33381075</pub-id></citation>
</ref>
<ref id="ref4">
<label>4.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Karako</surname> <given-names>K</given-names></name> <name><surname>Chen</surname> <given-names>Y</given-names></name> <name><surname>Oyama</surname> <given-names>K</given-names></name> <name><surname>Hu</surname> <given-names>L</given-names></name> <name><surname>Sakatani</surname> <given-names>K</given-names></name></person-group>. <article-title>Relationship between cognitive function, Oral conditions and systemic metabolic function in the elderly</article-title>. <source>Adv Exp Med Biol</source>. (<year>2022</year>) <volume>1438</volume>:<fpage>27</fpage>&#x2013;<lpage>31</lpage>. doi: <pub-id pub-id-type="doi">10.1007/978-3-031-42003-0_5</pub-id></citation>
</ref>
<ref id="ref5">
<label>5.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ali</surname> <given-names>H</given-names></name> <name><surname>Biswas</surname> <given-names>MR</given-names></name> <name><surname>Mohsen</surname> <given-names>F</given-names></name> <name><surname>Shah</surname> <given-names>U</given-names></name> <name><surname>Alamgir</surname> <given-names>A</given-names></name> <name><surname>Mousa</surname> <given-names>O</given-names></name> <etal/></person-group>. <article-title>The role of generative adversarial networks in brain MRI: a scoping review</article-title>. <source>Insights Imaging</source>. (<year>2022</year>) <volume>13</volume>:<fpage>98</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s13244-022-01237-0</pub-id>, PMID: <pub-id pub-id-type="pmid">35662369</pub-id></citation>
</ref>
<ref id="ref6">
<label>6.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>Y</given-names></name> <name><surname>Wang</surname> <given-names>Z</given-names></name> <name><surname>Zhang</surname> <given-names>Z</given-names></name> <name><surname>Liu</surname> <given-names>J</given-names></name> <name><surname>Feng</surname> <given-names>Y</given-names></name> <name><surname>Wee</surname> <given-names>L</given-names></name> <etal/></person-group>. <article-title>GAN-based one dimensional medical data augmentation</article-title>. <source>Soft Comput</source>. (<year>2023</year>) <volume>27</volume>:<fpage>10481</fpage>&#x2013;<lpage>91</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s00500-023-08345-z</pub-id></citation>
</ref>
<ref id="ref7">
<label>7.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Branco</surname> <given-names>P</given-names></name> <name><surname>Torgo</surname> <given-names>L</given-names></name> <name><surname>Ribeiro</surname> <given-names>R</given-names></name></person-group>. <article-title>SMOGN: a pre-processing approach for imbalanced regression</article-title>. <source>Proc Mach Learn Res</source>. (<year>2017</year>) <volume>74</volume>:<fpage>36</fpage>&#x2013;<lpage>50</lpage>.</citation>
</ref>
<ref id="ref8">
<label>8.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Goodfellow</surname> <given-names>I</given-names></name> <name><surname>Pouget-Abadie</surname> <given-names>J</given-names></name> <name><surname>Mirza</surname> <given-names>M</given-names></name> <name><surname>Xu</surname> <given-names>B</given-names></name> <name><surname>Warde-Farley</surname> <given-names>D</given-names></name> <name><surname>Ozair</surname> <given-names>S</given-names></name> <etal/></person-group>. <article-title>Generative adversarial nets</article-title>. <source>Adv Neural Inf Proces Syst</source>. (<year>2014</year>) <volume>27</volume>:<fpage>2672</fpage>&#x2013;<lpage>80</lpage>.</citation>
</ref>
<ref id="ref9">
<label>9.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>L</given-names></name> <name><surname>Skoularidou</surname> <given-names>M</given-names></name> <name><surname>Cuesta-Infante</surname> <given-names>A</given-names></name> <name><surname>Veeramachaneni</surname> <given-names>K</given-names></name></person-group>. <article-title>Modeling Tabular data using Conditional GAN</article-title>. <source>Adv Neural Inf Proces Syst</source>. (<year>2019</year>) <volume>32</volume>:<fpage>1</fpage>&#x2013;<lpage>11</lpage>.</citation>
</ref>
<ref id="ref10">
<label>10.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sarica</surname> <given-names>A</given-names></name> <name><surname>Cerasa</surname> <given-names>A</given-names></name> <name><surname>Quattrone</surname> <given-names>A</given-names></name></person-group>. <article-title>Random Forest algorithm for the classification of neuroimaging data in Alzheimer's disease: a systematic review</article-title>. <source>Front Aging Neurosci</source>. (<year>2017</year>) <volume>9</volume>:<fpage>329</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fnagi.2017.00329</pub-id>, PMID: <pub-id pub-id-type="pmid">29056906</pub-id></citation>
</ref>
<ref id="ref11">
<label>11.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tombaugh</surname> <given-names>TN</given-names></name> <name><surname>McIntyre</surname> <given-names>NJ</given-names></name></person-group>. <article-title>The mini-mental state examination: a comprehensive review</article-title>. <source>J Am Geriatr Soc</source>. (<year>1992</year>) <volume>40</volume>:<fpage>922</fpage>&#x2013;<lpage>35</lpage>. doi: <pub-id pub-id-type="doi">10.1111/j.1532-5415.1992.tb01992.x</pub-id></citation>
</ref>
<ref id="ref12">
<label>12.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kato</surname> <given-names>H</given-names></name> <name><surname>Takahashi</surname> <given-names>Y</given-names></name> <name><surname>Iseki</surname> <given-names>C</given-names></name> <name><surname>Igari</surname> <given-names>R</given-names></name> <name><surname>Sato</surname> <given-names>H</given-names></name> <name><surname>Sato</surname> <given-names>H</given-names></name> <etal/></person-group>. <article-title>Tooth loss-associated cognitive impairment in the elderly: a community-based study in Japan</article-title>. <source>Intern Med</source>. (<year>2019</year>) <volume>58</volume>:<fpage>1411</fpage>&#x2013;<lpage>6</lpage>. doi: <pub-id pub-id-type="doi">10.2169/internalmedicine.1896-18</pub-id>, PMID: <pub-id pub-id-type="pmid">30626824</pub-id></citation>
</ref>
<ref id="ref13">
<label>13.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nomura</surname> <given-names>Y</given-names></name> <name><surname>Tamaki</surname> <given-names>Y</given-names></name> <name><surname>Tanaka</surname> <given-names>T</given-names></name> <name><surname>Arakawa</surname> <given-names>H</given-names></name> <name><surname>Tsurumoto</surname> <given-names>A</given-names></name> <name><surname>Kirimura</surname> <given-names>K</given-names></name> <etal/></person-group>. <article-title>Screening of periodontitis with salivary enzyme tests</article-title>. <source>J Oral Sci</source>. (<year>2006</year>) <volume>48</volume>:<fpage>177</fpage>&#x2013;<lpage>83</lpage>. doi: <pub-id pub-id-type="doi">10.2334/josnusd.48.177</pub-id>, PMID: <pub-id pub-id-type="pmid">17220614</pub-id></citation>
</ref>
<ref id="ref14">
<label>14.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Oyama</surname> <given-names>K</given-names></name> <name><surname>Sakatani</surname> <given-names>K</given-names></name></person-group>. <article-title>Machine learning-based assessment of cognitive impairment using time-resolved near-infrared spectroscopy and basic blood test</article-title>. <source>Front Neurol</source>. (<year>2022</year>) <volume>12</volume>:<fpage>624063</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fneur.2021.624063</pub-id>, PMID: <pub-id pub-id-type="pmid">35153965</pub-id></citation>
</ref>
</ref-list>
</back>
</article>