<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Cardiovasc. Med.</journal-id>
<journal-title>Frontiers in Cardiovascular Medicine</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Cardiovasc. Med.</abbrev-journal-title>
<issn pub-type="epub">2297-055X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fcvm.2024.1399138</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Cardiovascular Medicine</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Performance of federated learning-based models in the Dutch TAVI population was comparable to central strategies and outperformed local strategies</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes"><name><surname>Yordanov</surname><given-names>Tsvetan R.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="cor1">&#x002A;</xref><uri xlink:href="https://loop.frontiersin.org/people/2557578/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author"><name><surname>Ravelli</surname><given-names>Anita C. J.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref><uri xlink:href="https://loop.frontiersin.org/people/1703759/overview" />
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author"><name><surname>Amiri</surname><given-names>Saba</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author"><name><surname>Vis</surname><given-names>Marije</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref><uri xlink:href="https://loop.frontiersin.org/people/2746777/overview" />
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author"><name><surname>Houterman</surname><given-names>Saskia</given-names></name>
<xref ref-type="aff" rid="aff6"><sup>6</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author"><name><surname>Van der Voort</surname><given-names>Sebastian R.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author"><name><surname>Abu-Hanna</surname><given-names>Ameen</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref><uri xlink:href="https://loop.frontiersin.org/people/2720493/overview" />
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib><on-behalf-of>the NHR THI Registration Committee</on-behalf-of>
</contrib-group>
<aff id="aff1"><label><sup>1</sup></label><institution>Department of Medical Informatics, Amsterdam University Medical Centers, University of Amsterdam</institution>, <addr-line>Amsterdam</addr-line>, <country>Netherlands</country></aff>
<aff id="aff2"><label><sup>2</sup></label><institution>Amsterdam Public Health Research Institute, Amsterdam University Medical Centers, University of Amsterdam</institution>, <addr-line>Amsterdam</addr-line>, <country>Netherlands</country></aff>
<aff id="aff3"><label><sup>3</sup></label><institution>Informatics Institute, University of Amsterdam</institution>, <addr-line>Amsterdam</addr-line>, <country>Netherlands</country></aff>
<aff id="aff4"><label><sup>4</sup></label><institution>Department of Cardiology, Amsterdam University Medical Centers, University of Amsterdam</institution>, <addr-line>Amsterdam</addr-line>, <country>Netherlands</country></aff>
<aff id="aff5"><label><sup>5</sup></label><institution>Amsterdam Cardiovascular Sciences Institute, Amsterdam University Medical Centers, University of Amsterdam</institution>, <addr-line>Amsterdam</addr-line>, <country>Netherlands</country></aff>
<aff id="aff6"><label><sup>6</sup></label><institution>Netherlands Heart Registration</institution>, <addr-line>Utrecht</addr-line>, <country>Netherlands</country></aff>
<author-notes>
<fn fn-type="edited-by"><p><bold>Edited by:</bold> Omneya Attallah, Technology and Maritime Transport (AASTMT), Egypt</p></fn>
<fn fn-type="edited-by"><p><bold>Reviewed by:</bold> Shitharth Selvarajan, Leeds Beckett University, United Kingdom</p>
<p>Thierry Caus, University of Picardie Jules Verne, France</p></fn>
<corresp id="cor1"><label>&#x002A;</label><bold>Correspondence:</bold> Tsvetan R. Yordanov <email>t.yordanov@amsterdamumc.nl</email></corresp>
</author-notes>
<pub-date pub-type="epub"><day>05</day><month>07</month><year>2024</year></pub-date>
<pub-date pub-type="collection"><year>2024</year></pub-date>
<volume>11</volume><elocation-id>1399138</elocation-id>
<history>
<date date-type="received"><day>11</day><month>03</month><year>2024</year></date>
<date date-type="accepted"><day>14</day><month>06</month><year>2024</year></date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2024 Yordanov, Ravelli, Amiri, Vis, Houterman, Van der Voort and Abu-Hanna.</copyright-statement>
<copyright-year>2024</copyright-year><copyright-holder>Yordanov, Ravelli, Amiri, Vis, Houterman, Van der Voort and Abu-Hanna</copyright-holder><license license-type="open-access" xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<sec><title>Background</title>
<p>Federated learning (FL) is a technique for learning prediction models without sharing records between hospitals. Compared to centralized training approaches, the adoption of FL could negatively impact model performance.</p>
</sec>
<sec><title>Aim</title>
<p>This study aimed to evaluate four types of multicenter model development strategies for predicting 30-day mortality for patients undergoing transcatheter aortic valve implantation (TAVI): (1) <italic>central</italic>, learning one model from a centralized dataset of all hospitals; (2) <italic>local</italic>, learning one model per hospital; (3) <italic>federated averaging</italic> (<italic>FedAvg</italic>), averaging of local model coefficients; and (4) <italic>ensemble</italic>, aggregating local model predictions.</p>
</sec>
<sec><title>Methods</title>
<p>Data from all 16 Dutch TAVI hospitals from 2013 to 2021 in the Netherlands Heart Registration (NHR) were used. All approaches were internally validated. For the <italic>central</italic> and federated approaches, external geographic validation was also performed. Predictive performance in terms of discrimination [the area under the ROC curve (AUC-ROC, hereafter referred to as AUC)] and calibration (intercept and slope, and calibration graph) was measured.</p>
</sec>
<sec><title>Results</title>
<p>The dataset comprised 16,661 TAVI records with a 30-day mortality rate of 3.4&#x0025;. In internal validation the AUCs of <italic>central</italic>, <italic>local</italic>, <italic>FedAvg</italic>, and <italic>ensemble</italic> models were 0.68, 0.65, 0.67, and 0.67, respectively. The <italic>central</italic> and <italic>local</italic> models were miscalibrated by slope, while the <italic>FedAvg</italic> and <italic>ensemble</italic> models were miscalibrated by intercept. During external geographic validation, <italic>central</italic>, <italic>FedAvg</italic>, and <italic>ensemble</italic> all achieved a mean AUC of 0.68. Miscalibration was observed for the <italic>central</italic>, <italic>FedAvg</italic>, and <italic>ensemble</italic> models in 44&#x0025;, 44&#x0025;, and 38&#x0025; of the hospitals, respectively.</p>
</sec>
<sec><title>Conclusion</title>
<p>Compared to centralized training approaches, FL techniques such as <italic>FedAvg</italic> and <italic>ensemble</italic> demonstrated comparable AUC and calibration. The use of FL techniques should be considered a viable option for clinical prediction model development.</p>
</sec>
</abstract>
<kwd-group>
<kwd>federated learning</kwd>
<kwd>multicenter</kwd>
<kwd>prediction models</kwd>
<kwd>TAVI</kwd>
<kwd>distributed machine learning</kwd>
<kwd>privacy-preserving algorithms</kwd>
<kwd>risk prediction</kwd>
<kwd>EHR</kwd>
</kwd-group>
<counts>
<fig-count count="3"/>
<table-count count="1"/><equation-count count="0"/><ref-count count="26"/><page-count count="11"/><word-count count="0"/></counts><custom-meta-wrap><custom-meta><meta-name>section-at-acceptance</meta-name><meta-value>General Cardiovascular Medicine</meta-value></custom-meta></custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro"><label>1</label><title>Introduction</title>
<p>The increasing adoption of electronic health records (EHRs) across healthcare facilities has led to a wealth of data that can be harnessed for developing prediction models for various medical applications. Such models may improve patient stratification, inform clinical decision-making, and ultimately enhance patient outcomes. In the field of cardiovascular medicine, combining records from multiple centers has successfully been used in training clinical prediction models (CPMs) (<xref ref-type="bibr" rid="B1">1</xref>). Such multicenter models tend to generalize better and are more robust than those derived from individual centers. Although models trained on data from a single center may perform well within their local hospital settings, they require a large number of records for training, and their performance often deteriorates when applied to new centers or other patient populations. However, sharing patient data between centers is not always straightforward. Concerns about patient privacy, the implementation of new regulations such as the General Data Protection Regulation (GDPR), and the challenges of integrating data from different centers all pose significant challenges. There is a growing need to implement strategies for training prediction models on multiple datasets without sharing records between them.</p>
<p>Federated learning (FL) has emerged as a promising approach to address this challenge. FL is a machine learning approach that enables multiple parties to build a shared prediction model without needing to exchange patient data.</p>
<p>However, implementing FL comes with its own set of challenges. Aside from logistical and communication issues, an important question is whether FL has a detrimental impact on the quality of learned models (<xref ref-type="bibr" rid="B2">2</xref>). While promising, the impact of FL on model quality has yet to be thoroughly examined in various areas of medicine.</p>
<p>Understanding the potential benefits and limitations of FL in developing multicenter prediction models helps facilitate a more effective and privacy-preserving use of electronic patient data in risk prediction. To that end, our analysis investigates the potential of FL as a viable strategy for multicenter prediction model development.</p>
<p>FL has rarely been studied in the cardiovascular context (<xref ref-type="bibr" rid="B3">3</xref>&#x2013;<xref ref-type="bibr" rid="B5">5</xref>) and not yet in the transcatheter aortic valve implantation (TAVI) population, which is the focus of this study. TAVI is a relatively new and minimally invasive treatment for severe aortic valve stenosis. The Netherlands Heart Registration (NHR) is a centralized registry that holds records of all cardiac interventions performed in the Netherlands, including those of TAVI patients who are treated in the 16 hospitals performing this operation. Across these 16 hospitals, the TAVI patient population could vary for a number of reasons, such as regional population demographic differences.</p>
<p>Risk prediction models for TAVI patients have been developed using data originating from a single hospital (<xref ref-type="bibr" rid="B6">6</xref>) or combining records from multiple centers (<xref ref-type="bibr" rid="B1">1</xref>, <xref ref-type="bibr" rid="B7">7</xref>, <xref ref-type="bibr" rid="B8">8</xref>). In a previous study, we evaluated the performance of one such centralized, multicenter TAVI early-mortality CPM and observed the model to have a moderate degree of external performance variability, most of which could be attributed to differences in hospital case-mix (<xref ref-type="bibr" rid="B9">9</xref>). However, the performance of such models in an FL approach, compared to a centralized or local approach, remains unknown.</p>
<p>We aimed to evaluate the impact of two important FL techniques: federated averaging (<italic>FedAvg</italic>) (<xref ref-type="bibr" rid="B10">10</xref>) and mean ensemble (henceforth referred to as <italic>ensemble</italic>) (<xref ref-type="bibr" rid="B11">11</xref>), explained further in the Materials and methods section, on the predictive performance of TAVI risk prediction models. This performance is compared to a <italic>centralized model</italic> and <italic>local</italic> center-specific models (<xref ref-type="table" rid="T1">Table&#x00A0;1</xref>).</p>
<table-wrap id="T1" position="float"><label>Table 1</label>
<caption><p>Method overview of model development strategies with respect to types of data sharing and validation performance evaluation (the main differences and similarities between the four model strategies used in the current experiments are shown).</p></caption>
<table frame="hsides" rules="groups">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th valign="top" align="left" colspan="4"/>
<th valign="top" align="center" colspan="4">Federated learning</th>
</tr>
<tr>
<th valign="top" align="center"/>
<th valign="top" align="center">Model strategy</th>
<th valign="top" align="center"><italic>Central</italic></th>
<th valign="top" align="center"><italic>Local</italic></th>
<th valign="top" align="center" colspan="2"><italic>FedAvg</italic></th>
<th valign="top" align="center" colspan="2"><italic>Ensemble</italic></th>
</tr>
<tr>
<th valign="top" align="center" colspan="2">Aspect</th>
<th valign="top" align="center"/>
<th valign="top" align="center"/>
<th valign="top" align="center">No recalibration</th>
<th valign="top" align="center">Recalibration</th>
<th valign="top" align="center">No recalibration</th>
<th valign="top" align="center">Recalibration</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left" rowspan="5">Sharing</td>
<td valign="top" align="left">Predictor data</td>
<td valign="top" align="left">Yes (by design)</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">No</td>
</tr>
<tr>
<td valign="top" align="left">Outcome data</td>
<td valign="top" align="left">Yes (by design)</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">Yes</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">Yes</td>
</tr>
<tr>
<td valign="top" align="left">Model parameters</td>
<td valign="top" align="left">Yes (by design)</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">Yes</td>
<td valign="top" align="left">Yes</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">No</td>
</tr>
<tr>
<td valign="top" align="left">Predictions</td>
<td valign="top" align="left">Yes (by design)</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">Yes</td>
<td valign="top" align="left">Yes</td>
<td valign="top" align="left">Yes</td>
</tr>
<tr>
<td valign="top" align="left">Optional: other model parameters</td>
<td valign="top" align="left">Central imputation</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">Local imputation</td>
<td valign="top" align="left">Local imputation; central recalibration</td>
<td valign="top" align="left">Local imputation</td>
<td valign="top" align="left">Local imputation; central recalibration</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">Calibration</td>
<td valign="top" align="left">Yes (by design)</td>
<td valign="top" align="left">Yes (by design, per center)</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">No</td>
</tr>
<tr>
<td valign="top" align="left"/>
<td valign="top" align="left">Recalibration</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">Local, central, federated</td>
<td valign="top" align="left">No</td>
<td valign="top" align="left">Local, central, federated</td>
</tr>
<tr>
<td valign="top" align="left" rowspan="3">Validation</td>
<td valign="top" align="left">Stacking predictions CV</td>
<td valign="top" align="left">Per (test) fold</td>
<td valign="top" align="left">Per (test) fold per center<xref ref-type="table-fn" rid="table-fn2"><sup>a</sup></xref></td>
<td valign="top" align="left">Per (test) fold</td>
<td valign="top" align="left">Per (test) fold</td>
<td valign="top" align="left">Per (test) fold</td>
<td valign="top" align="left">Per (test) fold</td>
</tr>
<tr>
<td valign="top" align="left">Stacking predictions LCOA</td>
<td valign="top" align="left">Per external center</td>
<td valign="top" align="left">Not Applicable</td>
<td valign="top" align="left">Per external center</td>
<td valign="top" align="left">Per external center</td>
<td valign="top" align="left">Per external center</td>
<td valign="top" align="left">Per external center</td>
</tr>
<tr>
<td valign="top" align="left">Obtaining performance</td>
<td valign="top" align="left">On stacked predictions</td>
<td valign="top" align="left">On stacked predictions</td>
<td valign="top" align="left">On stacked predictions</td>
<td valign="top" align="left">On stacked predictions</td>
<td valign="top" align="left">On stacked predictions</td>
<td valign="top" align="left">On stacked predictions</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="table-fn1"><p>Clarifications&#x2014;Sharing of predictor data: the predictor variable values of records from a center dataset. Sharing of outcome data: the outcome variable value of records from a center dataset. Sharing of model parameters: the weights and intercepts (coefficients) from a center-learned model. Sharing of predictions: the predicted probabilities from a center-learned model. Sharing of other model parameters (optional): imputation model for missing values, recalibration model. Calibration: does the model fitting process also calibrate the model predictions? Recalibration: recalibration (of any kind) applied to the model after its fitting? Stacking predictions CV: during CV, how were the model predictions from the test folds stacked (combined) before computing performance metrics? Stacking predictions LCOA: during LCOA, how were the model predictions from the test centers stacked (combined) before computing performance metrics? Obtaining performance: when computing performance metrics for a model, what set of predictions were used?</p></fn>
<fn id="table-fn2"><label><sup>a</sup></label><p>In the case of local models, for each individual center, the model predictions from all of its test folds during CV were stacked together. Each set of these stacked predictions was then used to obtain the per-center local model performance measures. Pooled performance across all center local models was then calculated with a REMA pooling of the individual center performance results.</p></fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s2" sec-type="methods"><label>2</label><title>Materials and methods</title>
<p>This study adhered to the Transparent Reporting of a Multivariable Prediction Model for Individual Prognosis or Diagnosis (TRIPOD) statement (<xref ref-type="bibr" rid="B12">12</xref>). This study meets all five of the CODE-EHR minimum framework standards for the use of structured healthcare data in clinical research (<xref ref-type="bibr" rid="B13">13</xref>).</p>
<sec id="s2a"><label>2.1</label><title>Dataset</title>
<p>In this nationwide retrospective multicenter cohort study, we included all patients who had a TAVI intervention in any Dutch hospital for the 9-year period from 1 January 2013 to 31 December 2021. Data were collected by the NHR (<xref ref-type="bibr" rid="B14">14</xref>). Permission was granted for this study to use the data and include a pseudonymized code indicating the center in the dataset (<xref ref-type="sec" rid="s11">Supplementary Appendix A</xref>).</p>
<p>The outcome of interest was the 30-day post-operative mortality. Mortality data were obtained by checking the regional municipal administration registration, Basisregistratie Personen (BRP).</p>
<p>Patients lacking the outcome measurement of 30-day mortality were excluded.</p>
<p>No ethical approval was needed according to the Dutch central committee of Human Research, as the study only used previously collected cohort registry data. All data in this study were fully anonymized before we accessed them. Approval for this study was granted by the Committee of Research and Ethics of the Netherlands Heart Registry on 2 February 2021.</p>
</sec>
<sec id="s2b"><label>2.2</label><title>Model strategies</title>
<p>Four different model development strategies (henceforth referred to as models) were considered in our experiments (<xref ref-type="table" rid="T1">Table&#x00A0;1</xref>). A <italic>central</italic> model was derived using the combined records from all hospitals (<xref ref-type="sec" rid="s11">Supplementary Figure S1</xref>). The derivation of such a model consists of two steps: (1) performing variable selection from the list of candidate predictor variables (explained further in Section <xref ref-type="sec" rid="s2c">2.3</xref>); and (2) fitting predictor variable coefficients. Leveraging the entire dataset enables capturing relationships between predictors and outcomes across multiple centers. Due to the nature of the central design, all data between hospitals are shared, including individual patient record variables.</p>
<p>In the next strategy, multiple <italic>local</italic> models were trained, one for each hospital&#x0027;s dataset (<xref ref-type="sec" rid="s11">Supplementary Figure S2</xref>). The derivation of each center local model would follow in much the same steps as in the centralized model strategy. As the <italic>local</italic> models are specific to each hospital, they avoid the need to share any data between centers.</p>
<p>In addition to these baselines, we considered two popular FL techniques: <italic>FedAvg</italic> and an <italic>ensemble</italic> model. To conceptualize the idea behind <italic>FedAvg</italic>, one can think of averaging the knowledge of a classroom where students train on their schoolwork and then share their key learnings with a central teacher who combines them to create a better understanding for everyone. In the case of the current study, each participating center trains a local model for one epoch (that is, one pass on all the data) and shares its model parameters with a central server (<xref ref-type="bibr" rid="B10">10</xref>). Once each center has shared model parameters, model updates are aggregated by the central server (<xref ref-type="sec" rid="s11">Supplementary Figure S3</xref>). This new aggregated model is then sent back to the centers for further local training in the next epoch. This process continues until convergence or a pre-specified number of epochs is reached.</p>
<p>The <italic>ensemble</italic> model approach is similar to combining votes from a diverse group, where the final prediction is the most popular choice (similar to how a majority vote wins an election). For the <italic>ensemble</italic> model in this case, a local model is fitted on each center&#x0027;s data. The ensemble&#x0027;s prediction for each patient is then formed by averaging the predictions of each local model from all hospitals (<xref ref-type="sec" rid="s11">Supplementary Figure S4</xref>) (<xref ref-type="bibr" rid="B15">15</xref>). With this strategy, only hospital-level models are transmitted between the centers.</p>
<p>For all model development strategies, we fitted logistic regression models with Least Absolute Shrinkage and Selection Operator (LASSO) penalization. This approach results in automatically selecting variables deemed predictive of the outcome.</p>
<sec id="s2b1"><label>2.2.1</label><title>Model recalibration</title>
<p><xref ref-type="table" rid="T1">Table&#x00A0;1</xref> provides a framework for summarizing, among others, the aspect of model recalibration. Ensuring a model is well-calibrated before its application in practice is critical. If a decision is to be made based on a predicted probability from a model, then the predicted probability should be as close as possible to the true patient risk probability. This is what calibration performance measures.</p>
<p>The recalibration aspect describes the addition of a final step to the model derivation process, where recalibration of the intercept and slope of the linear predictor is performed using the model&#x0027;s predictions on the training data. The derived recalibration function is used thereafter whenever the model makes predictions. Specifically, in recalibration, we align the true outcomes from the different centers with their corresponding model predictions followed by fitting the recalibration function. This is done by fitting a logistic regression model in which the predicted 30-day mortality probability is the sole covariate to predict the true 30-day mortality outcome. As listed in <xref ref-type="table" rid="T1">Table&#x00A0;1</xref>, recalibration could be done in a local, central, or federated manner. In the local case, a recalibration function would be learned for each individual hospital and then used for adjusting model predictions for patients belonging to the corresponding hospital. In the central case, a single recalibration function would be learned on the combined training dataset predictions from all hospitals. In the federated recalibration approach, a federated learning strategy (such as <italic>FedAvg</italic>) would be used to derive a single recalibration function while also avoiding the need to share the uncalibrated predictions between centers.</p>
<p>In our main analysis, we focused on the results from the FL techniques with central recalibration and did not investigate all options, such as learning the recalibration function in a federated manner.</p>
</sec>
</sec>
<sec id="s2c"><label>2.3</label><title>Candidate predictor variables</title>
<p>The TAVI dataset included variables for patient characteristics (e.g., age, sex, and body mass index), lab test results (e.g., serum creatinine), relevant medical history (e.g., chronic lung disease), and procedure characteristics (e.g., access route and use of anesthesia) (<xref ref-type="sec" rid="s11">Supplementary Table S1</xref>). All 33 candidate predictors were collected prior to the intervention. Threshold values for the body surface area (BSA) were used in summarizing patient characteristics.</p>
<p>In all model strategies, we used LASSO to perform automatic variable selection. In the case of <italic>FedAvg</italic>, LASSO was first used on each hospital dataset. Later, the selected predictors from each hospital-local LASSO were aggregated via center-weighted voting and a center agreement strength hyperparameter (<xref ref-type="sec" rid="s11">Supplementary Methods S1</xref>).</p>
</sec>
<sec id="s2d"><label>2.4</label><title>Experimental evaluation</title>
<p>We adopted two primary evaluation strategies to fit the type of evaluation: a 10-fold cross-validation (CV) approach for the internal validation and leave-center-out analysis (LCOA) for the geographic validation.</p>
<p>In some cases, a hospital-local model could not be fitted due to the insufficient number of records for the prevalence of outcomes. We compared the performance of the <italic>local</italic> model to the other models only in cases where a <italic>local</italic> model was successfully derived and reported the cases where fitting a <italic>local</italic> model failed.</p>
<sec id="s2d1"><label>2.4.1</label><title>Cross-validation</title>
<p>For cross-validation, we first randomly partitioned the entire dataset into ten equal subsets, stratified by the outcome. In each iteration of the CV, we utilized nine subsets (90&#x0025; of the records) for model training and held out the remaining subset (10&#x0025; of the records) for testing. This process was repeated 10 times, each with a different test set. In the case of <italic>local</italic>, <italic>FedAvg,</italic> and <italic>ensemble</italic>, the entire dataset was first partitioned by hospital, and thereafter each hospital dataset was randomly partitioned into 10 equal subsets stratified by the outcome.</p>
</sec>
<sec id="s2d2"><label>2.4.2</label><title>Leave-center-out analysis</title>
<p>For the federated strategies, we conducted a LCOA for a more robust external geographic validation (<xref ref-type="bibr" rid="B9">9</xref>). In this approach, we created as many train/test dataset pairs as there were hospitals in the dataset. For each pair, the training set encompassed records from all hospitals but one (the excluded hospital), while the test set solely contained records from the excluded hospital. This method allows us to evaluate how well each model performed when applied to a new center.</p>
</sec>
<sec id="s2d3"><label>2.4.3</label><title>Pooling results</title>
<p>In the context of CV, mean metric values and confidence intervals (CIs) for a model were derived from the individual metric results per test set. During this process, the predictive performance of each model was computed separately for each test set, generating 10 metric values. These 10 values were then averaged to arrive at the mean pooled metric, and their standard deviation was used to compute a 95&#x0025; CI.</p>
<p>During the LCOA, performance was calculated per external hospital and then pooled via random effects meta-analysis (REMA) with hospital as the random effect to give a mean estimate and 95&#x0025; CI.</p>
</sec>
<sec id="s2d4"><label>2.4.4</label><title>Performance metrics</title>
<p>Discrimination was evaluated using the area under the ROC curve (AUC-ROC, henceforth referred to as AUC). The AUC metric summarizes a model&#x0027;s ability to discriminate between events and cases. It involves sensitivity (also called recall in information retrieval) and specificity across all possible threshold values. Calibration was evaluated by the Cox method using the calibration intercept and slope and their corresponding 95&#x0025; CIs (<xref ref-type="bibr" rid="B16">16</xref>). A model&#x0027;s predictions were deemed to be miscalibrated if either (1) the 95&#x0025; CI for its Cox calibration intercept did not contain the value zero (miscalibration by intercept) or (2) the 95&#x0025; CI of its intercept did contain the value zero, but the 95&#x0025; CI for its Cox calibration slope did not contain the value one (miscalibration by the slope) (<xref ref-type="bibr" rid="B16">16</xref>).</p>
<p>In addition, calibration graphs showing a model&#x0027;s predicted probabilities vs. the observed frequencies of positive outcomes were drawn for visual inspection.</p>
<p>Net reclassification improvement (NRI) was calculated between the predictions of any two models in either validation strategy (CV and LCOA) (<xref ref-type="sec" rid="s11">Supplementary Methods S2</xref>).</p>
</sec>
<sec id="s2d5"><label>2.4.5</label><title>Significance testing</title>
<p>Bootstrapping with 3,000 samples was used to test for a difference in (paired) AUCs between two prediction models (<xref ref-type="bibr" rid="B17">17</xref>). This test was run per AUC of each test fold dataset during CV. Analogously, the test was applied per AUC of each external center dataset in the LCOA.</p>
</sec>
<sec id="s2d6"><label>2.4.6</label><title>Sensitivity analyses</title>
<p>Apart from the main experimental setup, we considered two additional modifications to it in the form of sensitivity analyses.</p>
<p>First, to see what effect the recalibration step was having on the two FL approaches (<italic>FedAvg</italic> and <italic>ensemble</italic>), we evaluated their performance without recalibration in a sensitivity analysis. In a second sensitivity analysis, we excluded hospitals with a low TAVI volume from the dataset and re-evaluated the models&#x2019; performance results. In this case, we defined a low TAVI volume to be any hospital that performed fewer than 10 TAVI procedures in any year of operation after its first year.</p>
</sec>
</sec>
<sec id="s2e"><label>2.5</label><title>Hyperparameter optimization</title>
<p>Hyperparameters for LASSO and <italic>FedAvg</italic> were optimized empirically on the training data (<xref ref-type="sec" rid="s11">Supplementary Methods S3</xref>). For LASSO, we optimized the regularization parameter lambda, while for <italic>FedAvg</italic>, we optimized on the learning rate, number of training epochs, and variable selection agreement strength (<xref ref-type="sec" rid="s11">Supplementary Methods S1</xref>).</p>
</sec>
<sec id="s2f"><label>2.6</label><title>Handling of missing data</title>
<p>Variables with more than 30&#x0025; missing values were not included as predictors.</p>
<p>The remaining missing values were assumed to be missing at random and were imputed using the Chain Equations (<xref ref-type="bibr" rid="B18">18</xref>). As shown in <xref ref-type="table" rid="T1">Table&#x00A0;1</xref>, imputation was done on the center-combined dataset for the <italic>central</italic> model, while for the other models, imputation was handled separately per-center dataset. In both validation strategies (CV and LCOA), missing values were imputed separately on the training and test sets (further information is provided in <xref ref-type="sec" rid="s11">Supplementary Methods S4</xref>).</p>
</sec>
<sec id="s2g"><label>2.7</label><title>Fitting final models</title>
<p>From each of the four strategies, a &#x201C;final&#x201D; version of their model was fitted using records from the complete dataset. We used the resulting final models to report on and compare the predictor variables selected by each model strategy. In the case of <italic>central</italic> and <italic>FedAvg</italic>, the &#x201C;final&#x201D; model comprised of just one single logistic regression model, while for <italic>local</italic>, the &#x201C;final&#x201D; model was a set of <italic>h</italic> hospital-local models (where <italic>h</italic> is the number of hospitals in the dataset). The <italic>ensemble</italic> produced a &#x201C;final&#x201D; model comprised of <italic>h</italic> local models and one top-level model, which averaged the predictions from the <italic>h</italic> hospital-level models.</p>
</sec>
<sec id="s2h"><label>2.8</label><title>Software</title>
<p>All statistical analyses were performed in the R programming language (version 4.2.1) and R studio (version 2023.03.1). The metamean function from the meta package was used for conducting the REMA, and the roc.test function from the pROC package was used for the bootstrap testing for the difference in AUCs between two models. The &#x201C;mice&#x201D; package in R was used for imputing missing values (mice version 3.14.0). All the source code used in this analysis was documented and made openly available on GitHub (<ext-link ext-link-type="uri" xlink:href="https://github.com/tsryo/evalFL">https://github.com/tsryo/evalFL</ext-link>). Experiments were carried out on a desktop machine with 16&#x2005;GB of memory and an i7-10700 2.9&#x2005;GHz processor and took approximately 2&#x2005;days to run.</p>
</sec>
</sec>
<sec id="s3" sec-type="results"><label>3</label><title>Results</title>
<p>The results in this section are structured into four sub-sections. First, we provide summary statistics of the TAVI dataset and its pre-processing. We then report on the models&#x2019; predictive performance measures from cross-validation and LOCA. Third, we report on selected predictor variables in each model type. Finally, we present results from the sensitivity analyses. <xref ref-type="fig" rid="F1">Figure&#x00A0;1</xref> provides a graphical overview of the experimental setup, methods, and key findings from our analyses.</p>
<fig id="F1" position="float"><label>Figure 1</label>
<caption><p>Graphical summary of the dataset used, prediction models considered, validation strategies employed, and main findings for the current study on 30-day mortality risk prediction models for TAVI patients. Key questions are as follows: In the context of multicenter TAVI risk prediction models, what is the impact on model performance from adopting two federated learning strategies (<italic>FedAvg</italic> and <italic>ensemble</italic>) compared to <italic>central</italic> and <italic>local</italic>-only model strategies? Key findings are as follows: Both federated learning strategies (<italic>FedAvg</italic> and <italic>ensemble</italic>) had comparable performance, in terms of discrimination and calibration, to that <italic>central</italic> models and outperformed the <italic>local</italic>-only models. Take-home message is as follows: Use of federated learning techniques should be considered a viable option for TAVI patient clinical prediction model development.</p></caption>
<graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="fcvm-11-1399138-g001.tif"/>
</fig>
<p>Between 2013 and 2021, there were 17,689 patients with a TAVI intervention in one of the 16 Dutch hospitals (labeled A&#x2013;P). In total, 1,028 patients lacked an outcome measurement; therefore, these patients were excluded from the analysis.</p>
<p>The final TAVI dataset consisted of 16,661 records with an average outcome prevalence of 30-day mortality of 3.4&#x0025;. The prevalence ranged from 1.2&#x0025; to 5.8&#x0025; between hospitals, with an intra-quartile range (IQR) of 2.8&#x0025;&#x2013;3.9&#x0025; (<xref ref-type="sec" rid="s11">Supplementary Table S2</xref>). From the list of all 33 candidate predictor variables, only the variable of frailty status was excluded for having more than 30&#x0025; of its records missing.</p>
<sec id="s3a"><label>3.1</label><title>Model performance</title>
<p>Predictive performance results for each model across both internal validation (CV) and external validation (LCOA) are reported in the following.</p>
<p>Due to the lower volume of TAVI records and low outcome prevalence, fitting a hospital-local model failed in some iterations of the CV analysis. From the 10 folds during CV, a local model could not be fitted in 100&#x0025; of folds in centers L and P, 90&#x0025; of folds in center M, 70&#x0025; in center I, 60&#x0025; in center K, 40&#x0025; in N, 20&#x0025; in C and J, and 10&#x0025; in centers F and O (<xref ref-type="sec" rid="s11">Supplementary Table S3</xref>).</p>
<sec id="s3a1"><label>3.1.1</label><title>Discrimination</title>
<sec id="s3a1a"><label>3.1.1.1</label><title>Cross-validation</title>
<p><italic>The central</italic> model had the highest mean AUC during internal validation (0.68, 95&#x0025; CI: 0.66&#x2013;0.70), followed by <italic>FedAvg</italic> (0.67, 95&#x0025; CI: 0.65&#x2013;0.68), <italic>ensemble</italic> (0.67, 95&#x0025; CI: 0.66&#x2013;0.68), and <italic>local</italic> (0.65, 95&#x0025; CI: 0.63&#x2013;0.67) (<xref ref-type="fig" rid="F2">Figure&#x00A0;2A</xref>, <xref ref-type="sec" rid="s11">Supplementary Table S4</xref>). Comparing model AUCs for significant differences with the bootstrap method showed <italic>central</italic> to outperform <italic>local</italic> and <italic>FedAvg</italic> in two (20&#x0025;) and 1 (10&#x0025;) out of 10 folds, respectively (<xref ref-type="sec" rid="s11">Supplementary Table S5</xref>). AUC results of local models ranged from 0.52 to 0.84 across centers (<xref ref-type="sec" rid="s11">Supplementary Table S6</xref>).</p>
<fig id="F2" position="float"><label>Figure 2</label>
<caption><p>AUCs from cross-validation (<bold>A</bold>) and leave-center-out analysis (<bold>B</bold>) of TAVI patient 30-day mortality risk prediction models. Next to each model&#x0027;s name, its mean AUC is given.</p></caption>
<graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="fcvm-11-1399138-g002.tif"/>
</fig>
</sec>
<sec id="s3a1b"><label>3.1.1.2</label><title>Leave-center-out</title>
<p>The meta-analysis pooled mean AUC from the LCOA was 0.68 (95&#x0025; CI: 0.66&#x2013;0.70) for the <italic>central</italic> model (<xref ref-type="fig" rid="F2">Figure&#x00A0;2B</xref>, <xref ref-type="sec" rid="s11">Supplementary Table S4</xref>), and AUC values ranged from 0.62 to 0.76 between external centers (<xref ref-type="sec" rid="s11">Supplementary Table S7</xref>). <italic>FedAvg</italic> also had a mean AUC of 0.68 (95&#x0025; CI: 0.65&#x2013;0.70), and its AUC values for the individual centers ranged from 0.56 to 0.80. For the <italic>ensemble</italic>, the mean AUC was 0.67 (95&#x0025; CI: 0.65&#x2013;0.70), and AUC values ranged from 0.46 to 0.76 between external hospitals.</p>
<p>Bootstrap AUC testing from LCOA showed that both <italic>FedAvg</italic> and <italic>ensemble</italic> outperformed <italic>central</italic> in one hospital (P) (<xref ref-type="sec" rid="s11">Supplementary Table S8</xref>). In another two centers (C and N), <italic>FedAvg</italic> outperformed <italic>ensemble</italic>, and in one hospital (H), <italic>central</italic> outperformed <italic>ensemble</italic>.</p>
</sec>
</sec>
<sec id="s3a2"><label>3.1.2</label><title>Calibration</title>
<p>Calibration performance results varied across the different models and validation strategies.</p>
<p>Calibration graphs showed that all models suffered from over-prediction in the higher-risk ranges. To better inspect the lower-risk probabilities (found in the majority of the records), calibration graphs for a model were visualized excluding the top 2.5&#x0025; of highest predicted probabilities.</p>
<sec id="s3a2a"><label>3.1.2.1</label><title>Cross-validation</title>
<p>From CV, all models showed miscalibration by slope when evaluated via the Cox method, and only <italic>ensemble</italic> showed miscalibration from its intercept (<xref ref-type="fig" rid="F3">Figure&#x00A0;3A</xref>, <xref ref-type="sec" rid="s11">Supplementary Table S9</xref>). <italic>Central</italic> had a calibration intercept of &#x2212;0.003 (95&#x0025; CI: &#x2212;0.03 to 0.02) and a calibration slope of 0.89 (95&#x0025; CI: 0.80&#x2013;0.98). <italic>Local</italic> models had a mean calibration intercept in CV of &#x2212;0.01 (95&#x0025; CI: &#x2212;0.04 to 0.01) but had a poor calibration slope of 0.54 (95&#x0025; CI: 0.40&#x2013;0.67). From the 14 hospitals where local models could be fitted (88&#x0025; of all the hospitals in the dataset), miscalibration occurred in 13 of them (93&#x0025;) (<xref ref-type="sec" rid="s11">Supplementary Table S10</xref>). For <italic>FedAvg</italic>, the calibration intercept was &#x2212;0.04 (95&#x0025; CI: &#x2212;0.07 to &#x2212;0.02), and the calibration slope was 0.86 (95&#x0025; CI: 0.78&#x2013;0.93). <italic>Ensemble</italic> models showed a calibration intercept of &#x2212;0.04 (95&#x0025; CI: &#x2212;0.06 to &#x2212;0.01) and a calibration slope of 0.89 (95&#x0025; CI: 0.82&#x2013;0.96). Calibration graphs of the four models showed <italic>central</italic> to most closely resemble the ideal calibration graph, followed by <italic>local</italic> (<xref ref-type="fig" rid="F3">Figure&#x00A0;3A</xref>).</p>
<fig id="F3" position="float"><label>Figure 3</label>
<caption><p>Calibration graphs from cross-validation (<bold>A</bold>) and leave-center-out analysis (<bold>B</bold>) results. The calibration graphs are shown after trimming the 2.5&#x0025; highest predicted probabilities to focus on the bulk of the sample. The legend shows the calibration intercept and slope of each model, respectively, as obtained from the Cox method (<xref ref-type="bibr" rid="B16">16</xref>). Mean values for cross-validation were obtained by computing performance metrics on the combined predictions from all corresponding test sets. An asterisk (&#x002A;) is placed after the names of the models where miscalibration was detected by way of the Cox method, and a hat (^) symbol is placed if miscalibration occurred in the calibration slope. The calibration intercept and slope values shown in the legend are calculated from all the predictions, including the 2.5&#x0025; highest predicted probabilities.</p></caption>
<graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="fcvm-11-1399138-g003.tif"/>
</fig>
</sec>
<sec id="s3a2b"><label>3.1.2.2</label><title>Leave-center-out</title>
<p>In the LCOA, the mean meta-analysis pooled calibration intercept for the <italic>central</italic> model was &#x2212;0.01 (95&#x0025; CI: &#x2212;0.16 to 0.15) and the calibration slope was 0.88 (95&#x0025; CI: 0.76&#x2013;1.01) (<xref ref-type="sec" rid="s11">Supplementary Table S9</xref>). Miscalibration was detected in 44&#x0025; of external hospital validations for the <italic>central</italic> model (<xref ref-type="sec" rid="s11">Supplementary Table S1</xref><xref ref-type="table" rid="T1">1</xref>). In <italic>FedAvg</italic>, the calibration intercept was 0.01 (95&#x0025; CI: &#x2212;0.16 to 0.18), the calibration slope was 1.04 (95&#x0025; CI: 0.89&#x2013;1.19), and miscalibration occurred in 44&#x0025; of the external hospitals. The <italic>ensemble</italic> model had a calibration intercept of 0.01 (95&#x0025; CI: &#x2212;0.14 to 0.16), the calibration slope was 0.97 (95&#x0025; CI: 0.82&#x2013;1.12), and miscalibration was seen in 38&#x0025; of centers.</p>
<p>Similar to the calibration graph from CV, the calibration graph in LCOA showed <italic>central</italic> to most closely follow the line of the ideal calibration graph (<xref ref-type="fig" rid="F3">Figure&#x00A0;3B</xref>).</p>
</sec>
</sec>
<sec id="s3a3"><label>3.1.3</label><title>Net reclassification improvement</title>
<p>NRI comparison results showed <italic>central</italic> models to outperform the rest in predicting positive outcomes during CV and LCOA. From CV, <italic>local</italic> models were superior to the rest for predicting negative outcomes, while during LCOA, <italic>central</italic> models showed a higher NRI than the rest for negative outcomes. In both CV and LCOA, <italic>FedAvg</italic> beat <italic>ensemble</italic> in the case-negative group. In the LCOA case-positive group, <italic>ensemble</italic> had a better NRI than <italic>FedAvg</italic>. Full results from comparing model predictions using NRI can be found in <xref ref-type="sec" rid="s11">Supplementary Results S1</xref>.</p>
</sec>
</sec>
<sec id="s3b"><label>3.2</label><title>Predictors selected</title>
<p>From the final models fitted using the whole dataset, <italic>FedAvg</italic> and <italic>ensemble</italic> both used the same set of 20 variables (<xref ref-type="sec" rid="s11">Supplementary Tables S12, S13</xref>).</p>
<p>The hospital-<italic>local</italic> models used between 2 and 14 variables (IQR 4&#x2013;9). Selected variables occurring in at least 50&#x0025; of all <italic>local</italic> models were age, left ventricular ejection fraction (LVEF), body mass index (BMI), BSA, and procedure access route. In the case of two hospitals (L and P), no <italic>local</italic> model could be trained due to insufficient TAVI record volumes with a positive outcome.</p>
<p>The <italic>central</italic> model selected 19 predictor variables (<xref ref-type="sec" rid="s11">Supplementary Table S12</xref>), which comprised 13 predictors already selected by the other strategies, plus an additional 6 new predictors [Canadian Cardiovascular Society (CCS) class IV angina, critical preoperative state, dialysis, previous aortic valve surgery, previous permanent pacemaker, and recent myocardial infarction]. More information on the considered and selected variables can be found in <xref ref-type="sec" rid="s11">Supplementary Table S14</xref>.</p>
</sec>
<sec id="s3c"><label>3.3</label><title>Sensitivity analyses</title>
<p>In such an analysis, where the recalibration step from model training was skipped for <italic>FedAvg</italic> and <italic>ensemble</italic> models, we saw that both performed significantly worse in calibration but not in AUC. In the second sensitivity analysis, where three low-volume heart centers (P, O, and N) were excluded from the analysis, the performance of the <italic>ensemble</italic> model remained mostly unchanged, while AUC was negatively affected for the other models. From this same analysis, an improvement was observed in calibration during CV for <italic>FedAvg</italic> and a worsening for <italic>central</italic> was observed during LCOA. Full results from the two sensitivity analyses are available in <xref ref-type="sec" rid="s11">Supplementary Results S2</xref>.</p>
</sec>
</sec>
<sec id="s4" sec-type="discussion"><label>4</label><title>Discussion</title>
<sec id="s4a"><label>4.1</label><title>Summary of findings</title>
<p>In this study, we investigated the performance of two FL approaches compared to <italic>central</italic> and <italic>local</italic> approaches for predicting early mortality in TAVI patients. We showed that <italic>FedAvg</italic> and <italic>ensemble</italic> models performed similarly compared to a <italic>central</italic> model. The hospital-<italic>local</italic> models were worse in terms of average AUC compared to the other approaches.</p>
<p>Testing for AUC differences showed the <italic>central</italic> model to outperform <italic>local</italic> and <italic>FedAvg</italic> models but not <italic>ensemble</italic> during internal validation. The <italic>local</italic> models, however, did not significantly outperform the federated ones, suggesting that the AUC performance of <italic>FedAvg</italic> and <italic>ensemble</italic> lied somewhere between that of the <italic>central</italic> and <italic>local</italic> models.</p>
<p><italic>Central</italic> and federated models performed similarly well in terms of calibration, whereas <italic>local</italic> model predictions were more frequently miscalibrated. Furthermore, in two cases, the <italic>local</italic> models could not be fitted due to the low number of positive outcome records in their datasets. Although <italic>local</italic> models were calibrated by design to their corresponding hospital-<italic>local</italic> training datasets (<xref ref-type="table" rid="T1">Table&#x00A0;1</xref>), this was often not sufficient to produce a good calibration on their corresponding test sets. While the federated models may not have been calibrated by design, they offered more options for recalibration (such as global, local, or federated recalibration). This could provide model developers with more fine-grained control over tradeoffs between maintaining data privacy and improving model calibration.</p>
<p>In the main experiments, the choice was made to use the <italic>central</italic> recalibration strategy (as opposed to <italic>local</italic> or federated) for the federated approaches. Although this approach requires the sharing of patient outcome data and model predictions between centers, it does offer the most promising recalibration approach of the three options.</p>
<p>In terms of NRI, there was an observed improvement from <italic>local</italic> to <italic>FedAvg</italic> and <italic>ensemble</italic> to <italic>central</italic> when looking at the outcome-positive group of records during internal validation (<xref ref-type="sec" rid="s11">Supplementary Results S1</xref>).</p>
<p>When comparing the two federated approaches, it is difficult to say that one strategy was better than the other, as both had strengths and weaknesses. In terms of discrimination, <italic>FedAvg</italic> seemed to be slightly superior to the <italic>ensemble</italic> model. For model calibration during internal validation, <italic>FedAvg</italic> and <italic>ensemble</italic> showed near-identical results; however, in the external validation, the <italic>ensemble</italic> approach was miscalibrated in fewer external hospitals.</p>
<p>From an interpretability standpoint, the <italic>FedAvg</italic> model would be preferred to the <italic>ensemble</italic> one, as it delivers a single parametric model with predictor variables and their coefficients. The <italic>ensemble</italic>, on the other hand would, comprise a list of parametric models (which may not all use the same variables), plus a top-level parametric model that combines the outputs from the aforementioned list. While the <italic>ensemble</italic> model is not as easily interpretable immediately, techniques like metamodeling could be useful to bridge this gap (<xref ref-type="bibr" rid="B19">19</xref>).</p>
<p>It is worth noting that, although easily interpretable, the <italic>FedAvg</italic> model was more costly to develop than the <italic>ensemble</italic> one regarding computing resources. Depending on the number of hyperparameter values considered, we saw that the training times for the <italic>FedAvg</italic> model could easily become orders of magnitude larger than those for the <italic>ensemble</italic> model. In the current experiments, we developed our in-house frameworks for both federated approaches and encountered more hurdles with the <italic>FedAvg</italic> strategy&#x2014;these included issues such as model convergence problems and the need to use a more elaborate variable selection strategy, which introduced the need for an additional hyperparameter.</p>
</sec>
<sec id="s4b"><label>4.2</label><title>Strengths and limitations</title>
<p>Our study has several strengths. It is the first study on employing federated learning in the TAVI population and one of the very few FL studies in cardiology. It is also based on a large national registry dataset consisting of all 16 hospitals performing TAVI interventions in the Netherlands. In addition, we provided a framework (in <xref ref-type="table" rid="T1">Table&#x00A0;1</xref>) of the various important elements to consider when adopting FL strategies in this context. We also considered multiple aspects of predictive performance and employed two validation strategies to prevent overfitting and optimism in the results. Finally, two sensitivity analyses were conducted to understand the robustness of our findings.</p>
<p>Our research also has limitations. We looked at FL prediction models for TAVI patients, considering only one outcome: the 30-day mortality. However, early post-operative mortality is a relevant and important clinical outcome in the TAVI patient group.</p>
<p>From a privacy perspective of local hospitals, we did not evaluate additional techniques that could be used to preserve patient privacy at local centers (such as differential privacy).</p>
<p>We also considered only one type of ensemble approach (mean volume-weighted ensemble) and only one type of federated aggregation approach (<italic>FedAvg</italic>),although a number of alternatives are available in both cases (<xref ref-type="bibr" rid="B11">11</xref>, <xref ref-type="bibr" rid="B20">20</xref>). Although relatively small, the group of patients excluded from the analysis due to missing outcome values could have somewhat biased our results in model performance. Changes over time in TAVI intervention modalities and patient selection protocols could also have impacted model performance estimates (<xref ref-type="bibr" rid="B21">21</xref>).</p>
<p>Finally, we did not extensively tune hyperparameters, which might have affected the performances of the <italic>FedAvg</italic> and <italic>ensemble</italic> models (<xref ref-type="bibr" rid="B22">22</xref>).</p>
</sec>
<sec id="s4c"><label>4.3</label><title>Comparison with literature</title>
<p>Few studies have investigated the impact of FL in the cardiology domain (<xref ref-type="bibr" rid="B23">23</xref>&#x2013;<xref ref-type="bibr" rid="B25">25</xref>). In only one study, the authors look at risk models for TAVI patients (<xref ref-type="bibr" rid="B23">23</xref>). In this study, Lopes et al. developed non-parametric models for predicting 1-year mortality after TAVI on a dataset from two hospitals. They compared hospital-local model performance against that of federated ensemble models and found the ensemble models to outperform the local ones. Our findings on the ensemble model&#x0027;s superior performance align with the study by Lopes et al. However, we expanded on their findings, first, by evaluating predictive performance with a much larger number of hospitals (16 vs. 2); second, by considering model calibration performance and NRI in addition to AUC; third, by performing additional geographic validation; and finally by considering a centralized model strategy as a baseline in addition to local and federated ones.</p>
<p>Another study by Goto et al. looked at training FL models to detect hypertrophic cardiomyopathy using ECG and echocardiogram data from three hospitals (<xref ref-type="bibr" rid="B24">24</xref>). The authors considered the AUC metric for discrimination and looked at <italic>FedAvg</italic> and local hospital models. They reported that the FL models outperform local models in terms of AUC, something we also observed in the current study.</p>
<p>In other medical domains, FL models have previously been evaluated on their performance compared to models derived from non-FL techniques.</p>
<p>A similar study to ours that described the benefits of using centralized models compared to federated and local ones is that by Vaid et al. (<xref ref-type="bibr" rid="B26">26</xref>). In their study, the authors developed prediction models for COVID-19 patient 7-day mortality outcomes and reported that in five out of five hospital datasets, the models derived from a central development strategy outperformed both local and federated models in terms of AUC. This finding was corroborated in our study for the local models but not for the federated models, which performed on par with the central ones. Differences in the domain of application and in the datasets may explain this. From inspecting the NRI of our models, however, it became clear that the central models offered an improvement on the federated ones, albeit not a statistically significant one. The findings from Vaid et al. (<xref ref-type="bibr" rid="B26">26</xref>), namely, that local models tended to underperform compared to central and federated models (in AUC but also in calibration), align with our findings.</p>
<p>Sadilek et al. (<xref ref-type="bibr" rid="B2">2</xref>) looked at eight previous studies of prediction models that used a centralized model approach and attempted to reproduce these eight models with the modification of using FL in their development strategies. From the eight models they evaluated, only one looked at hospital as the unit of the federation and reported a coefficient estimate for extrapulmonary tuberculosis in individuals with HIV. This coefficient differed significantly between the centralized and federated approaches. However, in a different setting, we observed similar findings with respect to the coefficients of our federated and centralized TAVI risk models.</p>
</sec>
<sec id="s4d"><label>4.4</label><title>Implications and future studies</title>
<p>For clinicians wanting to adopt a federated learning approach for developing prediction models for TAVI patients, our recommendation would be to use the <italic>ensemble</italic> strategy if predictive performance is most important, while the <italic>FedAvg</italic> strategy should be considered if one is willing to sacrifice a bit of model performance for better interpretability.</p>
<p>From the federated learning aspects overview (<xref ref-type="table" rid="T1">Table&#x00A0;1</xref>), possible model strategy setup options were described. While we attempted to make a comprehensive experimental setup, the purpose of this study was not to evaluate all possible options from this table. This methods&#x2019; overview could thus be further used to guide an evaluation of how predictive performance would change if one explored the various setup options.</p>
<p>Further studies should be done to refine the <italic>FedAvg</italic> and <italic>ensemble</italic> models, focusing on the use of additional techniques to enhance privacy-preservation and hyperparameter tuning (<xref ref-type="bibr" rid="B22">22</xref>). The evaluation of model performance should also be considered for other outcomes in addition to the 30-day post-operative mortality, as well as for other FL models in addition to the two types considered here. Future research could also investigate further aspects of model predictive performance by incorporating additional metrics, such as model&#x0027;s sharpness, the area under the precision-recall curve (AUC-PR), and the F1 score. In addition, the questions of investigating model performance in terms of scalability and computing resource requirements are important and merit future research.</p>
<p>The limitation of fitting a <italic>local</italic> model in centers with an insufficient number of case records emphasizes an issue that has not been extensively covered. This area represents a potential direction for future research to improve predictive modeling in such contexts.</p>
<p>Our experiments focused on parametric models, or more precisely models that use logistic regression. It is unclear whether the current findings would translate into federated learning for non-parametric or deep learning models.</p>
<p>Performance variations observed across different models emphasize the importance of selecting the appropriate model development strategy for each individual setting. Finally, examining the potential benefits and limitations of federated learning in cardiology, in general, merits future research.</p>
</sec>
</sec>
<sec id="s5" sec-type="conclusions"><label>5</label><title>Conclusion</title>
<p>Both the <italic>FedAvg</italic> and <italic>ensemble</italic> federated learning models had comparable AUC and calibration performance to the <italic>central</italic> risk prediction model of TAVI patients. This suggests the <italic>FedAvg</italic> and <italic>ensemble</italic> models are strong alternatives to the <italic>central</italic> model, emphasizing their potential effectiveness in the multicenter dataset.</p>
<p>The heterogeneity in performance across different hospitals underscores the importance of local context and sample size. Future research should further explore and enhance these distributed learning methods, particularly focusing on the robustness of federated learning models across diverse clinical settings.</p>
</sec>
</body>
<back>
<sec id="s6" sec-type="data-availability"><title>Data availability statement</title>
<p>The data analyzed in this study is subject to the following licenses/restrictions: permission to use the data underlying this article was granted by the Netherlands Heart Registration (NHR). Permission to reuse the data can be addressed to the NHR (<ext-link ext-link-type="uri" xlink:href="https://nhr.nl/wetenschappelijk-onderzoek/#aanvragen">https://nhr.nl/wetenschappelijk-onderzoek/&#x0023;aanvragen</ext-link>). Requests to access these datasets should be directed to <ext-link ext-link-type="uri" xlink:href="https://nhr.nl/wetenschappelijk-onderzoek/#aanvragen">https://nhr.nl/wetenschappelijk-onderzoek/&#x0023;aanvragen</ext-link>.</p>
</sec>
<sec id="s7" sec-type="ethics-statement"><title>Ethics statement</title>
<p>The studies involving humans were approved by the Committee of Research and Ethics of the Netherlands Heart Registry. The studies were conducted in accordance with the local legislation and institutional requirements. Written informed consent for participation was not required from the participants or the participants&#x2019; legal guardians/next of kin in accordance with the national legislation and institutional requirements.</p>
</sec>
<sec id="s8" sec-type="author-contributions"><title>Author contributions</title>
<p>TY: Conceptualization, Methodology, Software, Validation, Visualization, Writing &#x2013; original draft. AR: Conceptualization, Methodology, Supervision, Validation, Writing &#x2013; review &#x0026; editing. SA: Validation, Writing &#x2013; review &#x0026; editing. MV: Validation, Writing &#x2013; review &#x0026; editing. SH: Data curation, Validation, Writing &#x2013; review &#x0026; editing. SV: Validation, Writing &#x2013; review &#x0026; editing. AA-H: Conceptualization, Methodology, Supervision, Validation, Writing &#x2013; review &#x0026; editing.</p>
</sec>
<sec id="s9" sec-type="funding-information"><title>Funding</title>
<p>The authors declare that no financial support was received for the research, authorship, and/or publication of this article.</p>
</sec>
<ack><title>Acknowledgments</title>
<p>The authors acknowledge the Netherlands Heart Registration (<ext-link ext-link-type="uri" xlink:href="https://www.nhr.nl">https://www.nhr.nl</ext-link>) for making available the data that underpins this work.</p>
</ack>
<sec id="s10" sec-type="COI-statement"><title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s12" sec-type="disclaimer"><title>Publisher&#x0027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s11" sec-type="supplementary-material"><title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fcvm.2024.1399138/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fcvm.2024.1399138/full&#x2002;&#x0023;supplementary-material</ext-link>.</p>
<supplementary-material id="SD1" content-type="local-data">
<media mimetype="application" mime-subtype="pdf" xlink:href="Datasheet1.pdf"/>
</supplementary-material>
</sec>
<ref-list><title>References</title>
<ref id="B1"><label>1.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Al-Farra</surname><given-names>H</given-names></name><name><surname>Ravelli</surname><given-names>ACJ</given-names></name><name><surname>Henriques</surname><given-names>JPS</given-names></name><name><surname>Ter Burg</surname><given-names>WJ</given-names></name><name><surname>Houterman</surname><given-names>S</given-names></name><name><surname>De Mol</surname><given-names>BAJM</given-names></name><etal/></person-group> <article-title>Development and validation of a prediction model for early-mortality after transcatheter aortic valve implantation (TAVI) based on The Netherlands Heart Registration (NHR): the TAVI-NHR risk model</article-title>. <source>Catheter Cardiovasc Interv</source>. (<year>2022</year>) <volume>101</volume>(<issue>2</issue>):<fpage>879</fpage>&#x2013;<lpage>89</lpage>. <pub-id pub-id-type="doi">10.1002/ccd.30398</pub-id></citation></ref>
<ref id="B2"><label>2.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sadilek</surname><given-names>A</given-names></name><name><surname>Liu</surname><given-names>L</given-names></name><name><surname>Nguyen</surname><given-names>D</given-names></name><name><surname>Kamruzzaman</surname><given-names>M</given-names></name><name><surname>Serghiou</surname><given-names>S</given-names></name><name><surname>Rader</surname><given-names>B</given-names></name><etal/></person-group> <article-title>Privacy-first health research with federated learning</article-title>. <source>NPJ Digit Med</source>. (<year>2021</year>) <volume>4</volume>(<issue>1</issue>):<fpage>132</fpage>&#x2013;<lpage>41</lpage>. <pub-id pub-id-type="doi">10.1038/s41746-021-00489-2</pub-id><pub-id pub-id-type="pmid">34493770</pub-id></citation></ref>
<ref id="B3"><label>3.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lee</surname><given-names>EW</given-names></name><name><surname>Xiong</surname><given-names>L</given-names></name><name><surname>Hertzberg</surname><given-names>VS</given-names></name><name><surname>Simpson</surname><given-names>RL</given-names></name><name><surname>Ho</surname><given-names>JC</given-names></name></person-group>. <article-title>Privacy-preserving sequential pattern mining in distributed EHRs for predicting cardiovascular disease</article-title>. <source>AMIA Jt Summits Transl Sci Proc</source>. (<year>2021</year>) <volume>2021</volume>:<fpage>384</fpage>&#x2013;<lpage>93</lpage>. <pub-id pub-id-type="pmid">34457153</pub-id>; <pub-id pub-id-type="pmid">8378625</pub-id>.<pub-id pub-id-type="pmid">34457153</pub-id></citation></ref>
<ref id="B4"><label>4.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>van Egmond</surname><given-names>MB</given-names></name><name><surname>Spini</surname><given-names>G</given-names></name><name><surname>van der Galien</surname><given-names>O</given-names></name><name><surname>Ijpma</surname><given-names>A</given-names></name><name><surname>Veugen</surname><given-names>T</given-names></name><name><surname>Kraaij</surname><given-names>W</given-names></name><etal/></person-group> <article-title>Privacy-preserving dataset combination and lasso regression for healthcare predictions</article-title>. <source>BMC Med Inform Decis Mak</source>. (<year>2021</year>) <volume>21</volume>(<issue>1</issue>):<fpage>266</fpage>&#x2013;<lpage>82</lpage>. <pub-id pub-id-type="doi">10.1186/s12911-021-01582-y</pub-id><pub-id pub-id-type="pmid">34530824</pub-id></citation></ref>
<ref id="B5"><label>5.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Puiu</surname><given-names>A</given-names></name><name><surname>Vizitiu</surname><given-names>A</given-names></name><name><surname>Nita</surname><given-names>C</given-names></name><name><surname>Itu</surname><given-names>L</given-names></name><name><surname>Sharma</surname><given-names>P</given-names></name><name><surname>Comaniciu</surname><given-names>D</given-names></name></person-group>. <article-title>Privacy-preserving and explainable AI for cardiovascular imaging</article-title>. <source>Stud Inform Control</source>. (<year>2021</year>) <volume>30</volume>(<issue>2</issue>):<fpage>21</fpage>&#x2013;<lpage>32</lpage>. <pub-id pub-id-type="doi">10.24846/v30i2y202102</pub-id></citation></ref>
<ref id="B6"><label>6.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zusman</surname><given-names>O</given-names></name><name><surname>Kornowski</surname><given-names>R</given-names></name><name><surname>Witberg</surname><given-names>G</given-names></name><name><surname>Lador</surname><given-names>A</given-names></name><name><surname>Orvin</surname><given-names>K</given-names></name><name><surname>Levi</surname><given-names>A</given-names></name><etal/></person-group> <article-title>Transcatheter aortic valve implantation futility risk model development and validation among treated patients with aortic stenosis</article-title>. <source>Am J Cardiol</source>. (<year>2017</year>) <volume>120</volume>(<issue>12</issue>):<fpage>2241</fpage>&#x2013;<lpage>6</lpage>. <pub-id pub-id-type="doi">10.1016/j.amjcard.2017.09.007</pub-id><pub-id pub-id-type="pmid">29037446</pub-id></citation></ref>
<ref id="B7"><label>7.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Edwards</surname><given-names>FH</given-names></name><name><surname>Cohen</surname><given-names>DJ</given-names></name><name><surname>O&#x2019;Brien</surname><given-names>SM</given-names></name><name><surname>Peterson</surname><given-names>ED</given-names></name><name><surname>Mack</surname><given-names>MJ</given-names></name><name><surname>Shahian</surname><given-names>DM</given-names></name><etal/></person-group> <article-title>Development and validation of a risk prediction model for in-hospital mortality after transcatheter aortic valve replacement</article-title>. <source>JAMA Cardiol</source>. (<year>2016</year>) <volume>1</volume>(<issue>1</issue>):<fpage>46</fpage>&#x2013;<lpage>52</lpage>. <pub-id pub-id-type="doi">10.1001/jamacardio.2015.0326</pub-id><pub-id pub-id-type="pmid">27437653</pub-id></citation></ref>
<ref id="B8"><label>8.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Iung</surname><given-names>B</given-names></name><name><surname>Laou&#x00E9;nan</surname><given-names>C</given-names></name><name><surname>Himbert</surname><given-names>D</given-names></name><name><surname>Eltchaninoff</surname><given-names>H</given-names></name><name><surname>Chevreul</surname><given-names>K</given-names></name><name><surname>Donzeau-Gouge</surname><given-names>P</given-names></name><etal/></person-group> <article-title>Predictive factors of early mortality after transcatheter aortic valve implantation: individual risk assessment using a simple score</article-title>. <source>Heart</source>. (<year>2014</year>) <volume>100</volume>(<issue>13</issue>):<fpage>1016</fpage>&#x2013;<lpage>23</lpage>. <pub-id pub-id-type="doi">10.1136/heartjnl-2013-305314</pub-id><pub-id pub-id-type="pmid">24740804</pub-id></citation></ref>
<ref id="B9"><label>9.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yordanov</surname><given-names>TR</given-names></name><name><surname>Lopes</surname><given-names>RR</given-names></name><name><surname>Ravelli</surname><given-names>ACJ</given-names></name><name><surname>Vis</surname><given-names>M</given-names></name><name><surname>Houterman</surname><given-names>S</given-names></name><name><surname>Marquering</surname><given-names>H</given-names></name><etal/></person-group> <article-title>An integrated approach to geographic validation helped scrutinize prediction model performance and its variability</article-title>. <source>J Clin Epidemiol</source>. (<year>2023</year>) <volume>157</volume>:<fpage>13</fpage>&#x2013;<lpage>21</lpage>. <pub-id pub-id-type="doi">10.1016/j.jclinepi.2023.02.021</pub-id><pub-id pub-id-type="pmid">36822443</pub-id></citation></ref>
<ref id="B10"><label>10.</label><citation citation-type="book"><person-group person-group-type="author"><name><surname>McMahan</surname><given-names>B</given-names></name><name><surname>Moore</surname><given-names>E</given-names></name><name><surname>Ramage</surname><given-names>D</given-names></name><name><surname>Hampson</surname><given-names>S</given-names></name><name><surname>Arcas</surname><given-names>B</given-names></name></person-group>. &#x201C;<article-title>Communication-efficient learning of deep networks from decentralized data</article-title>&#x201D;. In: <person-group person-group-type="editor"><name><surname>Aarti</surname><given-names>S</given-names></name><name><surname>Jerry</surname><given-names>Z</given-names></name></person-group>, editors. <source>Proceedings of the 20th International Conference on Artificial Intelligence and Statistics; Proceedings of Machine Learning Research: PMLR</source>. London, United Kingdom: BioMed Central (<year>2017</year>). p. <fpage>1273</fpage>&#x2013;<lpage>82</lpage>.</citation></ref>
<ref id="B11"><label>11.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Reps</surname><given-names>JM</given-names></name><name><surname>Williams</surname><given-names>RD</given-names></name><name><surname>Schuemie</surname><given-names>MJ</given-names></name><name><surname>Ryan</surname><given-names>PB</given-names></name><name><surname>Rijnbeek</surname><given-names>PR</given-names></name></person-group>. <article-title>Learning patient-level prediction models across multiple healthcare databases: evaluation of ensembles for increasing model transportability</article-title>. <source>BMC Med Inform Decis Mak</source>. (<year>2022</year>) <volume>22</volume>(<issue>1</issue>):<fpage>142</fpage>&#x2013;<lpage>57</lpage>. <pub-id pub-id-type="doi">10.1186/s12911-022-01879-6</pub-id><pub-id pub-id-type="pmid">35614485</pub-id></citation></ref>
<ref id="B12"><label>12.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Collins</surname><given-names>GS</given-names></name><name><surname>Reitsma</surname><given-names>JB</given-names></name><name><surname>Altman</surname><given-names>DG</given-names></name><name><surname>Moons</surname><given-names>KG</given-names></name></person-group>. <article-title>Transparent reporting of a multivariable prediction model for individual prognosis or diagnosis (TRIPOD): the TRIPOD statement. The TRIPOD group</article-title>. <source>Circulation</source>. (<year>2015</year>) <volume>131</volume>(<issue>2</issue>):<fpage>211</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1161/CIRCULATIONAHA.114.014508</pub-id><pub-id pub-id-type="pmid">25561516</pub-id></citation></ref>
<ref id="B13"><label>13.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kotecha</surname><given-names>D</given-names></name><name><surname>Asselbergs</surname><given-names>FW</given-names></name><name><surname>Achenbach</surname><given-names>S</given-names></name><name><surname>Anker</surname><given-names>SD</given-names></name><name><surname>Atar</surname><given-names>D</given-names></name><name><surname>Baigent</surname><given-names>C</given-names></name><etal/></person-group> <article-title>CODE-EHR best practice framework for the use of structured electronic healthcare records in clinical research</article-title>. <source>Eur Heart J</source>. (<year>2022</year>) <volume>43</volume>(<issue>37</issue>):<fpage>3578</fpage>&#x2013;<lpage>88</lpage>. <pub-id pub-id-type="doi">10.1093/eurheartj/ehac426</pub-id><pub-id pub-id-type="pmid">36208161</pub-id></citation></ref>
<ref id="B14"><label>14.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Timmermans</surname><given-names>MJC</given-names></name><name><surname>Houterman</surname><given-names>S</given-names></name><name><surname>Daeter</surname><given-names>ED</given-names></name><name><surname>Danse</surname><given-names>PW</given-names></name><name><surname>Li</surname><given-names>WW</given-names></name><name><surname>Lipsic</surname><given-names>E</given-names></name><etal/></person-group> <article-title>Using real-world data to monitor and improve quality of care in coronary artery disease: results from The Netherlands Heart Registration</article-title>. <source>Neth Heart J</source>. (<year>2022</year>) 30(12):<fpage>546</fpage>&#x2013;<lpage>56</lpage>. <pub-id pub-id-type="doi">10.1007/s12471-022-01672-0</pub-id><pub-id pub-id-type="pmid">35389133</pub-id></citation></ref>
<ref id="B15"><label>15.</label><citation citation-type="book"><person-group person-group-type="editor"><name><surname>Fumera</surname><given-names>G</given-names></name><name><surname>Roli</surname><given-names>F</given-names></name></person-group>, editors. <source>Performance Analysis and Comparison of Linear Combiners for Classifier Fusion. Structural, Syntactic, and Statistical Pattern Recognition</source>. <publisher-loc>Berlin</publisher-loc>: <publisher-name>Springer</publisher-name> (<year>2002</year>).</citation></ref>
<ref id="B16"><label>16.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cox</surname><given-names>DR</given-names></name></person-group>. <article-title>Two further applications of a model for binary regression</article-title>. <source>Biometrika</source>. (<year>1958</year>) <volume>45</volume>(<issue>3&#x2013;4</issue>):<fpage>562</fpage>&#x2013;<lpage>5</lpage>. <pub-id pub-id-type="doi">10.1093/biomet/45.3-4.562</pub-id></citation></ref>
<ref id="B17"><label>17.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Efron</surname><given-names>B</given-names></name></person-group>. <article-title>Bootstrap methods: another look at the jackknife</article-title>. <source>Ann Stat</source>. (<year>1979</year>) <volume>7</volume>(<issue>1</issue>):<fpage>1</fpage>&#x2013;<lpage>26</lpage>. <pub-id pub-id-type="doi">10.1214/aos/1176344552</pub-id></citation></ref>
<ref id="B18"><label>18.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>van Buuren</surname><given-names>S</given-names></name><name><surname>Groothuis-Oudshoorn</surname><given-names>K</given-names></name></person-group>. <article-title>Mice: multivariate imputation by chained equations in R</article-title>. <source>J Stat Softw</source>. (<year>2011</year>) <volume>45</volume>(<issue>3</issue>):<fpage>1</fpage>&#x2013;<lpage>67</lpage>.</citation></ref>
<ref id="B19"><label>19.</label><citation citation-type="book"><person-group person-group-type="author"><name><surname>Alaa</surname><given-names>AM</given-names></name><name><surname>van der Schaar</surname><given-names>M</given-names></name></person-group>. &#x201C;<article-title>Demystifying black-box models with symbolic metamodels</article-title>&#x201D;. In: <person-group person-group-type="editor"><name><surname>Wallach</surname><given-names>H</given-names></name><name><surname>Larochelle</surname><given-names>H</given-names></name><name><surname>Beygelzimer</surname><given-names>A</given-names></name><name><surname>d&#x2019; Alch&#x00E9;-Buc</surname><given-names>F</given-names></name><name><surname>Fox</surname><given-names>E</given-names></name><name><surname>Garnett</surname><given-names>R</given-names></name></person-group>, editors. <source>Advances in Neural Information Processing Systems 2019</source>. <publisher-name>Basel, Switzerland: Curran Associates, Inc.</publisher-name> (<year>2019</year>).</citation></ref>
<ref id="B20"><label>20.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Moshawrab</surname><given-names>M</given-names></name><name><surname>Adda</surname><given-names>M</given-names></name><name><surname>Bouzouane</surname><given-names>A</given-names></name><name><surname>Ibrahim</surname><given-names>H</given-names></name><name><surname>Raad</surname><given-names>A</given-names></name></person-group>. <article-title>Reviewing federated learning aggregation algorithms; strategies, contributions, limitations and future perspectives</article-title>. <source>Electronics (Basel)</source>. (<year>2023</year>) <volume>12</volume>(<issue>10</issue>):<fpage>2287</fpage>&#x2013;<lpage>322</lpage>. <pub-id pub-id-type="doi">10.3390/electronics12102287</pub-id></citation></ref>
<ref id="B21"><label>21.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lopes</surname><given-names>RR</given-names></name><name><surname>Yordanov</surname><given-names>TTR</given-names></name><name><surname>Ravelli</surname><given-names>A</given-names></name><name><surname>Houterman</surname><given-names>S</given-names></name><name><surname>Vis</surname><given-names>M</given-names></name><name><surname>de Mol</surname><given-names>B</given-names></name><etal/></person-group> <article-title>Temporal validation of 30-day mortality prediction models for transcatheter aortic valve implantation using statistical process control&#x2014;an observational study in a national population</article-title>. <source>Heliyon</source>. (<year>2023</year>) <volume>9</volume>(<issue>6</issue>):<fpage>e17139</fpage>. <pub-id pub-id-type="doi">10.1016/j.heliyon.2023.e17139</pub-id><pub-id pub-id-type="pmid">37484279</pub-id></citation></ref>
<ref id="B22"><label>22.</label><citation citation-type="other"><person-group person-group-type="author"><name><surname>Wang</surname><given-names>J</given-names></name><name><surname>Charles</surname><given-names>ZB</given-names></name><name><surname>Xu</surname><given-names>Z</given-names></name><name><surname>Joshi</surname><given-names>G</given-names></name><name><surname>McMahan</surname><given-names>HB</given-names></name><name><surname>Arcas</surname><given-names>B</given-names></name><etal/></person-group> <comment><italic>A Field Guide to Federated Optimization</italic>. arXiv. abs/2107.06917</comment> (<year>2021</year>).</citation></ref>
<ref id="B23"><label>23.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lopes</surname><given-names>RR</given-names></name><name><surname>Mamprin</surname><given-names>M</given-names></name><name><surname>Zelis</surname><given-names>JM</given-names></name><name><surname>Tonino</surname><given-names>PAL</given-names></name><name><surname>van Mourik</surname><given-names>MS</given-names></name><name><surname>Vis</surname><given-names>MM</given-names></name><etal/></person-group> <article-title>Local and distributed machine learning for inter-hospital data utilization: an application for TAVI outcome prediction</article-title>. <source>Front Cardiovasc Med</source>. (<year>2021</year>) <volume>8</volume>:787246. <pub-id pub-id-type="doi">10.3389/fcvm.2021.787246</pub-id></citation></ref>
<ref id="B24"><label>24.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Goto</surname><given-names>S</given-names></name><name><surname>Solanki</surname><given-names>D</given-names></name><name><surname>John</surname><given-names>JE</given-names></name><name><surname>Yagi</surname><given-names>R</given-names></name><name><surname>Homilius</surname><given-names>M</given-names></name><name><surname>Ichihara</surname><given-names>G</given-names></name><etal/></person-group> <article-title>Multinational federated learning approach to train ECG and echocardiogram models for hypertrophic cardiomyopathy detection</article-title>. <source>Circulation</source>. (<year>2022</year>) <volume>146</volume>(<issue>10</issue>):<fpage>755</fpage>&#x2013;<lpage>69</lpage>. <pub-id pub-id-type="doi">10.1161/CIRCULATIONAHA.121.058696</pub-id><pub-id pub-id-type="pmid">35916132</pub-id></citation></ref>
<ref id="B25"><label>25.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sangha</surname><given-names>V</given-names></name><name><surname>Nargesi</surname><given-names>AA</given-names></name><name><surname>Dhingra</surname><given-names>LS</given-names></name><name><surname>Khunte</surname><given-names>A</given-names></name><name><surname>Mortazavi</surname><given-names>BJ</given-names></name><name><surname>Ribeiro</surname><given-names>AH</given-names></name><etal/></person-group> <article-title>Detection of left ventricular systolic dysfunction from electrocardiographic images</article-title>. <source>Circulation</source>. (<year>2023</year>) <volume>148</volume>(<issue>9</issue>):<fpage>765</fpage>&#x2013;<lpage>77</lpage>. <pub-id pub-id-type="doi">10.1161/CIRCULATIONAHA.122.062646</pub-id><pub-id pub-id-type="pmid">37489538</pub-id></citation></ref>
<ref id="B26"><label>26.</label><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vaid</surname><given-names>A</given-names></name><name><surname>Jaladanki</surname><given-names>SK</given-names></name><name><surname>Xu</surname><given-names>J</given-names></name><name><surname>Teng</surname><given-names>S</given-names></name><name><surname>Kumar</surname><given-names>A</given-names></name><name><surname>Lee</surname><given-names>S</given-names></name><etal/></person-group> <article-title>Federated learning of electronic health records to improve mortality prediction in hospitalized patients with COVID-19: machine learning approach</article-title>. <source>JMIR Med Inform</source>. (<year>2021</year>) <volume>9</volume>(<issue>1</issue>):<fpage>e24207</fpage>. <pub-id pub-id-type="doi">10.2196/24207</pub-id><pub-id pub-id-type="pmid">33400679</pub-id></citation></ref></ref-list>
</back>
</article>