<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Archiving and Interchange DTD v2.3 20070202//EN" "archivearticle.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="methods-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Radiol.</journal-id>
<journal-title>Frontiers in Radiology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Radiol.</abbrev-journal-title>
<issn pub-type="epub">2673-8740</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fradi.2021.777030</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Radiology</subject>
<subj-group>
<subject>Methods</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Integrating Transcriptomics, Genomics, and Imaging in Alzheimer&#x00027;s Disease: A Federated Model</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Wu</surname> <given-names>Jianfeng</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1264815/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Chen</surname> <given-names>Yanxi</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Wang</surname> <given-names>Panwen</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Caselli</surname> <given-names>Richard J.</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Thompson</surname> <given-names>Paul M.</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Wang</surname> <given-names>Junwen</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/605336/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Wang</surname> <given-names>Yalin</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c002"><sup>&#x0002A;</sup></xref>
<xref ref-type="author-notes" rid="fn002"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/226907/overview"/>
</contrib>
<on-behalf-of>the Alzheimer&#x00027;s Disease Neuroimaging Initiative</on-behalf-of>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>School of Computing and Augmented Intelligence, Arizona State University</institution>, <addr-line>Tempe, AZ</addr-line>, <country>United States</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of Health Sciences Research and Center for Individualized Medicine, Mayo Clinic Arizona</institution>, <addr-line>Scottsdale, AZ</addr-line>, <country>United States</country></aff>
<aff id="aff3"><sup>3</sup><institution>Department of Neurology, Mayo Clinic Arizona</institution>, <addr-line>Scottsdale, AZ</addr-line>, <country>United States</country></aff>
<aff id="aff4"><sup>4</sup><institution>Imaging Genetics Center, Stevens Institute for Neuroimaging and Informatics, Keck School of Medicine, University of Southern California</institution>, <addr-line>Los Angeles, CA</addr-line>, <country>United States</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Li Shen, University of Pennsylvania, United States</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Jingwen Yan, Purdue University Indianapolis, United States; Eun Jeong Min, Catholic University of Korea, South Korea</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Junwen Wang <email>wang.junwen&#x00040;mayo.edu</email></corresp>
<corresp id="c002">Yalin Wang <email>ylwang&#x00040;asu.edu</email></corresp>
<fn fn-type="other" id="fn001"><p>This article was submitted to Artificial Intelligence in Radiology, a section of the journal Frontiers in Radiology</p></fn>
<fn fn-type="equal" id="fn002"><p>&#x02020;These authors have contributed equally to this work</p></fn></author-notes>
<pub-date pub-type="epub">
<day>21</day>
<month>01</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2021</year>
</pub-date>
<volume>1</volume>
<elocation-id>777030</elocation-id>
<history>
<date date-type="received">
<day>14</day>
<month>09</month>
<year>2021</year>
</date>
<date date-type="accepted">
<day>21</day>
<month>12</month>
<year>2021</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2022 Wu, Chen, Wang, Caselli, Thompson, Wang and Wang.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Wu, Chen, Wang, Caselli, Thompson, Wang and Wang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Alzheimer&#x00027;s disease (AD) affects more than 1 in 9 people age 65 and older and becomes an urgent public health concern as the global population ages. In clinical practice, structural magnetic resonance imaging (sMRI) is the most accessible and widely used diagnostic imaging modality. Additionally, genome-wide association studies (GWAS) and transcriptomics&#x02014;the study of gene expression&#x02014;also play an important role in understanding AD etiology and progression. Sophisticated imaging genetics systems have been developed to discover genetic factors that consistently affect brain function and structure. However, most studies to date focused on the relationships between brain sMRI and GWAS or brain sMRI and transcriptomics. To our knowledge, few methods have been developed to discover and infer multimodal relationships among sMRI, GWAS, and transcriptomics. To address this, we propose a novel federated model, Genotype-Expression-Imaging Data Integration (GEIDI), to identify genetic and transcriptomic influences on brain sMRI measures. The relationships between brain imaging measures and gene expression are allowed to depend on a person&#x00027;s genotype at the single-nucleotide polymorphism (SNP) level, making the inferences adaptive and personalized. We performed extensive experiments on publicly available Alzheimer&#x00027;s Disease Neuroimaging Initiative (ADNI) dataset. Experimental results demonstrated our proposed method outperformed state-of-the-art expression quantitative trait loci (eQTL) methods for detecting genetic and transcriptomic factors related to AD and has stable performance when data are integrated from multiple sites. Our GEIDI approach may offer novel insights into the relationship among image biomarkers, genotypes, and gene expression and help discover novel genetic targets for potential AD drug treatments.</p></abstract>
<kwd-group>
<kwd>Alzheimer&#x00027;s disease</kwd>
<kwd>brain imaging</kwd>
<kwd>GWAS</kwd>
<kwd>transcriptomics</kwd>
<kwd>chow test</kwd>
<kwd>federated learning</kwd>
</kwd-group>
<contract-num rid="cn001">P30AG072980</contract-num>
<contract-num rid="cn001">R01AG069453</contract-num>
<contract-num rid="cn001">R21AG065942</contract-num>
<contract-num rid="cn001">RF1AG051710</contract-num>
<contract-num rid="cn001">U01AG068057</contract-num>
<contract-num rid="cn002">R01LM013438</contract-num>
<contract-num rid="cn003">R01EB025032</contract-num>
<contract-num rid="cn004">R01EY032125</contract-num>
<contract-sponsor id="cn001">National Institute on Aging<named-content content-type="fundref-id">10.13039/100000049</named-content></contract-sponsor>
<contract-sponsor id="cn002">U.S. National Library of Medicine<named-content content-type="fundref-id">10.13039/100000092</named-content></contract-sponsor>
<contract-sponsor id="cn003">National Institute of Biomedical Imaging and Bioengineering<named-content content-type="fundref-id">10.13039/100000070</named-content></contract-sponsor>
<contract-sponsor id="cn004">National Eye Institute<named-content content-type="fundref-id">10.13039/100000053</named-content></contract-sponsor>
<contract-sponsor id="cn005">Arizona Alzheimer&#x00027;s Consortium<named-content content-type="fundref-id">10.13039/100014813</named-content></contract-sponsor>
<counts>
<fig-count count="5"/>
<table-count count="5"/>
<equation-count count="2"/>
<ref-count count="75"/>
<page-count count="14"/>
<word-count count="10305"/>
</counts>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>Introduction</title>
<p>Alzheimer&#x00027;s disease (AD) is a major public health concern, with the number of affected individuals expected to triple, reaching 13.8 million, by the year 2050 in the U.S. alone (<xref ref-type="bibr" rid="B1">1</xref>). Current therapeutic failures in patients with dementia due to AD may be due to interventions that are too late or targets that are secondary effects and less relevant to disease initiation and early progression (<xref ref-type="bibr" rid="B2">2</xref>). Mounting evidence suggests that germline mutations, e.g., DNA single nucleotide polymorphisms (SNPs), play an important role in AD etiology and progression (<xref ref-type="bibr" rid="B3">3</xref>, <xref ref-type="bibr" rid="B4">4</xref>). Among various genetic risk factors, Apolipoprotein E (<italic>APOE</italic>) has the strongest association to late-onset AD, and the e4 allele is associated with increased risk, whereas the e2 allele is associated with decreased risk (<xref ref-type="bibr" rid="B5">5</xref>). Known genetic risk variants could be used to identify presymptomatic individuals at risk for AD and support diagnostic assessment of symptomatic subjects. By taking into account patients&#x00027; genetic risk factors, at-risk individuals could be more readily identified, diagnostic precision could be improved, and targetable disease mechanisms for new drug development may be discovered (<xref ref-type="bibr" rid="B6">6</xref>&#x02013;<xref ref-type="bibr" rid="B9">9</xref>). By enabling each patient to receive earlier diagnoses, risk assessments, and optimal treatments, personalized or precision medicine holds promise for improving early AD intervention while also lowering costs (<xref ref-type="bibr" rid="B10">10</xref>).</p>
<p>Recent clinical trials targeting single molecular mechanisms have failed (<xref ref-type="bibr" rid="B11">11</xref>, <xref ref-type="bibr" rid="B12">12</xref>). Rather, it might be necessary to tackle the problem from a holistic or multimodality perspective (<xref ref-type="bibr" rid="B13">13</xref>, <xref ref-type="bibr" rid="B14">14</xref>). Indeed, the NIH and the scientific community realized this problem a while ago and have already started to produce multi-omics data. For example, the Alzheimer&#x00027;s Disease Sequencing Project (ADSP) data repository contains genomic level data derived from genome-wide association studies (GWAS) (<xref ref-type="bibr" rid="B4">4</xref>, <xref ref-type="bibr" rid="B15">15</xref>), whole-exome sequencing (WES) (<xref ref-type="bibr" rid="B16">16</xref>, <xref ref-type="bibr" rid="B17">17</xref>), and whole-genome sequencing (WGS), and RNA level data including mRNA, miRNA, and long non-coding RNA profiling from either microarray or RNA-Seq (<xref ref-type="bibr" rid="B18">18</xref>). And transcriptome-wide association studies (TWASs) provides a way to use eQTLs and expression data to guide GWAS of AD (<xref ref-type="bibr" rid="B19">19</xref>). Brain imaging has played a significant role in the study of Alzheimer&#x00027;s disease (<xref ref-type="bibr" rid="B20">20</xref>). Integrating imaging data and omics data is becoming an emerging data science field known as imaging genomics (<xref ref-type="bibr" rid="B21">21</xref>). The major task of this field is to perform integrated analysis of imaging and omics data, often combined with other biomarkers, as well as clinical and environmental data. The ultimate goal is to gain new insights into the underlying mechanisms of human health and disease, to better inform the development of new diagnostic, therapeutic, and preventative approaches.</p>
<p>Various imaging genetics methods have been developed to integrate imaging and genetic data. However, most studies have focused on imaging, imaging combined with GWAS data (<xref ref-type="bibr" rid="B22">22</xref>&#x02013;<xref ref-type="bibr" rid="B24">24</xref>), imaging with transcriptomics (<xref ref-type="bibr" rid="B25">25</xref>), or GWAS with transcriptomics (<xref ref-type="bibr" rid="B26">26</xref>). For example, imaging genetics methods have been used to link SNPs with image features (<xref ref-type="bibr" rid="B27">27</xref>), and expression quantitative trait loci (eQTL) have been used to discover <italic>APOE</italic>-related genes (<xref ref-type="bibr" rid="B28">28</xref>). However, relatively few methods have been developed to integrate GWAS/WES/WGS, imaging, and transcriptomic data to infer multimodal relationships. For instance, Liu et al.(<xref ref-type="bibr" rid="B29">29</xref>) use a brain-wide gene expression profile available in the Allen Human Brain Atlas (AHBA) as a 2-D prior to guide the brain imaging genetics association analysis. Their transcriptome-guided SCCA (TG-SCCA) framework incorporates the gene expression information into the traditional SCCA model. Such a multimodal approach may give us a more holistic view of the evidence from multiple sources to provide novel insights on the molecular mechanisms of AD pathogenesis and prognosis. Besides, both gene expression and imaging features are dynamic and change with time and throughout the disease, whereas germline SNPs are unchanged over an individual&#x00027;s lifetime. We need a better model for studying SNP-image-gene expression relationships to consider both the dynamic changes in imaging and gene expression features and understand how they are affected by an individual&#x00027;s SNPs. Such knowledge will provide novel insights into the relationship among image biomarkers, genotypes and gene expression, and may help discover novel genetic targets for pharmaceutical interventions.</p>
<p>AD is a complex multifactorial disorder that involves many biological processes. The launch of the Alzheimer&#x00027;s Precision Medicine Initiative (APMI) and its associated cohort program in 2016&#x02014;facilitated by the academic core coordinating center run by the Sorbonne University Clinical Research Group in Alzheimer&#x00027;s Precision Medicine&#x02014;is intended to improve clinical diagnostics and drug development research in Alzheimer&#x00027;s disease (<xref ref-type="bibr" rid="B30">30</xref>). Hampel et al. (<xref ref-type="bibr" rid="B30">30</xref>) indicate the challenges for precision medicine, including secure data access accompanied by rigorous privacy protection and the availability of data to qualified researchers who may use them to exercise their creative thinking with an <italic>a posteriori</italic> approach or to test their <italic>a priori</italic> hypotheses. Integrating data from multiple sites and sources is common practice to achieve larger sample sizes and increase the statistical power. Unprecedentedly large amounts of biomedical data now exist across hospitals and research institutions. However, different institutions may not be readily able to share biomedical research data due to patient privacy concerns, data restrictions based on patient consent or institutional review board (IRB) regulations, and legal complexities; this can present a major obstacle to pooling large scale datasets to discover and understand AD-related genetic information. To remedy this distributed problem, a large-scale collaborative network, ENIGMA consortium, was built (<xref ref-type="bibr" rid="B31">31</xref>). Federated learning is an important direction of interest in multi-site neuroimaging research; the use of distributed computing offers an approach to learn from data spread across multiple sites without having to share the raw data directly or to centralize it in any one location. Even so, most ENIGMA and other GWAS studies currently focus on the influence of genetic variants on human brain structures (<xref ref-type="bibr" rid="B22">22</xref>, <xref ref-type="bibr" rid="B32">32</xref>&#x02013;<xref ref-type="bibr" rid="B34">34</xref>) or functional measures (<xref ref-type="bibr" rid="B35">35</xref>) and relatively few have studied the relationships among image biomarkers, genotypes, and gene expression.</p>
<p>In this paper, we propose a novel Federated Genotype-Expression-Image Data Integration model (GEIDI) based on the Chow test (<xref ref-type="bibr" rid="B36">36</xref>). The intuition behind our multi-omics framework is illustrated in <xref ref-type="fig" rid="F1">Figure 1</xref>. Some important image-expression relationships (correlations) may be diluted when the population is mixed together. Still, when we stratify the population based on their genotypes (a gene like <italic>APOE</italic> or a SNP like <italic>rs942439</italic>), we can observe strong correlations (AA and BB groups) across subgroups. Accordingly, as shown in <xref ref-type="fig" rid="F2">Figure 2</xref>, our model is designed to detect if the relationships between X (imaging biomarker) and Y (gene expression) are different among the subgroups. The <italic>p</italic>-value of the model is then used to prioritize the trios (genotype-expression-image).</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Schematic view of our multi-omics approach. <bold>(A)</bold> When patients are mixed together, image-expression correlation may be low. <bold>(B)</bold> When a certain genotype stratifies patients, some subsets (AA, BB) have a high correlation.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fradi-01-777030-g0001.tif"/>
</fig>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>The federated GEIDI model on ADNI data. <bold>(A)</bold> Stratify samples into subgroups with different genotypes of a gene (e.g., <italic>APOE</italic>) or at a specific SNP locus (e.g., <italic>rs942439</italic>) <bold>(B)</bold> Federated GEIDI is used to detect if the relationships between X (imaging biomarker) and Y (gene expression) are different among the subgroups. The <italic>p</italic>-value of federated GEIDI will then be used to prioritize the trios (genotype-expression-image).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fradi-01-777030-g0002.tif"/>
</fig>
<p>We further design various experiments on publicly available data from the Alzheimer&#x00027;s Disease Neuroimaging Initiative (ADNI, <ext-link ext-link-type="uri" xlink:href="http://adni.loni.usc.edu">adni.loni.usc.edu</ext-link>) to demonstrate that our model may detect the genetic factors most related to AD better than the state-of-the-art Matrix eQTL. The overall intent of the work is to detect relationships that inform the design or repurposing of drugs to target these subgroups to achieve precision medicine. We first use a hypergeometric analysis and an AD-related gene list from <ext-link ext-link-type="uri" xlink:href="http://alzgene.org/">alzgene.org</ext-link> to evaluate the ability of our federated GEIDI model to discover AD-related gene expression. To further aid in the discovery of genes that may be potential AD drug targets, we also use Pearson correlations analyses to demonstrate the divergence in stratified populations. Additionally, we design experiments to show that our model can discover more AD-related SNPs, based on tests with 1,217 known AD-associated SNPs and 1,217 randomly selected SNPs. Finally, we evaluate the stability of our model under different multi-site conditions. With the ADNI dataset, we set off to test our hypothesis that the proposed federated GEIDI model may be an effective federated model that can provide novel insights into the relationship among image biomarkers, genotypes, and gene expressions and the discovery of novel genes for potential AD drug targets.</p>
</sec>
<sec id="s2">
<title>Data and Methods</title>
<sec>
<title>Data Preprocessing</title>
<p>The data in this work are from the Alzheimer&#x00027;s Disease Neuroimaging Initiative (ADNI, <ext-link ext-link-type="uri" xlink:href="http://adni.loni.usc.edu">adni.loni.usc.edu</ext-link>) and the TADPOLE challenge (<ext-link ext-link-type="uri" xlink:href="http://tadpole.grand-challenge.org">tadpole.grand-challenge.org</ext-link>) (<xref ref-type="bibr" rid="B37">37</xref>). The ADNI was launched in 2003 as a public-private partnership led by Principal Investigator Michael W. Weiner, MD. The primary goal of ADNI has been to test whether serial MRI, PET, other biological markers, and clinical and neuropsychological assessments can be combined to measure the progression of MCI and early AD. The genome-wide association study of ADNI is designed to provide researchers with the opportunity to combine genetics with imaging and clinical data to help investigate the mechanisms of the disease. For up-to-date information, see <ext-link ext-link-type="uri" xlink:href="http://adni.loni.usc.edu/data-samples/data-types/genetic-data/">adni.loni.usc.edu/data-samples/data-types/genetic-data/</ext-link>. From the ADNI GWAS, we analyzed data from 697 subjects, including AD patients, people with mild cognitive impairment (MCI), and cognitively unimpaired (CU) subjects, for whom the demographic information is shown in <xref ref-type="table" rid="T1">Table 1</xref>. Each sample has three types of modalities of data: genotypes of known AD risk genes (e.g., <italic>APOE</italic>) and SNPs from genome-wide association studies (GWAS), gene expression measurements (for 20,211 genes) from microarray-based transcriptomic profiling of samples&#x00027; blood, and imaging biomarkers from structural magnetic resonance imaging (sMRI) data of subjects&#x00027; brains. We use <italic>plink</italic> to perform a quality check of the genotype data. The SNPs in the normal group that deviate significantly from Hardy-Weinberg equilibrium are removed (<xref ref-type="bibr" rid="B38">38</xref>). The LINNORM package (<xref ref-type="bibr" rid="B39">39</xref>) was adopted to perform data transformation on the expression data for normality and homoscedasticity. Recent evaluations (<xref ref-type="bibr" rid="B40">40</xref>, <xref ref-type="bibr" rid="B41">41</xref>) show that LINNORM typically performs better than current DEG analysis methods for both single-cell and bulk RNA-Seq, such as Seurat (<xref ref-type="bibr" rid="B42">42</xref>) and DESeq2 (<xref ref-type="bibr" rid="B43">43</xref>).</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Demographic information for the subjects we study from the ADNI.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th valign="top" align="left"><bold>Group</bold></th>
<th valign="top" align="center"><bold>Sex (M/F)</bold></th>
<th valign="top" align="center"><bold>Age</bold></th>
<th valign="top" align="center"><bold>MMSE</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">AD (<italic>n</italic> = 96)</td>
<td valign="top" align="center">59/37</td>
<td valign="top" align="center">74.8 &#x000B1; 7.5</td>
<td valign="top" align="center">21.8 &#x000B1; 4.1</td>
</tr>
<tr>
<td valign="top" align="left">MCI (<italic>n</italic> = 366)</td>
<td valign="top" align="center">209/157</td>
<td valign="top" align="center">72.0 &#x000B1; 7.5</td>
<td valign="top" align="center">28.0 &#x000B1; 1.7</td>
</tr>
<tr>
<td valign="top" align="left">CU (<italic>n</italic> = 235)</td>
<td valign="top" align="center">115/120</td>
<td valign="top" align="center">74.4 &#x000B1; 5.8</td>
<td valign="top" align="center">29.1 &#x000B1; 1.2</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p><italic>Values are mean &#x000B1; standard deviation, where applicable</italic>.</p>
</table-wrap-foot>
</table-wrap>
<p>Eventually, we get 2,059,586 SNPs, <italic>APOE</italic> genotype, and expression data for 20,211 genes for each sample. Besides, from the TADPOLE challenge, we obtained two brain imaging biomarkers for each subject calculated using FreeSurfer (<xref ref-type="bibr" rid="B44">44</xref>) with sMRI, including the volume of the hippocampus and middle temporal gyrus (MidTemp). To adjust for individual differences in head size, the volume of each region is adjusted by the intracranial vault volume (ICV) of each subject (volume/ICV). The difference between the dates for gene expression collection and MRI scan is &#x0003C;5 months.</p>
</sec>
<sec>
<title>Federated Genotype-Expression-Image Data Integration Framework</title>
<p>Econometrician Gregory Chow first proposed the Chow test in 1960 (<xref ref-type="bibr" rid="B36">36</xref>) to determine whether correlation coefficients estimated in two subgroups are significantly different. In econometrics, it is most commonly used in time series analysis to test for the presence of a structural break at a period that can be assumed to be known as <italic>a priori</italic> (for instance, a significant historical event such as a war). For example, we can model the data as <italic>y</italic> &#x0003D; <italic>wX</italic>&#x0002B;&#x003F5;. Then, the data can be broken into two groups according to some event and fitted to the regression model as, <italic>y</italic><sub>1</sub> &#x0003D; <italic>w</italic><sub>1</sub><italic>x</italic><sub>1</sub> &#x0002B; &#x003F5; and <italic>y</italic><sub>2</sub> &#x0003D; <italic>w</italic><sub>2</sub><italic>x</italic><sub>2</sub> &#x0002B; &#x003F5;. The null hypothesis of the Chow test asserts that <italic>w</italic><sub>1</sub> &#x0003D; <italic>w</italic><sub>2</sub> and the model errors &#x003F5; are independent and identically distributed from a normal distribution with unknown variance. Let <italic>S</italic><sub><italic>C</italic></sub>, <italic>S</italic><sub>1</sub>, and <italic>S</italic><sub>2</sub> be the sum of squared residuals for the three regression models, respectively, <italic>N</italic><sub>1</sub> and <italic>N</italic><sub>2</sub> are the number of observations in each group, and <italic>k</italic> is the number of parameters. The Chow test statistic is <inline-formula><mml:math id="M1"><mml:mrow><mml:mi>F</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>S</mml:mi><mml:mi>C</mml:mi></mml:msub><mml:mo>&#x02212;</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>S</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>S</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>)</mml:mo><mml:mo>/</mml:mo><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>S</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>S</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>/</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>N</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>N</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo>&#x02212;</mml:mo><mml:mn>2</mml:mn><mml:mi>k</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mfrac></mml:mrow></mml:math></inline-formula>, which follows the <italic>F</italic>-distribution with <italic>k</italic> and <italic>N</italic><sub>1</sub> &#x0002B; <italic>N</italic><sub>2</sub> &#x02212; 2<italic>k</italic> degrees of freedom.</p>
<p>Although the Chow test is commonly used in the financial industry, it is seldom used in the biomedical field (<xref ref-type="bibr" rid="B45">45</xref>). In this work, we first generalize the Chow test model to estimate the multi-subgroup condition and further introduce a federated learning technique to the model. We apply the proposed model to the ADNI dataset to detect the significant trios among genotype, gene expression, and imaging biomarkers and discover the dominant genetic and transcriptomics factors for brain structures.</p>
<sec>
<title>Standardization</title>
<p>We simulate the multi-site condition by separating all the samples into <italic>I</italic> hypothetical institutions (<italic>I</italic> &#x0003D; 5) on Apache Spark (<ext-link ext-link-type="uri" xlink:href="https://spark.apache.org/">spark.apache.org</ext-link>), a state-of-the-art distributed computing platform (Although the ADNI data can be centralized, such a federated analysis would allow the method to be scaled up to much larger datasets, including genomic data that is difficult to centralize for logistic or regulatory reasons). As illustrated in <xref ref-type="fig" rid="F2">Figure 2</xref>, the samples in each institution can be further partitioned into at most three subgroups (<italic>g</italic> &#x0003D; 1, 2, 3) according to the subject&#x00027;s genotype at certain SNP loci (e.g., GG, GA, AA) or a gene (e.g., stratified by the three <italic>APOE</italic> genotypes considered in this study). Accordingly, <inline-formula><mml:math id="M2"><mml:msubsup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> and <inline-formula><mml:math id="M3"><mml:msubsup><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula>, respectively, represent the image biomarkers and gene expression values in the <italic>g</italic>th group of the <italic>i</italic>th institution. The data from the <italic>g</italic>th group in all <italic>I</italic> institutions will be fitted into a regression model in a federated strategy.</p>
</sec>
<sec>
<title>Federated Chow Test Analysis</title>
<p>Using federated linear regression, we can calculate four linear models for all the <italic>I</italic> institutions, including three models for three subgroups and one for all samples in the three subgroups. <inline-formula><mml:math id="M4"><mml:mover accent="false" class="mml-overline"><mml:mrow><mml:msup><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow><mml:mo accent="true">&#x000AF;</mml:mo></mml:mover></mml:math></inline-formula>, <inline-formula><mml:math id="M5"><mml:mover accent="false" class="mml-overline"><mml:mrow><mml:msup><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow><mml:mo accent="true">&#x000AF;</mml:mo></mml:mover></mml:math></inline-formula>, <inline-formula><mml:math id="M6"><mml:mover accent="false" class="mml-overline"><mml:mrow><mml:msup><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>3</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow><mml:mo accent="true">&#x000AF;</mml:mo></mml:mover></mml:math></inline-formula> and <inline-formula><mml:math id="M7"><mml:mover accent="false" class="mml-overline"><mml:mrow><mml:msup><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow><mml:mo accent="true">&#x000AF;</mml:mo></mml:mover></mml:math></inline-formula> are their optimal coefficient vectors. The Chow test assumes that the errors &#x003F5; are independent and identically distributed from a normal distribution by an unknown variance. The null hypothesis of the Chow test asserts that <inline-formula><mml:math id="M8"><mml:mover accent="false" class="mml-overline"><mml:mrow><mml:msup><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow><mml:mo accent="true">&#x000AF;</mml:mo></mml:mover></mml:math></inline-formula>, <inline-formula><mml:math id="M9"><mml:mover accent="false" class="mml-overline"><mml:mrow><mml:msup><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow><mml:mo accent="true">&#x000AF;</mml:mo></mml:mover></mml:math></inline-formula>, and <inline-formula><mml:math id="M10"><mml:mover accent="false" class="mml-overline"><mml:mrow><mml:msup><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>3</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow><mml:mo accent="true">&#x000AF;</mml:mo></mml:mover></mml:math></inline-formula> are equal. The predictive test suggested by Chow is then:</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M11"><mml:mtable class="eqnarray" columnalign="right center left"><mml:mtr><mml:mtd><mml:mi>F</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mo>-</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:msup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:msup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>3</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>/</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>2</mml:mn><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:msup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:msup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>3</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>/</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:msup><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:msup><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>3</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mo>-</mml:mo><mml:mn>3</mml:mn><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>S</italic><sup>(<italic>C</italic>)</sup> is the sum of squared residuals from the combined data from the three subgroups, <italic>S</italic><sup>(1)</sup> is the sum of squared residuals from the first group, and so on for <italic>S</italic><sup>(2)</sup> and <italic>S</italic><sup>(3)</sup>. <italic>N</italic><sup>(1)</sup>, <italic>N</italic><sup>(2)</sup>, and <italic>N</italic><sup>(3)</sup> are the number of samples in each subgroup, and <italic>k</italic> is the number of parameters. Under the null hypothesis, the test statistic follows the <italic>F</italic>-distribution with 2<italic>k</italic> and <italic>N</italic><sup>(1)</sup> &#x0002B; <italic>N</italic><sup>(2)</sup> &#x0002B; <italic>N</italic><sup>(3)</sup> &#x02212; 3<italic>k</italic> degrees of freedom. The global center will calculate <italic>F</italic> by gathering all the least square losses and the number of subjects for each subgroup and combined data from each institution. For example, for the first subgroup, the global least-square loss is <inline-formula><mml:math id="M12"><mml:msup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>I</mml:mi></mml:mrow></mml:munderover><mml:msubsup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:math></inline-formula> and the global subject number is <inline-formula><mml:math id="M13"><mml:msup><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>I</mml:mi></mml:mrow></mml:munderover><mml:msubsup><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:math></inline-formula>. Eventually, the <italic>p</italic>-value will be calculated at the global coordinating center and assigned to each institution.</p>
</sec>
<sec>
<title>Federated Linear Regression</title>
<p>Many regression models may be selected for the Chow test model, such as linear regression (<xref ref-type="bibr" rid="B46">46</xref>), polynomial regression (<xref ref-type="bibr" rid="B47">47</xref>), ridge regression (<xref ref-type="bibr" rid="B48">48</xref>), and so on. In this study, we focus on studying the differences in the relationships between imaging biomarkers and gene expression among different groups. Complex regression models, like polynomial regression, may lead to over-fitting and meaningless results. Also, sparse or penalized regression methods, such as ridge regression, require an appropriate regularization parameter. Therefore, in this work, linear regression would be the most rational choice.</p>
<p>Since the federated regression models for each subgroup are the same, we omit the group superscripts here. For the data in one subgroup of all the <italic>I</italic> institutions, we can calculate the linear regression equation as: <italic>y</italic> &#x0003D; <italic>Xw</italic> &#x0002B; &#x003F5;, where <italic>X</italic> &#x02208; <italic>R</italic><sup><italic>N</italic>&#x000D7;<italic>k</italic></sup> represents the independent variables, <italic>y</italic> &#x02208; <italic>R</italic><sup><italic>N</italic></sup> is a vector of the observations on a dependent variable, <italic>w</italic> &#x02208; <italic>R</italic><sup><italic>k</italic></sup> is a coefficient vector, and &#x003F5; &#x02208; <italic>R</italic><sup><italic>N</italic></sup> is the disturbance vector. <italic>N</italic> is the number of observations in the group, and <italic>k</italic> is the number of parameters. Then, the coefficient vector <italic>w</italic> can be estimated by minimizing the least squared function, <inline-formula><mml:math id="M14"><mml:mi>S</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:mfrac><mml:mtext>&#x000A0;</mml:mtext><mml:mo>|</mml:mo><mml:mo>|</mml:mo><mml:mi>X</mml:mi><mml:mi>w</mml:mi><mml:mo>-</mml:mo><mml:mi>y</mml:mi><mml:mo>|</mml:mo><mml:msubsup><mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:math></inline-formula>.</p>
<p>To avoid centralizing the data, (<italic>X</italic><sub><italic>i</italic></sub>, <italic>y</italic><sub><italic>i</italic></sub>), from each institution, we first rewrite the minimization problem as, <inline-formula><mml:math id="M15"><mml:mo class="qopname">min</mml:mo><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo class="qopname">&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>I</mml:mi></mml:mrow></mml:munderover><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>w</mml:mi><mml:mo>;</mml:mo><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:mfrac><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo class="qopname">&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>I</mml:mi></mml:mrow></mml:munderover><mml:mo>|</mml:mo><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mi>w</mml:mi><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo><mml:msubsup><mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:math></inline-formula>. Then, the global gradient can be calculated as, <inline-formula><mml:math id="M16"><mml:mo>&#x02207;</mml:mo><mml:mi>S</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>X</mml:mi><mml:mi>w</mml:mi><mml:mo>-</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>I</mml:mi></mml:mrow></mml:munderover><mml:msubsup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mi>w</mml:mi><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>I</mml:mi></mml:mrow></mml:munderover><mml:mo>&#x02207;</mml:mo><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>. Therefore, instead of centralizing the data, the global center only needs to gather the partial gradient, &#x02207;<italic>S</italic><sub><italic>i</italic></sub>(<italic>w</italic>), which is calculated with (<italic>X</italic><sub><italic>i</italic></sub>, <italic>y</italic><sub><italic>i</italic></sub>) at each local institution. After computing the global gradient, &#x02207;<italic>S</italic>(<italic>w</italic>), the global center will send it back to <italic>i</italic>th local institution. Finally, <italic>w</italic> will be updated at each institution by gradient descent with the same learning rate, <italic>w</italic> &#x02190; <italic>w</italic> &#x02212; &#x003B7;&#x02207;<italic>S</italic>(<italic>w</italic>). The reason for not updating <italic>w</italic> at the global center is to avoid possible data reconstruction. When <italic>w</italic> is zero, the local gradient sent to the center is <inline-formula><mml:math id="M17"><mml:mo>-</mml:mo><mml:msubsup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msubsup><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>. Then, the global center can easily acquire <inline-formula><mml:math id="M18"><mml:msubsup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msubsup><mml:msub><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mi>w</mml:mi></mml:math></inline-formula> and <italic>X</italic><sub><italic>i</italic></sub> might be reconstructed if <italic>w</italic> is known to the center. Consequently, our optimization strategy is able to preserve data privacy for all institutions. The whole framework of our federated Genotype-Expression-Image Integration model is summarized in <xref ref-type="table" rid="T6">Algorithm 1</xref>. And the code can be downloaded at our website, <ext-link ext-link-type="uri" xlink:href="https://github.com/JianfengWu1993/GEIDI">https://github.com/JianfengWu1993/GEIDI</ext-link>.</p>
<table-wrap position="float" id="T6">
<label>Algorithm 1</label>
<caption><p>Federated Genotype-Expression-Image Data Integration Model.</p></caption>
<table frame="hsides" rules="groups">
<tbody>
<tr>
<td align="left" valign="top">&#x000A0;&#x000A0;<bold>Input</bold>: Data pairs of the <italic>I</italic> institutions, (<italic>X</italic><sub>1</sub>, <italic>y</italic><sub>1</sub>), &#x02026;, (<italic>X</italic><sub><italic>i</italic></sub>, <italic>y</italic><sub><italic>i</italic></sub>), &#x02026;, (<italic>X</italic><sub><italic>I</italic></sub>, <italic>y</italic><sub><italic>I</italic></sub>) and the sample numbers of each group, <inline-formula><mml:math id="M19"><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mn>3</mml:mn></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mn>3</mml:mn></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>I</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>I</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>I</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mn>3</mml:mn></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:math></inline-formula></td>
</tr>
<tr>
<td align="left" valign="top">&#x000A0;&#x000A0;<bold>Output</bold>: <italic>p</italic>-value of the studying Genotype-Expression-Image trio</td>
</tr>
<tr>
<td align="left" valign="top">&#x000A0;&#x000A0;<bold>Initialize</bold>: <italic>w</italic><sup>(1)</sup>, <italic>w</italic><sup>(2)</sup>, <italic>w</italic><sup>(3)</sup>, <italic>w</italic><sup>(<italic>C</italic>)</sup> &#x0003D; <bold>0</bold></td>
</tr>
<tr>
<td align="left" valign="top">&#x000A0;&#x000A0;&#x000A0;&#x000A0;1:&#x000A0;&#x000A0;<bold>for</bold> <italic>g</italic> &#x0003D; {1, 2, 3, <italic>C</italic>} <bold>do</bold></td>
</tr>
<tr>
<td align="left" valign="top">&#x000A0;&#x000A0;&#x000A0;&#x000A0;2:&#x000A0;&#x000A0;&#x000A0;&#x000A0;<bold>while</bold> <italic>convergence and maximum number of iterations are not reached</italic> <bold>do</bold></td>
</tr>
<tr>
<td align="left" valign="top">&#x000A0;&#x000A0;&#x000A0;&#x000A0;3:&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;Get an image patch <bold>x</bold><sub><bold>i</bold></sub> from <bold>X</bold>.</td>
</tr>
<tr>
<td align="left" valign="top">&#x000A0;&#x000A0;&#x000A0;&#x000A0;4:&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;Each institution computes the gradient: <inline-formula><mml:math id="M20"><mml:mrow><mml:mo>&#x02207;</mml:mo><mml:msubsup><mml:mi>S</mml:mi><mml:mi>i</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>g</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msubsup><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msup><mml:mi>w</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>g</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mo stretchy='false'>[</mml:mo><mml:msubsup><mml:mi>X</mml:mi><mml:mi>i</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>g</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msubsup><mml:mo stretchy='false'>]</mml:mo></mml:mrow><mml:mi>T</mml:mi></mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msubsup><mml:mi>X</mml:mi><mml:mi>i</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>g</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msubsup><mml:msup><mml:mi>w</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>g</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo>&#x02212;</mml:mo><mml:msubsup><mml:mi>y</mml:mi><mml:mi>i</mml:mi><mml:mrow><mml:mi>g</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>g</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msubsup></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mrow></mml:math></inline-formula></td>
</tr>
<tr>
<td align="left" valign="top">&#x000A0;&#x000A0;&#x000A0;&#x000A0;5:&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;Global center computes and sends global gradient to each institution: <inline-formula><mml:math id="M21"><mml:mo>&#x02207;</mml:mo><mml:msup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>I</mml:mi></mml:mrow></mml:munderover><mml:mo>&#x02207;</mml:mo><mml:msubsup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:math></inline-formula></td>
</tr>
<tr>
<td align="left" valign="top">&#x000A0;&#x000A0;&#x000A0;&#x000A0;6:&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;Each institution updates the coefficient with the global gradient: <italic>w</italic><sup>(<italic>g</italic>)</sup> &#x02190; <italic>w</italic><sup>(<italic>g</italic>)</sup> &#x02212; &#x003B7;&#x02207;<italic>S</italic><sup>(<italic>g</italic>)</sup> (<italic>w</italic><sup>(<italic>g</italic>)</sup>).</td>
</tr>
<tr>
<td align="left" valign="top">&#x000A0;&#x000A0;&#x000A0;&#x000A0;7:&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;<bold>end while</bold></td>
</tr>
<tr>
<td align="left" valign="top">&#x000A0;&#x000A0;&#x000A0;&#x000A0;8:&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;Each institution calculates the sum of squared residual: <inline-formula><mml:math id="M22"><mml:msubsup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mo>;</mml:mo><mml:msubsup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:math></inline-formula></td>
</tr>
<tr>
<td align="left" valign="top">&#x000A0;&#x000A0;&#x000A0;&#x000A0;9:&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;Global center gathers the global sum of squared residual: <inline-formula><mml:math id="M23"><mml:msup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>I</mml:mi></mml:mrow></mml:munderover><mml:msubsup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:math></inline-formula>.</td>
</tr>
<tr>
<td align="left" valign="top">&#x000A0;&#x000A0;&#x000A0;&#x000A0;10:&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;Global center gathers the global sample numbers: <inline-formula><mml:math id="M24"><mml:msup><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>I</mml:mi></mml:mrow></mml:munderover><mml:msubsup><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:math></inline-formula>.</td>
</tr>
<tr>
<td align="left" valign="top">&#x000A0;&#x000A0;&#x000A0;&#x000A0;11:&#x000A0;&#x000A0;<bold>end for</bold></td>
</tr>
<tr>
<td align="left" valign="top">&#x000A0;&#x000A0;&#x000A0;&#x000A0;12:&#x000A0;&#x000A0;Global center calculates F value with equation (1) and then computes and sends <italic>p</italic>-value to all institutions.</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec>
<title>Performance Evaluation Protocol</title>
<p>We firstly use our model to identify AD-related gene expression. From the publicly available database, <ext-link ext-link-type="uri" xlink:href="http://alzgene.org">alzgene.org</ext-link>, and the GWAS results from International Genomics of Alzheimer&#x00027;s Project (IGAP) (<xref ref-type="bibr" rid="B6">6</xref>), we select 632 known AD-related genes. When we fix the genotype and imaging biomarker, we can calculate a <italic>p</italic>-value for each of the 20,211 gene expressions. We rank the 20,211 <italic>p</italic>-values and identify which of the known AD-related genes are featured in the top <italic>N</italic> gene expressions. In section Discovering AD-Related Gene Expressions, the top <italic>N</italic> gene expressions are the ones with a <italic>p</italic>-value &#x0003C; 0.05. In section Discovering AD-Related SNPs, <italic>N</italic> is a fixed number (100 and 200). We introduce hypergeometric analysis (<xref ref-type="bibr" rid="B49">49</xref>) to evaluate the model&#x00027;s performance to detect the known AD-related genes. The probability mass function of hypergeometric analysis is defined as,</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M25"><mml:mtable class="eqnarray" columnalign="right center left"><mml:mtr><mml:mtd><mml:mi>p</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mi>M</mml:mi><mml:mo>,</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mi>n</mml:mi><mml:mo>,</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mi>N</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none none none none none none none none none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:mi>n</mml:mi></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>k</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mtext>&#x000A0;</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none none none none none none none none none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:mi>M</mml:mi><mml:mo>-</mml:mo><mml:mi>n</mml:mi></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>N</mml:mi><mml:mo>-</mml:mo><mml:mi>k</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none none none none none none none none none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:mi>M</mml:mi></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>N</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>In our case, the number of population (<italic>M</italic>) is 20,211, the sample size (<italic>n</italic>) is 632, the number of samples drawn from the population (<italic>n</italic>) is the selected top <italic>N</italic> gene expressions, and the number of the observed successes (<italic>k</italic>) is the number of overlapping genes between 632 known AD-related genes and the top <italic>N</italic> gene expressions.</p>
<p>Secondly, with different genotypes, the pattern of hypergeometric enrichment will vary. The AD-related genotypes should, in general, have a more significant hypergeometric enrichment. From <ext-link ext-link-type="uri" xlink:href="http://alzgene.org">alzgene.org</ext-link>, we also obtain 1217 known AD-related SNPs. And we randomly select 1217 SNPs from the ADNI database as non-AD-related SNPs. After ranking the SNPs with the <italic>p</italic>-value based on hypergeometric analysis, we compute the number of AD-related SNPs found in the top <italic>m</italic> SNPs as true positive rate (TPR) and evaluate the performance of the models with TPR.</p>
<p>Finally, to prove the stability of our federated GEIDI, we compare the residuals of the federated linear regression model under different multi-site conditions. If the residuals are the same under different conditions, the <italic>F</italic> value and <italic>p</italic>-value will stay unchanged.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>Results</title>
<sec>
<title>Discovering AD-Related Gene Expressions</title>
<sec>
<title>APOE Related Gene Expressions</title>
<p><italic>APOE</italic> genotype is a well-known genetic biomarker for predicting subjects&#x00027; risk for AD. We stratify 697 subjects into three subgroups based on their <italic>APOE</italic> genotype status: non-carriers (e3/e3), heterozygotes (e3/e4), and homozygotes (e4/e4). Federated GEIDI is then adopted to discover genes correlated with hippocampus volume differentially across the three subgroups. We first run federated GEIDI with the volume of both sides of the hippocampus and the expression measures for 20,211 genes. Next, 1,625 gene expression measures are selected with <italic>p</italic> &#x0003C; 0.05. We evaluate the enrichment of these genes and the 632 AD-related genes annotated on <ext-link ext-link-type="uri" xlink:href="http://alzgene.org">alzgene.org</ext-link> and find 73 overlapping genes, yielding a hypergeometric enrichment <italic>p</italic> &#x0003D; 0.00039. Among the 73 overlapping genes, the top ten gene expressions are those measured for <italic>CAST, CST3, GSTO1, LSS, MS4A4A, NPC1, PMVK, PPM1H, PPP2R2B</italic>, and <italic>SORCS2</italic>. Besides, the top ten genes in the 1,625 gene expressions are <italic>IK, BRPF3, BTN3A2, LOC101929275, TDRG1, PAFAH1B1, SERINC3, ALKBH6, VPS45, and LGALS1</italic>. We also perform the false discovery rate (FDR) (<xref ref-type="bibr" rid="B50">50</xref>) test on the 20,211 <italic>p</italic>-values but none of the corrected <italic>p</italic>-values are significant. The list for these selected gene expressions is attached in <xref ref-type="supplementary-material" rid="SM1">Supplementary Material (table 1.csv)</xref>.</p>
<p>Additionally, we perform the same experiments on the volume of the middle temporal gyrus (MidTemp); the results are shown in <xref ref-type="table" rid="T2">Table 2</xref>. 2,415 gene expressions are significant and 92 of them overlap with the 632 AD-related genes - with a hypergeometric enrichment <italic>p</italic> &#x0003D; 0.00624. The top ten gene expressions are those measured for <italic>ABCA2, COL11A1, CST3, GNA11, HMOX1, HSPA1B, MAOA, MS4A4A, PRKAB2</italic>, and <italic>SORCS2</italic>. And the top ten genes in the 2,415 gene expressions are <italic>GLRA3, CAMK2N2, MCOLN2, BPIFA1, KIT, CST3, SLC20A2, LGALS4, TNFSF8</italic>, and <italic>LCOR</italic>. After performing FDR on the 20,211 genes, three gene expressions are significant, including <italic>GLRA3, CAMK2N2</italic>, and <italic>BPIFA1</italic>. The list for these selected gene expressions is attached in <xref ref-type="supplementary-material" rid="SM2">Supplementary Material (table 2.csv)</xref>.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Hypergeometric statistics for <italic>APOE</italic>.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th valign="top" align="left"><bold>Structures</bold></th>
<th valign="top" align="center"><bold>Selected genes</bold></th>
<th valign="top" align="center"><bold>Overlapping genes</bold></th>
<th valign="top" align="center"><bold><italic>P</italic>-value</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Hippocampus</td>
<td valign="top" align="center">1,625</td>
<td valign="top" align="center">73</td>
<td valign="top" align="center">0.00039</td>
</tr>
<tr>
<td valign="top" align="left">MidTemp</td>
<td valign="top" align="center">2,415</td>
<td valign="top" align="center">92</td>
<td valign="top" align="center">0.00624</td>
</tr>
<tr>
<td valign="top" align="left">Linear regression</td>
<td valign="top" align="center">2,657</td>
<td valign="top" align="center">98</td>
<td valign="top" align="center">0.00976</td>
</tr>
<tr>
<td valign="top" align="left">ANOVA</td>
<td valign="top" align="center">3,234</td>
<td valign="top" align="center">110</td>
<td valign="top" align="center">0.02665</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Matrix eQTL (<xref ref-type="bibr" rid="B51">51</xref>) is a state-of-the-art software to study the association between genotype and gene expression. We also leverage the linear model and the ANOVA model in Matrix eQTL to evaluate the <italic>APOE</italic> genotype and the measured expression levels of the 20,211 genes. For the linear model, there are 2,657 significant gene expressions and 98 overlapping genes, leading to a hypergeometric enrichment <italic>p</italic> &#x0003D; 9.76<italic>E</italic> &#x02212; 03. For the ANOVA model, 3,234 gene expressions are selected, and 110 known genes are found, which leads to a <italic>p</italic>-value &#x0003D; 2.665<italic>E</italic> &#x02212; 02. The results show that our federated GEIDI can detect the most gene candidates that are significantly enriched for known AD genes. As the volume of hippocampus has the best performance in detecting AD-related genes, we use it as the imaging biomarker for all the remaining experiments.</p>
</sec>
<sec>
<title>SNP Related Gene Expressions</title>
<p>In this experiment, we stratify the subjects into three subgroups based on their SNP status. We choose <italic>rs942439</italic>, as this SNP was reported in <ext-link ext-link-type="uri" xlink:href="http://alzgene.org">alzgene.org</ext-link>, and also one of the top hits in our experiment of discovering AD-related SNPs (the details about selecting AD-related SNPs will be introduced in section Discovering AD-Related SNPs). And we use the volume of both sides of hippocampus as the imaging biomarker because of its superior performance in the first experiment. Federated GEIDI is used to detect any known AD gene whose expression is differentially associated with hippocampus volume in the subgroups stratified by the genotype at <italic>rs942439</italic> locus.</p>
<p>As shown in <xref ref-type="table" rid="T3">Table 3</xref>, 1,587 gene expressions are significant and 60 of them are reported in <ext-link ext-link-type="uri" xlink:href="http://alzgene.org">alzgene.org</ext-link> and IGAP GWAS results, leading to a hypergeometric enrichment <italic>p</italic> &#x0003D; 0.017. Of these 60 gene expression measures, the top ten genes are <italic>ADRB1, ALOX5, ATXN1, CBS, FGF1, FLOT1, HSPA1A, RFTN1, SORL1</italic>, and <italic>XRCC1</italic>. Besides, the top ten genes in the 1,587 gene expressions are <italic>AIF1L, KRT23, CA2, C2ORF88, HSPA1A, LRGUK, LGALS3BP, IFT46, DDX23</italic>, and <italic>FAM166B</italic>. After performing the FDR test on the 20,211 genes, none of the gene expression is significant. The list for these selected gene expressions is attached in <xref ref-type="supplementary-material" rid="SM3">Supplementary Material (table 3.csv)</xref>.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Hypergeometric statistics for <italic>rs942439</italic>.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th valign="top" align="left"><bold>Structures</bold></th>
<th valign="top" align="center"><bold>Selected genes</bold></th>
<th valign="top" align="center"><bold>Overlapping genes</bold></th>
<th valign="top" align="center"><bold><italic>P</italic>-value</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Hippocampus</td>
<td valign="top" align="center">1,587</td>
<td valign="top" align="center">60</td>
<td valign="top" align="center">0.017</td>
</tr>
<tr>
<td valign="top" align="left">Linear regression</td>
<td valign="top" align="center">1,794</td>
<td valign="top" align="center">66</td>
<td valign="top" align="center">0.021</td>
</tr>
<tr>
<td valign="top" align="left">ANOVA</td>
<td valign="top" align="center">1,347</td>
<td valign="top" align="center">49</td>
<td valign="top" align="center">0.033</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>We also perform eQTL analysis on the SNP, <italic>rs942439</italic>. For a linear regression model, 1,794 gene expressions are selected, and, of these, 66 genes are reported in <ext-link ext-link-type="uri" xlink:href="http://alzgene.org">alzgene.org</ext-link> and IGAP GWAS results, yielding a hypergeometric enrichment <italic>p</italic> &#x0003D; 0.021. For the ANOVA model, 1,347 gene expression values are significant and, of these, 49 genes are reported in <ext-link ext-link-type="uri" xlink:href="http://alzgene.org">alzgene.org</ext-link> and IGAP; in this case, the hypergeometric enrichment was <italic>p</italic> &#x0003D; 0.033.</p>
<p>In the experiment, one of the most significant gene expression measures is for <italic>XRCC1</italic>, for which the <italic>p</italic>-value is 4.332<italic>E</italic> &#x02212; 03. <italic>XRCC1</italic> is a gene coding for the X-ray repair cross-complementing protein; it was previously reported to be weakly associated with AD in a Turkish population (<xref ref-type="bibr" rid="B52">52</xref>).</p>
<p>As shown in <xref ref-type="fig" rid="F3">Figure 3</xref>, we further adopt Pearson&#x00027;s correlation to evaluate the relationship between the hippocampal volume (<italic>x</italic>-axis) (adjusted for ICV) and <italic>XRCC1</italic> gene expression (<italic>y</italic>-axis) of each subgroup. <xref ref-type="fig" rid="F3">Figure 3A</xref> illustrates the distribution for all the samples. <xref ref-type="fig" rid="F3">Figures 3B&#x02013;D</xref> show the distribution for the samples with &#x0201C;GG&#x0201D;, &#x0201C;GA&#x0201D; and &#x0201C;AA&#x0201D; genotype, respectively. Above each subfigure, <italic>R</italic> and <italic>p</italic> are the Pearson correlation coefficient and <italic>p</italic>-value, and N is the number of subjects. Even so, there is always some missing information in the genotype data. Hence, before we run federated GEIDI as well as the Pearson correlation statistics, we remove the subjects without the specific genotype. Because of this, the total number N in <xref ref-type="fig" rid="F3">Figure 3A</xref> is 579 instead of 697. We find samples with an &#x0201C;AA&#x0201D; genotype had hippocampal volume negatively correlated with expression levels of <italic>XRCC1</italic> (<italic>N</italic> = 37, <italic>R</italic> = 0.37, <italic>p</italic> = 0.022). In contrast, the analysis in all samples (<xref ref-type="fig" rid="F3">Figure 3A</xref>) or subjects with either &#x0201C;GG&#x0201D; or &#x0201C;GA&#x0201D; genotype (<xref ref-type="fig" rid="F3">Figures 3B,C</xref>) showed that the Pearson correlation coefficients were not significant in the overall, pooled sample. This result indicates that our method can establish associations among SNP, imaging, and gene expression data that include known AD risk factors.</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>Correlation of image biomarkers and <italic>XRCC1</italic> gene expression in subpopulations stratified by the sample&#x00027;s genotype at <italic>rs942439</italic>. <bold>(A)</bold> all samples <bold>(B)</bold> individuals with &#x0201C;GG&#x0201D; genotype; <bold>(C)</bold> those with &#x0201C;GA&#x0201D; genotype <bold>(D)</bold> those with &#x0201C;AA&#x0201D; genotype.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fradi-01-777030-g0003.tif"/>
</fig>
<p>We further apply the above procedure to discover genes that have never been reported to be associated with AD. As shown in <xref ref-type="fig" rid="F4">Figure 4D</xref>, <italic>SEC14L2</italic> gene expression is negatively associated with hippocampal volume only in the subpopulation with &#x0201C;AA&#x0201D; genotype at rs942439 locus (<italic>N</italic> = 37, <italic>R</italic> = &#x02212;0.47, <italic>p</italic> = 0.003). Interestingly, the opposite correlation is found in a subpopulation with &#x0201C;GA&#x0201D; genotype (<xref ref-type="fig" rid="F4">Figure 4C</xref>, <italic>N</italic> = 208, <italic>R</italic> = 0.15, <italic>P</italic> = 0.03), and when applied to all pooled subjects, the total population does not show significant correlations (<xref ref-type="fig" rid="F4">Figure 4A</xref>, <italic>N</italic> = 579, <italic>R</italic> = 0.07, <italic>p</italic> = 0.09) and the subpopulation with &#x0201C;GG&#x0201D; genotype doesn&#x00027;t show any significant correlations (<xref ref-type="fig" rid="F4">Figure 4B</xref>, <italic>N</italic> = 334, <italic>R</italic> = 0.062, <italic>p</italic> = 0.258). The <italic>SEC14L2</italic> gene encodes a protein that stimulates squalene monooxygenase, a downstream enzyme in the cholesterol biosynthesis pathway. This gene has never been reported to be associated with AD, but high cholesterol levels have been linked to early-onset AD (<xref ref-type="bibr" rid="B53">53</xref>). This result indicates that our method can detect strong correlations in specific subpopulations that cannot be detected in the whole population. We also observe conflicting directions in different subpopulations, as shown by &#x0201C;GA&#x0201D; and &#x0201C;AA&#x0201D; subpopulations showing opposite correlations. This also highlights the importance of individualized medicine in patient management, as the same drug may have opposing effects in different groups of samples. Thus, federated GEIDI offers a new approach to discover novel genes related to AD as potential drug targets.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>Correlation of image biomarkers and <italic>SEC14L2</italic> gene expression in subpopulation stratified by the sample&#x00027;s genotype at <italic>rs942439</italic>. <bold>(A)</bold> all samples <bold>(B)</bold> those with &#x0201C;GG&#x0201D; genotype; <bold>(C)</bold> those with &#x0201C;GA&#x0201D; genotype; <bold>(D)</bold> those with &#x0201C;AA&#x0201D; genotype.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fradi-01-777030-g0004.tif"/>
</fig>
</sec>
</sec>
<sec>
<title>Discovering AD-Related SNPs</title>
<p>In the experiments of section Discovering AD-Related Gene Expressions, we used hypergeometric statistics to evaluate the ability of our proposed model to discover AD-related gene expressions that are differentially associated with imaging measures in populations stratified by <italic>APOE</italic> haplotype. In this experiment, we also use hypergeometric statistics to assess the discovery rate of known AD-related genes, in the set of genes whose expression shows different correlations with imaging markers, in samples stratified according to different genotypes. Sets that are enriched in AD-related SNPs will have a more significant <italic>p</italic>-value in the hypergeometric test that assesses enrichment. Since the hippocampal volume measure showed superior performance for this task, among all the imaging biomarkers in section Discovering AD-Related Gene Expressions, we adopt it as the brain imaging measure in this experiment. To illustrate the effectiveness of our GEIDI model, we perform the same experiment with the linear model in Matrix eQTL, which can evaluate the associations between SNPs and gene expression. To adjust for multiple comparisons, we will convert raw <italic>p</italic>-values to false discovery rate (FDR) and consider trios with FDR &#x0003C; 0.05 as functionally important.</p>
<p>When we analyze each SNP with our federated GEIDI and Matrix eQTL, we will obtain a <italic>p</italic>-value for each of the 20,211 expressed genes. Instead of selecting the significant gene expressions with a <italic>p</italic>-value &#x0003C; 0.05, we respectively rank the <italic>p</italic>-value of all the gene expressions calculated by the two methods and select the top <italic>N</italic> (100 and 200) gene expressions to apply the hypergeometric analysis. With the <italic>p</italic>-value from this hypergeometric analysis (which assesses enrichment for known AD-associated genes), we may rank the SNPs and obtain the most AD-related ones. Then, we try to prove that our GEIDI is able to detect more AD-related SNPs. From <ext-link ext-link-type="uri" xlink:href="https://www.alzGene.org">alzgene.org</ext-link>, we also created a list of 1,217 AD-related SNPs, and we randomly selected another 1,217 SNPs as the non-AD-related ones. After ranking the SNPs with the <italic>p</italic>-value computed by the two methods, we calculate the true positive rate (TPR) for the top <italic>m</italic> SNPs, which measures the percentage of AD-related SNPs in the selected top <italic>m</italic> SNPs. For example, the last number in <xref ref-type="table" rid="T4">Table 4</xref> is 0.57, which means 57% of the top 500 SNPs are AD-related ones. As the results in <xref ref-type="table" rid="T4">Table 4</xref>, our federated GEIDI can always achieve superior performance than Matrix eQTL.</p>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>True Positive Rates of AD-related SNPs in the top <italic>m</italic> SNPs.</p></caption>
<table frame="box" rules="all">
<thead><tr>
<th valign="top" align="left"><bold><inline-graphic xlink:href="fradi-01-777030-i0001.tif"/></bold></th>
<th valign="top" align="center"><bold>10</bold></th>
<th valign="top" align="center"><bold>50</bold></th>
<th valign="top" align="center"><bold>100</bold></th>
<th valign="top" align="center"><bold>200</bold></th>
<th valign="top" align="center"><bold>500</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left" colspan="6"><bold>Matrix eQTL: linear regression</bold></td>
</tr>
<tr>
<td valign="top" align="left">100<break/>200</td>
<td valign="top" align="center">0.50<break/>0.50</td>
<td valign="top" align="center">0.58<break/>0.60</td>
<td valign="top" align="center">0.52<break/>0.55</td>
<td valign="top" align="center">0.50<break/>0.58</td>
<td valign="top" align="center">0.53<break/>0.54</td>
</tr>
<tr>
<td valign="top" align="left" colspan="6"><bold>Matrix eQTL: ANOVA</bold></td>
</tr>
<tr>
<td valign="top" align="left">100<break/>200</td>
<td valign="top" align="center">0.60<break/><bold>0.60</bold></td>
<td valign="top" align="center">0.58<break/>0.60</td>
<td valign="top" align="center">0.52<break/>0.57</td>
<td valign="top" align="center">0.49<break/>0.55</td>
<td valign="top" align="center">0.55<break/>0.55</td>
</tr>
<tr>
<td valign="top" align="left" colspan="6"><bold>Federated GEIDI</bold></td>
</tr>
<tr>
<td valign="top" align="left">100<break/>200</td>
<td valign="top" align="center"><bold>0.60</bold><break/><bold>0.60</bold></td>
<td valign="top" align="center"><bold>0.60</bold><break/><bold>0.62</bold></td>
<td valign="top" align="center"><bold>0.60</bold><break/><bold>0.61</bold></td>
<td valign="top" align="center"><bold>0.61</bold><break/><bold>0.58</bold></td>
<td valign="top" align="center"><bold>0.60</bold><break/><bold>0.57</bold></td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p><italic>The SNPs are ranked with the p-value from hypergeometric analysis with the top. N gene expressions as the number of samples drawn from the population</italic>.</p>
</table-wrap-foot>
</table-wrap>
<p>In <xref ref-type="fig" rid="F5">Figure 5</xref>, we visualize the <italic>p</italic>-values of these 2,434 SNPs from hypergeometric analysis in the Manhattan plots. The top figure is the Manhattan plot for the result with the top 100 gene expressions and the bottom one is for the result of the top 200 gene expressions. The SNPs, <italic>rs4889013</italic> and <italic>rs11940059</italic>, are the top-ranked ones for both results. When we select 100 or 200 as the number of samples drawn from the population, three parameters in Equation (2) are fixed and only the number of observed successes, <italic>k</italic>, varies for different SNPs. Therefore, the <italic>p</italic>-value from different SNPs might be the same if their numbers of observed successes are the same. This explains why results of some SNPs locate at the same horizontal position.</p>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>Manhattan plots for the results of federated GEIDI. The top figure is the Manhattan plot for the results from hypergeometric analysis with the top 100 gene expressions as the number of samples drawn from the population and the bottom one is for the results with the top 200 gene expressions. The SNPs, <italic>rs4889013</italic> and <italic>rs11940059</italic>, are the top-ranked ones for both results.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fradi-01-777030-g0005.tif"/>
</fig>
</sec>
<sec>
<title>Federated Learning Stability Analysis</title>
<p>In this experiment, we aim to demonstrate that the performance of our federated GEIDI model is not greatly affected by different data distribution models across institutions. In practice, it would be convenient and efficient to run association tests on data that might be distributed across multiple servers without transferring it all to a centralized location. We developed this algorithm with R language and simulated the distributed condition on a cluster with several conventional x86 nodes, of which each contains two Intel Xeon E5-2680 v4 CPUs running at 2.40 GHz. Each institution is assigned one computing node. We synthesized 1,000 samples and randomly assigned them to different independent hypothetical institutions, including one institution, three institutions, five institutions and seven institutions. We compared the residuals from each linear regression model for each condition and found the residuals remained unchanged, as shown in <xref ref-type="table" rid="T5">Table 5</xref>. The first column is the ground truth residual and the rest are the residuals for our federated linear model under different data distribution conditions. The residuals are the same, which means that the results of our Federated GEIDI will remain stable under different multi-site conditions. Therefore, these results demonstrate the correctness and stability of our federated GEIDI model.</p>
<table-wrap position="float" id="T5">
<label>Table 5</label>
<caption><p>Stability analysis of federated GEIDI across different institutional settings.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th/>
<th valign="top" align="left"><bold>Ground truth</bold></th>
<th valign="top" align="center"><bold>1-institution</bold></th>
<th valign="top" align="center"><bold>3-institution</bold></th>
<th valign="top" align="center"><bold>5-institution</bold></th>
<th valign="top" align="center"><bold>7-institution</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Residual</td>
<td valign="top" align="left">3.9553</td>
<td valign="top" align="center">3.9553</td>
<td valign="top" align="center">3.9553</td>
<td valign="top" align="center">3.9553</td>
<td valign="top" align="center">3.9553</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>Discussion</title>
<p>In this work, we propose a novel federated Genotype-Expression-Imaging Data Integration (GEIDI) model to identify the genetic and transcriptomic influences on brain sMRI measures. We performed various experiments with our model on the publicly available ADNI dataset, and we have two main findings. First, our federated GEIDI is an effective multimodal approach that provides novel insights into the relationship among image biomarkers, genotypes, and gene expression, and may be useful to discover novel genes as potential AD drug targets. It has better performance in detecting AD-related gene expressions and SNPs than the linear regression model and ANOVA model in the state-of-the-art Matrix eQTL approach. In addition, our model may not only detect known AD-associated genes as potential drug targets, such as <italic>XRCC1</italic>, but may also help in discovering novel genes as potential drug targets, such as <italic>SEC14L2</italic>. Second, compared to Matrix eQTL, our federated GEIDI provides a way to investigate extremely large datasets from different institutions without violating data privacy. The statistical power of the model will also be increased with the larger sample size. Our work may lay down a solid foundation for future multi-site large-scale imaging genetics research.</p>
<sec>
<title>Comparison Analysis of Federated GEIDI and Matrix eQTL</title>
<p>Expression quantitative trait loci (eQTL) analysis (<xref ref-type="bibr" rid="B54">54</xref>, <xref ref-type="bibr" rid="B55">55</xref>) is designed to identify the significant associations between SNPs and gene expression, which can help understand the biochemical processes occurring in living systems, discover the genetic factors that influence the onset and progression of certain diseases, and determine the pathways affected by them. There are many eQTL analysis methods, including linear regression, ANOVA models, Bayesian regression (<xref ref-type="bibr" rid="B56">56</xref>), and so on. Matrix eQTL (<xref ref-type="bibr" rid="B51">51</xref>) is the state-of-the-art software for computationally efficient eQTL analysis, and it supports additive linear and ANOVA models. It has been widely used in the study of human genetic traits and diseases. However, it has two main limitations. First, although Matrix eQTL is very computationally efficient, it cannot work on data that is distributed across different institutions. Nowadays, unprecedentedly large volumes of biomedical and genetic data have been collected by different hospitals and research institutions, and this aggregate of available data may significantly advance the study of factors influencing disease. However, data restrictions, legal complexities, and patient privacy have all been major obstacles for researchers to obtain or share these data. Therefore, federated machine learning and distributed statistical models are becoming advantageous for current research on medical data (<xref ref-type="bibr" rid="B57">57</xref>, <xref ref-type="bibr" rid="B58">58</xref>). Second, the models in Matrix eQTL cannot jointly consider the information from images. Changes in brain structures can play a vital role in the study and diagnosis of Alzheimer&#x00027;s disease, and many researchers have attempted to detect associations between genetic factors and imaging features (<xref ref-type="bibr" rid="B21">21</xref>, <xref ref-type="bibr" rid="B59">59</xref>, <xref ref-type="bibr" rid="B60">60</xref>). Therefore, introducing imaging information may greatly assist the detection of genetic factors that influence disease as an intermediate phenotype that might reflect relevant disease processes.</p>
<p>Our proposed federated Genotype-Expression-Imaging Data Integration model can effectively overcome these two obstacles. In the Methods section, we detailed how our model maintains each institutional data private. Additionally, our federated GEIDI model integrates GWAS data, gene expression, and imaging data. The experimental results demonstrate that our federated GEIDI model has better performance in detecting AD-related genes and SNPs. In detecting AD-related gene expression, our model achieves the strongest hypergeometric enrichment with the volume of the hippocampus. In our tests detecting AD-related SNPs, our federated GEIDI model generally obtained a higher TPR than the linear regression model and ANOVA model. Besides, compared with existing methods, our proposed model offers novel insights into the relationship among image biomarkers, genotypes, and gene expression by considering both imaging and gene expression features&#x02014;which can vary over time&#x02014;and understanding how they are affected by an individual&#x00027;s SNPs. Compared with Matrix eQTL, the only caveat of our model is the computation time on a single computing node. Our proposed model may require more computation time compared to Matrix eQTL. For each trio, our framework has to solve four linear regression models. But Matrix eQTL only needs to calculate one correlation matrix to evaluate all the trios. Therefore, our model may require more computation time. We perform the experiments of section APOE Related Gene Expressions on a single computing node. Matrix eQTL only takes 1.4 s, while our model may require 265 s. However, due to the federated learning nature, our work may be applied to different computation nodes parallelly. It may make our work scalable to large datasets and result in comparable computation times with Matrix eQTL.</p>
</sec>
<sec>
<title>Drug Target for Precision Medicine of AD</title>
<p>Increasingly, a major challenge in healthcare is that many drugs are adequate for only small subgroups of patients (<xref ref-type="bibr" rid="B61">61</xref>). Some patients may not only suffer from adverse side effects but also waste money on ineffective drugs. Precision medicine has the potential to tailor therapy based on the best expected response and highest safety margin to ensure better patient care. By enabling each patient to receive earlier diagnoses, risk assessments, and optimal treatments, personalized medicine holds promise for improving health care while also potentially lowering costs (<xref ref-type="bibr" rid="B10">10</xref>). In this work, our multi-omics approach offers potential in genome-guided drug discovery. Compared to state-of-the-art methods, our model performs better in detecting AD-related genes and SNPs. Moreover, our model not only detects known genes for target drugs, like <italic>XRCC1</italic>, but also discovers novel potential gene expressions, like <italic>SEC14L2</italic>. Meanwhile, our federated framework may integrate data from multiple sources without violating the data privacy and the obtained larger sample size may help discover and understand more AD-related genetic information. Therefore, we believe our federated GEIDI model will play an important role in the study of precision medicine for AD in the future.</p>
</sec>
<sec>
<title>Limitations and Future Work</title>
<p>Despite the promising results of our federated GEIDI model, there are four caveats. Firstly, we only evaluated our model on data from 697 subjects from the publicly available ADNI dataset. In the future, we will add other datasets to make results more robust and reliable. For example, the Arizona APOE cohort (AZ APOE cohort) recruited 450 actively followed participants matched by age, sex, and education&#x02014;including homozygous <italic>APOE-e4</italic> carriers and non-<italic>e4</italic> carriers since 1994 (<xref ref-type="bibr" rid="B62">62</xref>). The UK Biobank project (<xref ref-type="bibr" rid="B63">63</xref>) collects both large-scale genetic-genomic and phenotypic data as well as health-related information from around 500,000 volunteer participants in the UK. Assessments include biological measures, blood- and urine-based biomarkers, body, and brain imaging scans, and lifestyle parameters (<xref ref-type="bibr" rid="B64">64</xref>, <xref ref-type="bibr" rid="B65">65</xref>). Second, the volumes of specific subcortical structures may not be ideal imaging measurements for the multiple biological processes involved in Alzheimer&#x00027;s disease. Surface-based morphometry analyses have achieved excellent performance for early AD detection (<xref ref-type="bibr" rid="B66">66</xref>&#x02013;<xref ref-type="bibr" rid="B68">68</xref>). In recent work (<xref ref-type="bibr" rid="B69">69</xref>, <xref ref-type="bibr" rid="B70">70</xref>), the authors created tools to generate a univariate morphometry index (UMI) for surface morphometry features on regions of interest (ROIs) that are related to beta-amyloid deposition. This induced UMI may reflect intrinsic morphological changes induced by processes of amyloid accumulation in AD and has greater signal-to-noise ratio and strong generalizability to new subjects. If we were to use such brain pathology induced UMI measures instead of volumes, our federated GEIDI model may detect additional AD-related genes whose expression is influenced by SNPs. Thirdly, all the three methods (our GEIDI model, linear regression model, and ANOVA model) can only get several significant corrected <italic>p</italic>-values on this dataset with the FDR test since only about 10 percent of the 20,211 gene expressions can get significant raw <italic>p</italic>-values. Therefore, we use hypergeometric analysis to evaluate the top findings of these methods. The hypergeometric analysis is also a classical approach to evaluate the discovery significance in genetics research (<xref ref-type="bibr" rid="B71">71</xref>&#x02013;<xref ref-type="bibr" rid="B73">73</xref>). It may justify our model. In the future, we will try to apply our work to larger datasets and apply the multiple test correction models for justification. Finally, in ongoing work on blood-based biomarkers (<xref ref-type="bibr" rid="B74">74</xref>, <xref ref-type="bibr" rid="B75">75</xref>), plasma levels of amyloid-beta (plasma A&#x003B2;) may provide an alternative but highly accurate estimate of brain amyloid positivity. In (<xref ref-type="bibr" rid="B75">75</xref>), plasma P-Tau181 accurately discriminated AD dementia from non-AD neurodegenerative diseases with an excellent AUC (0.94). Similarly, such plasma measures might be used in conjunction with our federated GEIDI model to better understand the effects of AD-related genotypes. We plan to analyze such datasets to further evaluate our model in the future.</p>
</sec>
</sec>
<sec sec-type="conclusions" id="s5">
<title>Conclusion</title>
<p>We propose a novel federated Genotype-Expression-Image Data Integration model. Compared to similar studies, this work achieves state-of-the-art performance in discovering downstream effects of AD-related genes and SNPs. Besides, the model provides novel insights into the relationship among image biomarkers, genotypes, and gene expression and could discover novel drug targets for precision medicine. In the future, we will further validate our model with more datasets and more advanced imaging biomarkers. Specifically, we will introduce blood-based biomarkers into our model when such data are available.</p>
</sec>
<sec sec-type="data-availability" id="s6">
<title>Data Availability Statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: <ext-link ext-link-type="uri" xlink:href="http://adni.loni.usc.edu/data-samples/data-types/genetic-data/">http://adni.loni.usc.edu/data-samples/data-types/genetic-data/</ext-link>.</p>
</sec>
<sec id="s7">
<title>Ethics Statement</title>
<p>The studies involving human participants were reviewed and approved by Alzheimer&#x00027;s Disease Neuroimaging Initiative. Written informed consent for participation was not required for this study in accordance with the national legislation and the institutional requirements.</p>
</sec>
<sec id="s8">
<title>Alzheimer&#x00027;s Disease Neuroimaging Initiative</title>
<p>Data collection and sharing for this project was funded by the Alzheimer&#x00027;s Disease Neuroimaging Initiative (ADNI) (National Institutes of Health Grant U01 AG024904) and DoD ADNI (Department of Defense award number W81XWH-12-2-0012). ADNI was funded by the National Institute on Aging, the National Institute of Biomedical Imaging and Bioengineering, and through generous contributions from the following: Alzheimer&#x00027;s Association; Alzheimer&#x00027;s Drug Discovery Foundation; BioClinica, Inc.; Biogen Idec Inc.; Bristol-Myers Squibb Company; Eisai Inc.; Elan Pharmaceuticals, Inc.; Eli Lilly and Company; F. Hoffmann-La Roche Ltd and its affiliated company Genentech, Inc.; GE Healthcare; Innogenetics, N.V.; IXICO Ltd.; Janssen Alzheimer&#x00027;s Immunotherapy Research &#x00026; Development, LLC.; Johnson &#x00026; Johnson Pharmaceutical Research &#x00026; Development LLC.; Medpace, Inc.; Merck &#x00026; Co., Inc.; Meso Scale Diagnostics, LLC.; NeuroRx Research; Novartis Pharmaceuticals Corporation; Pfizer Inc.; Piramal Imaging; Servier; Synarc Inc.; and Takeda Pharmaceutical Company. The Canadian Institutes of Health Research is providing funds to support ADNI clinical sites in Canada. Private sector contributions are facilitated by the Foundation for the National Institutes of Health (<ext-link ext-link-type="uri" xlink:href="http://www.fnih.org">www.fnih.org</ext-link>). The grantee organization is the Northern California Institute for Research and Education, and the study is coordinated by the Alzheimer&#x00027;s Disease Cooperative Study at the University of California, San Diego. ADNI data are disseminated by the Laboratory for Neuro Imaging at the University of Southern California.</p>
</sec>
<sec id="s9">
<title>Author Contributions</title>
<p>JWu: methodology, investigation, formal analysis, and writing&#x02014;original draft. YC: investigation. PW: methodology and conceptualization. RC: review and editing. PT: methodology and review and editing. JWa: conceptualization, supervision, funding acquisition, and writing&#x02014;review and editing. YW: conceptualization, investigation, supervision, funding acquisition, and writing&#x02014;review and editing. All authors contributed to the article and approved the submitted version.</p>
</sec>
<sec sec-type="funding-information" id="s10">
<title>Funding</title>
<p>Algorithm development and image analysis for this study were partially supported by the ASU/Mayo Seed Grant Program, the National Institute on Aging (RF1AG051710, R21AG065942, U01AG068057, R01AG069453, and P30AG072980), the National Library of Medicine (R01LM013438), the National Institute of Biomedical Imaging and Bioengineering (R01EB025032), the National Eye Institute (R01EY032125), and the Arizona Alzheimer&#x00027;s Consortium.</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x00027;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
</body>
<back>
<ack><p>Data used in preparing this article were obtained from the Alzheimer&#x00027;s Disease Neuroimaging Initiative (ADNI) database (<ext-link ext-link-type="uri" xlink:href="https://www.adni.loni.usc.edu">adni.loni.usc.edu</ext-link>). As such, many investigators within the ADNI contributed to the design and implementation of ADNI and/or provided data but did not participate in the analysis or writing of this report. A complete listing of ADNI investigators can be found at: <ext-link ext-link-type="uri" xlink:href="http://adni.loni.usc.edu/wp-content/uploads/how_to_apply/ADNI_Acknowledgement_List.pdf">http://adni.loni.usc.edu/wp-content/uploads/how_to_apply/ADNI_Acknowledgement_List.pdf</ext-link>.</p>
</ack><sec sec-type="supplementary-material" id="s12">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fradi.2021.777030/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fradi.2021.777030/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Table_1.csv" id="SM1" mimetype="text/csv" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_2.csv" id="SM2" mimetype="text/csv" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table_3.csv" id="SM3" mimetype="text/csv" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Brookmeyer</surname> <given-names>R</given-names></name> <name><surname>Johnson</surname> <given-names>E</given-names></name> <name><surname>Ziegler-Graham</surname> <given-names>K</given-names></name> <name><surname>Arrighi</surname> <given-names>HM</given-names></name></person-group>. <article-title>Forecasting the global burden of Alzheimer&#x00027;s disease</article-title>. <source>Alzheimer Dementia.</source> (<year>2007</year>) <volume>3</volume>:<fpage>186</fpage>&#x02013;<lpage>91</lpage>. <pub-id pub-id-type="doi">10.1016/j.jalz.2007.04.381</pub-id><pub-id pub-id-type="pmid">19595937</pub-id></citation></ref>
<ref id="B2">
<label>2.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hyman</surname> <given-names>BT</given-names></name></person-group>. <article-title>Amyloid-dependent and amyloid-independent stages of Alzheimer disease</article-title>. <source>Arch Neurol.</source> (<year>2011</year>) <volume>68</volume>:<fpage>1062</fpage>&#x02013;<lpage>4</lpage>. <pub-id pub-id-type="doi">10.1001/archneurol.2011.70</pub-id><pub-id pub-id-type="pmid">21482918</pub-id></citation></ref>
<ref id="B3">
<label>3.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jill</surname> <given-names>M</given-names></name> <name><surname>Martin</surname> <given-names>F</given-names></name> <name><surname>Bernardino</surname> <given-names>G</given-names></name> <name><surname>Benson</surname> <given-names>MD</given-names></name></person-group>. <article-title>A mutation in the amyloid precursor protein associated with hereditary Alzheimer&#x00027;s disease</article-title>. <source>Science.</source> (<year>1991</year>) <volume>254</volume>:<fpage>97</fpage>&#x02013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1126/science.1925564</pub-id><pub-id pub-id-type="pmid">1925564</pub-id></citation></ref>
<ref id="B4">
<label>4.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kunkle</surname> <given-names>BW</given-names></name> <name><surname>Grenier-Boley</surname> <given-names>B</given-names></name> <name><surname>Sims</surname> <given-names>R</given-names></name> <name><surname>Bis</surname> <given-names>JC</given-names></name> <name><surname>Damotte</surname> <given-names>V</given-names></name> <name><surname>Naj</surname> <given-names>AC</given-names></name> <etal/></person-group>. <article-title>Genetic meta-analysis of diagnosed Alzheimer&#x00027;s disease identifies new risk loci and implicates A&#x003B2;, tau, immunity and lipid processing</article-title>. <source>Nat Genet.</source> (<year>2019</year>) <volume>51</volume>:<fpage>414</fpage>&#x02013;<lpage>30</lpage>. <pub-id pub-id-type="doi">10.1038/s</pub-id>,41588-019-0358-2<pub-id pub-id-type="pmid">31417202</pub-id></citation></ref>
<ref id="B5">
<label>5.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bertram</surname> <given-names>L</given-names></name> <name><surname>McQueen</surname> <given-names>MB</given-names></name> <name><surname>Mullin</surname> <given-names>K</given-names></name> <name><surname>Blacker</surname> <given-names>D</given-names></name> <name><surname>Tanzi</surname> <given-names>RE</given-names></name></person-group>. <article-title>Systematic meta-analyses of Alzheimer disease genetic association studies: the AlzGene database</article-title>. <source>Nat Genet.</source> (<year>2007</year>) <volume>39</volume>:<fpage>17</fpage>&#x02013;<lpage>23</lpage>. <pub-id pub-id-type="doi">10.1038/ng1934</pub-id><pub-id pub-id-type="pmid">17192785</pub-id></citation></ref>
<ref id="B6">
<label>6.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lambert</surname> <given-names>J-C</given-names></name> <name><surname>Ibrahim-Verbaas</surname> <given-names>CA</given-names></name> <name><surname>Harold</surname> <given-names>D</given-names></name> <name><surname>Naj</surname> <given-names>AC</given-names></name> <name><surname>Sims</surname> <given-names>R</given-names></name> <etal/></person-group>. <article-title>Meta-analysis of 74,046 individuals identifies 11 new susceptibility loci for Alzheimer&#x00027;s disease</article-title>. <source>Nat Genet.</source> (<year>2013</year>) <volume>45</volume>:<fpage>1452</fpage>&#x02013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1038/ng.2802</pub-id><pub-id pub-id-type="pmid">24162737</pub-id></citation></ref>
<ref id="B7">
<label>7.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mormino</surname> <given-names>EC</given-names></name> <name><surname>Sperling</surname> <given-names>RA</given-names></name> <name><surname>Holmes</surname> <given-names>AJ</given-names></name> <name><surname>Buckner</surname> <given-names>RL</given-names></name> <name><surname>de Jager</surname> <given-names>PL</given-names></name> <name><surname>Smoller</surname> <given-names>JW</given-names></name> <etal/></person-group>. <article-title>Polygenic risk of Alzheimer disease is associated with early- and late-life processes</article-title>. <source>Neurology.</source> (<year>2016</year>) <volume>87</volume>:<fpage>481</fpage>&#x02013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1212/WNL.0000000000002922</pub-id><pub-id pub-id-type="pmid">27385740</pub-id></citation></ref>
<ref id="B8">
<label>8.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Singanamalli</surname> <given-names>A</given-names></name> <name><surname>Wang</surname> <given-names>H</given-names></name> <name><surname>Madabhushi</surname> <given-names>A</given-names></name> <name><surname>Initiative</surname> <given-names>ADN</given-names></name></person-group>. <article-title>Cascaded multi-view canonical correlation (CaMCCo) for early diagnosis of Alzheimer&#x00027;s disease via fusion of clinical, imaging and omic features</article-title>. <source>Sci Rep.</source> (<year>2017</year>) <volume>7</volume>:<fpage>8137</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-017-03925-0</pub-id><pub-id pub-id-type="pmid">28811553</pub-id></citation></ref>
<ref id="B9">
<label>9.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Freudenberg-Hua</surname> <given-names>Y</given-names></name> <name><surname>Li</surname> <given-names>W</given-names></name> <name><surname>Davies</surname> <given-names>P</given-names></name></person-group>. <article-title>The role of genetics in advancing precision medicine for alzheimer&#x00027;s disease&#x02014;a narrative review</article-title>. <source>Front Med.</source> (<year>2018</year>) <volume>5</volume>:<fpage>108</fpage>. <pub-id pub-id-type="doi">10.3389/fmed.2018.00108</pub-id><pub-id pub-id-type="pmid">29740579</pub-id></citation></ref>
<ref id="B10">
<label>10.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vogenberg</surname> <given-names>FR</given-names></name> <name><surname>Isaacson Barash</surname> <given-names>C</given-names></name> <name><surname>Pursel</surname> <given-names>M</given-names></name></person-group>. <article-title>Personalized medicine: part 1: evolution and development into theranostics</article-title>. <source>PT.</source> (<year>2010</year>) <volume>35</volume>:<fpage>560</fpage>&#x02013;<lpage>76</lpage>. <pub-id pub-id-type="pmid">21037908</pub-id></citation></ref>
<ref id="B11">
<label>11.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cummings</surname> <given-names>JL</given-names></name> <name><surname>Morstorf</surname> <given-names>T</given-names></name> <name><surname>Zhong</surname> <given-names>K</given-names></name></person-group>. <article-title>Alzheimer&#x00027;s disease drug-development pipeline: few candidates, frequent failures</article-title>. <source>Alzheimers Res Ther.</source> (<year>2014</year>) <volume>6</volume>:<fpage>37</fpage>. <pub-id pub-id-type="doi">10.1186/alzrt269</pub-id><pub-id pub-id-type="pmid">25024750</pub-id></citation></ref>
<ref id="B12">
<label>12.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mehta</surname> <given-names>D</given-names></name> <name><surname>Jackson</surname> <given-names>R</given-names></name> <name><surname>Paul</surname> <given-names>G</given-names></name> <name><surname>Shi</surname> <given-names>J</given-names></name> <name><surname>Sabbagh</surname> <given-names>M</given-names></name></person-group>. <article-title>Why do trials for Alzheimer&#x00027;s disease drugs keep failing? A discontinued drug perspective for 2010-2015</article-title>. <source>Exp Opin Invest Drugs.</source> (<year>2017</year>) <volume>26</volume>:<fpage>735</fpage>&#x02013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1080/13543784.2017.1323868</pub-id><pub-id pub-id-type="pmid">28460541</pub-id></citation></ref>
<ref id="B13">
<label>13.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pimplikar</surname> <given-names>SW</given-names></name></person-group>. <article-title>Multi-omics and Alzheimer&#x00027;s disease: a slower but surer path to an efficacious therapy?</article-title> <source>Am J Physiol Cell Physiol.</source> (<year>2017</year>) <volume>313</volume>:<fpage>C1</fpage>&#x02013;<lpage>2</lpage>. <pub-id pub-id-type="doi">10.1152/ajpcell.00109.2017</pub-id><pub-id pub-id-type="pmid">28515086</pub-id></citation></ref>
<ref id="B14">
<label>14.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xicota</surname> <given-names>L</given-names></name> <name><surname>Ichou</surname> <given-names>F</given-names></name> <name><surname>Lejeune</surname> <given-names>F-X</given-names></name> <name><surname>Colsch</surname> <given-names>B</given-names></name> <name><surname>Tenenhaus</surname> <given-names>A</given-names></name> <etal/></person-group>. <article-title>Multi-omics signature of brain amyloid deposition in asymptomatic individuals at-risk for Alzheimer&#x00027;s disease: The INSIGHT-preAD study</article-title>. <source>EBio Med.</source> (<year>2019</year>) <volume>47</volume>:<fpage>518</fpage>&#x02013;<lpage>28</lpage>. <pub-id pub-id-type="doi">10.1016/j.ebiom.2019.08.051</pub-id><pub-id pub-id-type="pmid">31492558</pub-id></citation></ref>
<ref id="B15">
<label>15.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Saykin</surname> <given-names>AJ</given-names></name> <name><surname>Shen</surname> <given-names>L</given-names></name> <name><surname>Yao</surname> <given-names>X</given-names></name> <name><surname>Kim</surname> <given-names>S</given-names></name> <name><surname>Nho</surname> <given-names>K</given-names></name> <name><surname>Risacher</surname> <given-names>SL</given-names></name> <etal/></person-group>. <article-title>Genetic studies of quantitative MCI and AD phenotypes in ADNI: progress, opportunities, and plans</article-title>. <source>Alzheimers Dementia.</source> (<year>2015</year>) <volume>11</volume>:<fpage>792</fpage>&#x02013;<lpage>814</lpage>. <pub-id pub-id-type="doi">10.1016/j.jalz.2015.05.009</pub-id><pub-id pub-id-type="pmid">26194313</pub-id></citation></ref>
<ref id="B16">
<label>16.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Simino</surname> <given-names>J</given-names></name> <name><surname>Wang</surname> <given-names>Z</given-names></name> <name><surname>Bressler</surname> <given-names>J</given-names></name> <name><surname>Chouraki</surname> <given-names>V</given-names></name> <name><surname>Yang</surname> <given-names>Q</given-names></name> <name><surname>Younkin</surname> <given-names>SG</given-names></name> <etal/></person-group>. <article-title>Whole exome sequence-based association analyses of plasma amyloid-&#x003B2; in African and European Americans; the atherosclerosis risk in communities-neurocognitive study</article-title>. <source>PLoS ONE.</source> (<year>2017</year>) <volume>12</volume>:<fpage>e0180046</fpage>&#x02013;<lpage>0046</lpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0180046</pub-id><pub-id pub-id-type="pmid">28704393</pub-id></citation></ref>
<ref id="B17">
<label>17.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bis</surname> <given-names>JC</given-names></name> <name><surname>Jian</surname> <given-names>X</given-names></name> <name><surname>Kunkle</surname> <given-names>BW</given-names></name> <name><surname>Chen</surname> <given-names>Y</given-names></name> <name><surname>Hamilton-Nelson</surname> <given-names>KL</given-names></name> <name><surname>Bush</surname> <given-names>WS</given-names></name> <etal/></person-group>. <article-title>Whole exome sequencing study identifies novel rare and common Alzheimer&#x00027;s-Associated variants involved in immune response and transcriptional regulation</article-title>. <source>Mol Psychiatry.</source> (<year>2020</year>) <volume>25</volume>:<fpage>1859</fpage>&#x02013;<lpage>75</lpage>. <pub-id pub-id-type="doi">10.1038/s41380-018-0112-7</pub-id><pub-id pub-id-type="pmid">31636380</pub-id></citation></ref>
<ref id="B18">
<label>18.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Piras</surname> <given-names>IS</given-names></name> <name><surname>Krate</surname> <given-names>J</given-names></name> <name><surname>Schrauwen</surname> <given-names>I</given-names></name> <name><surname>Corneveaux</surname> <given-names>JJ</given-names></name> <name><surname>Serrano</surname> <given-names>GE</given-names></name> <name><surname>Sue</surname> <given-names>L</given-names></name> <etal/></person-group>. <article-title>Whole transcriptome profiling of the human hippocampus suggests an involvement of the KIBRA rs17070145 polymorphism in differential activation of the MAPK signaling pathway</article-title>. <source>Hippocampus.</source> (<year>2017</year>) <volume>27</volume>:<fpage>784</fpage>&#x02013;<lpage>93</lpage>. <pub-id pub-id-type="doi">10.1002/hipo.22731</pub-id><pub-id pub-id-type="pmid">28380666</pub-id></citation></ref>
<ref id="B19">
<label>19.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Luningham</surname> <given-names>JM</given-names></name> <name><surname>Chen</surname> <given-names>J</given-names></name> <name><surname>Tang</surname> <given-names>S</given-names></name> <name><surname>de Jager</surname> <given-names>PL</given-names></name> <name><surname>Bennett</surname> <given-names>DA</given-names></name> <name><surname>Buchman</surname> <given-names>AS</given-names></name> <etal/></person-group>. <article-title>Bayesian genome-wide TWAS method to leverage both cis- and trans-eQTL information through summary statistics</article-title>. <source>Am J Human Genet.</source> (<year>2020</year>) <volume>107</volume>:<fpage>714</fpage>&#x02013;<lpage>26</lpage>. <pub-id pub-id-type="doi">10.1016/j.ajhg.2020.08.022</pub-id><pub-id pub-id-type="pmid">32961112</pub-id></citation></ref>
<ref id="B20">
<label>20.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Johnson</surname> <given-names>KA</given-names></name> <name><surname>Fox</surname> <given-names>NC</given-names></name> <name><surname>Sperling</surname> <given-names>RA</given-names></name> <name><surname>Klunk</surname> <given-names>WE</given-names></name></person-group>. <article-title>Brain imaging in Alzheimer disease</article-title>. <source>Cold Spring Harbor Perspect Med.</source> (<year>2012</year>) <volume>2</volume>:<fpage>a006213</fpage>. <pub-id pub-id-type="doi">10.1101/cshperspect.a006213</pub-id><pub-id pub-id-type="pmid">22474610</pub-id></citation></ref>
<ref id="B21">
<label>21.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shen</surname> <given-names>L</given-names></name> <name><surname>Thompson</surname> <given-names>PM</given-names></name></person-group>. <article-title>Brain imaging genomics: integrated analysis and machine learning</article-title>. <source>Proc IEEE Inst Electr Electron Eng.</source> (<year>2020</year>) <volume>108</volume>:<fpage>125</fpage>&#x02013;<lpage>62</lpage>. <pub-id pub-id-type="doi">10.1109/JPROC.2019.2947272</pub-id><pub-id pub-id-type="pmid">31902950</pub-id></citation></ref>
<ref id="B22">
<label>22.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chauhan</surname> <given-names>G</given-names></name> <name><surname>Adams</surname> <given-names>HHH</given-names></name> <name><surname>Bis</surname> <given-names>JC</given-names></name> <name><surname>Weinstein</surname> <given-names>G</given-names></name> <name><surname>Yu</surname> <given-names>L</given-names></name> <name><surname>T&#x000F6;glhofer</surname> <given-names>AM</given-names></name> <etal/></person-group>. <article-title>Association of Alzheimer&#x00027;s disease GWAS loci with MRI markers of brain aging</article-title>. <source>Neurobiol Aging.</source> (<year>2015</year>) <volume>36</volume>:<fpage>1765.e7</fpage>&#x02013;<lpage>16</lpage>. <pub-id pub-id-type="doi">10.1016/j.neurobiolaging.2014.12.028</pub-id><pub-id pub-id-type="pmid">25670335</pub-id></citation></ref>
<ref id="B23">
<label>23.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>J-Q</given-names></name> <name><surname>Wang</surname> <given-names>H-F</given-names></name> <name><surname>Zhu</surname> <given-names>X-C</given-names></name> <name><surname>Sun</surname> <given-names>F-R</given-names></name> <name><surname>Tan</surname> <given-names>M-S</given-names></name> <name><surname>Tan</surname> <given-names>C-C</given-names></name> <etal/></person-group>. <article-title>GWAS-linked loci and neuroimaging measures in Alzheimer&#x00027;s disease</article-title>. <source>Mol Neurobiol.</source> (<year>2017</year>) <volume>54</volume>:<fpage>146</fpage>&#x02013;<lpage>53</lpage>. <pub-id pub-id-type="doi">10.1007/s12035-015-9669-1</pub-id><pub-id pub-id-type="pmid">26732597</pub-id></citation></ref>
<ref id="B24">
<label>24.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Grasby</surname> <given-names>KL</given-names></name> <name><surname>Jahanshad</surname> <given-names>N</given-names></name> <name><surname>Painter</surname> <given-names>JN</given-names></name> <name><surname>Colodro-Conde</surname> <given-names>L</given-names></name> <name><surname>Bralten</surname> <given-names>J</given-names></name> <name><surname>Hibar</surname> <given-names>DP</given-names></name> <etal/></person-group>. <article-title>The genetic architecture of the human cerebral cortex</article-title>. <source>Science.</source> (<year>2021</year>) <volume>367</volume>:<fpage>eaay6690</fpage>. <pub-id pub-id-type="doi">10.1126/science.aay6690</pub-id><pub-id pub-id-type="pmid">34672763</pub-id></citation></ref>
<ref id="B25">
<label>25.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ritchie</surname> <given-names>J</given-names></name> <name><surname>Pantazatos</surname> <given-names>SP</given-names></name> <name><surname>French</surname> <given-names>L</given-names></name></person-group>. <article-title>Transcriptomic characterization of MRI contrast with focus on the T1-w/T2-w ratio in the cerebral cortex</article-title>. <source>NeuroImage.</source> (<year>2018</year>) <volume>174</volume>:<fpage>504</fpage>&#x02013;<lpage>17</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuroimage.2018.03.027</pub-id><pub-id pub-id-type="pmid">29567503</pub-id></citation></ref>
<ref id="B26">
<label>26.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Albert</surname> <given-names>FW</given-names></name> <name><surname>Kruglyak</surname> <given-names>L</given-names></name></person-group>. <article-title>The role of regulatory variation in complex traits and disease</article-title>. <source>Nat Rev Genet.</source> (<year>2015</year>) <volume>16</volume>:<fpage>197</fpage>&#x02013;<lpage>212</lpage>. <pub-id pub-id-type="doi">10.1038/nrg3891</pub-id><pub-id pub-id-type="pmid">25707927</pub-id></citation></ref>
<ref id="B27">
<label>27.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Stein</surname> <given-names>JL</given-names></name> <name><surname>Hua</surname> <given-names>X</given-names></name> <name><surname>Lee</surname> <given-names>S</given-names></name> <name><surname>Ho</surname> <given-names>AJ</given-names></name> <name><surname>Leow</surname> <given-names>AD</given-names></name> <name><surname>Toga</surname> <given-names>AW</given-names></name> <etal/></person-group>. <article-title>Voxelwise genome-wide association study (vGWAS)</article-title>. <source>NeuroImage.</source> (<year>2010</year>) <volume>53</volume>:<fpage>1160</fpage>&#x02013;<lpage>74</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuroimage.2010.02.032</pub-id><pub-id pub-id-type="pmid">20171287</pub-id></citation></ref>
<ref id="B28">
<label>28.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>A</given-names></name> <name><surname>Zhao</surname> <given-names>Q</given-names></name> <name><surname>Xu</surname> <given-names>D</given-names></name> <name><surname>Jiang</surname> <given-names>S</given-names></name></person-group>. <article-title>Brain APOE expression quantitative trait loci-based association study identified one susceptibility locus for Alzheimer&#x00027;s disease by interacting with APOE &#x003B5;4</article-title>. <source>Sci Rep.</source> (<year>2018</year>) <volume>8</volume>:<fpage>8068</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-018-26398-1</pub-id><pub-id pub-id-type="pmid">29795290</pub-id></citation></ref>
<ref id="B29">
<label>29.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>K</given-names></name> <name><surname>Yao</surname> <given-names>X</given-names></name> <name><surname>Yan</surname> <given-names>J</given-names></name> <name><surname>Chasioti</surname> <given-names>D</given-names></name> <name><surname>Risacher</surname> <given-names>S</given-names></name> <name><surname>Nho</surname> <given-names>K</given-names></name> <etal/></person-group>. <article-title>Transcriptome-guided imaging genetic analysis via a novel sparse CCA algorithm</article-title>. <source>Graphs Biomed Image Anal Comput Anat Imaging Genet</source>. (<year>2017</year>) <volume>10551</volume>:<fpage>220</fpage>&#x02013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-319-67675-3_20</pub-id><pub-id pub-id-type="pmid">30294724</pub-id></citation></ref>
<ref id="B30">
<label>30.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hampel</surname> <given-names>H</given-names></name> <name><surname>Vergallo</surname> <given-names>A</given-names></name> <name><surname>Perry</surname> <given-names>G</given-names></name> <name><surname>Lista</surname> <given-names>S</given-names></name></person-group>. <article-title>The Alzheimer precision medicine initiative</article-title>. <source>J Alzheimers Dis.</source> (<year>2019</year>) <volume>68</volume>:<fpage>1</fpage>&#x02013;<lpage>24</lpage>. <pub-id pub-id-type="doi">10.3233/JAD-181121</pub-id><pub-id pub-id-type="pmid">31801830</pub-id></citation></ref>
<ref id="B31">
<label>31.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Thompson</surname> <given-names>PM</given-names></name> <name><surname>Jahanshad</surname> <given-names>N</given-names></name> <name><surname>Ching</surname> <given-names>CRK</given-names></name> <name><surname>Salminen</surname> <given-names>LE</given-names></name> <name><surname>Thomopoulos</surname> <given-names>SI</given-names></name> <name><surname>Bright</surname> <given-names>J</given-names></name> <etal/></person-group>. <article-title>ENIGMA and global neuroscience: a decade of large-scale studies of the brain in health and disease across more than 40 countries</article-title>. <source>Transl Psychiatry.</source> (<year>2020</year>) <volume>10</volume>:<fpage>100</fpage>. <pub-id pub-id-type="doi">10.1038/s41398-020-0705-1</pub-id><pub-id pub-id-type="pmid">32198361</pub-id></citation></ref>
<ref id="B32">
<label>32.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hibar</surname> <given-names>DP</given-names></name> <name><surname>Stein</surname> <given-names>JL</given-names></name> <name><surname>Renteria</surname> <given-names>ME</given-names></name> <name><surname>Arias-Vasquez</surname> <given-names>A</given-names></name> <name><surname>Desrivi&#x000E8;res</surname> <given-names>S</given-names></name> <name><surname>Jahanshad</surname> <given-names>N</given-names></name> <etal/></person-group>. <article-title>Common genetic variants influence human subcortical brain structures</article-title>. <source>Nature.</source> (<year>2015</year>) <volume>520</volume>:<fpage>224</fpage>&#x02013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1038/nature14101</pub-id><pub-id pub-id-type="pmid">25607358</pub-id></citation></ref>
<ref id="B33">
<label>33.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Satizabal</surname> <given-names>CL</given-names></name> <name><surname>Adams</surname> <given-names>HHH</given-names></name> <name><surname>Hibar</surname> <given-names>DP</given-names></name> <name><surname>White</surname> <given-names>CC</given-names></name> <name><surname>Knol</surname> <given-names>MJ</given-names></name> <name><surname>Stein</surname> <given-names>JL</given-names></name> <etal/></person-group>. <article-title>Genetic architecture of subcortical brain structures in 38,851 individuals</article-title>. <source>Nat Genet.</source> (<year>2019</year>) <volume>51</volume>:<fpage>1624</fpage>&#x02013;<lpage>36</lpage>. <pub-id pub-id-type="doi">10.1038/s41588-019-0511-y</pub-id><pub-id pub-id-type="pmid">31636452</pub-id></citation></ref>
<ref id="B34">
<label>34.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhao</surname> <given-names>B</given-names></name> <name><surname>Li</surname> <given-names>T</given-names></name> <name><surname>Yang</surname> <given-names>Y</given-names></name> <name><surname>Wang</surname> <given-names>X</given-names></name> <name><surname>Luo</surname> <given-names>T</given-names></name> <name><surname>Shan</surname> <given-names>Y</given-names></name> <etal/></person-group>. <article-title>Common genetic variation influencing human white matter microstructure</article-title>. <source>Science.</source> (<year>2021</year>) <volume>372</volume>:<fpage>eabf3736</fpage>. <pub-id pub-id-type="doi">10.1126/science.abf3736</pub-id><pub-id pub-id-type="pmid">34140357</pub-id></citation></ref>
<ref id="B35">
<label>35.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Smit</surname> <given-names>DJA</given-names></name> <name><surname>Wright</surname> <given-names>MJ</given-names></name> <name><surname>Meyers</surname> <given-names>JL</given-names></name> <name><surname>Martin</surname> <given-names>NG</given-names></name> <name><surname>Ho</surname> <given-names>YYW</given-names></name> <name><surname>Malone</surname> <given-names>SM</given-names></name> <etal/></person-group>. <article-title>Genome-wide association analysis links multiple psychiatric liability genes to oscillatory brain activity</article-title>. <source>Human Brain Mapping.</source> (<year>2018</year>) <volume>39</volume>:<fpage>4183</fpage>&#x02013;<lpage>95</lpage>. <pub-id pub-id-type="doi">10.1002/hbm.24238</pub-id><pub-id pub-id-type="pmid">29947131</pub-id></citation></ref>
<ref id="B36">
<label>36.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chow</surname> <given-names>GC</given-names></name></person-group>. <article-title>Tests of equality between sets of coefficients in two linear regressions</article-title>. <source>Econometrica.</source> (<year>1960</year>) <volume>28</volume>:<fpage>591</fpage>&#x02013;<lpage>605</lpage>. <pub-id pub-id-type="doi">10.2307/1910133</pub-id></citation>
</ref>
<ref id="B37">
<label>37.</label>
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Marinescu</surname> <given-names>RV</given-names></name> <name><surname>Oxtoby</surname> <given-names>NP</given-names></name> <name><surname>Young</surname> <given-names>AL</given-names></name> <name><surname>Bron</surname> <given-names>EE</given-names></name> <name><surname>Toga</surname> <given-names>AW</given-names></name></person-group>. <article-title>The Alzheimer&#x00027;s Disease Prediction Of Longitudinal Evolution (TADPOLE) Challenge: Results after 1 Year Follow-up</article-title>. (<year>2020</year>). Available online at: <ext-link ext-link-type="uri" xlink:href="http://arxiv.org/abs/2002.03419">http://arxiv.org/abs/2002.03419</ext-link> (accessed December 26, 2021).</citation>
</ref>
<ref id="B38">
<label>38.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Purcell</surname> <given-names>S</given-names></name> <name><surname>Neale</surname> <given-names>B</given-names></name> <name><surname>Todd-Brown</surname> <given-names>K</given-names></name> <name><surname>Thomas</surname> <given-names>L</given-names></name> <name><surname>Ferreira</surname> <given-names>MAR</given-names></name> <name><surname>Bender</surname> <given-names>D</given-names></name> <etal/></person-group>. <article-title>PLINK: a tool set for whole-genome association and population-based linkage analyses</article-title>. <source>Am J Human Genet.</source> (<year>2007</year>) <volume>81</volume>:<fpage>559</fpage>&#x02013;<lpage>75</lpage>. <pub-id pub-id-type="doi">10.1086/519795</pub-id><pub-id pub-id-type="pmid">17701901</pub-id></citation></ref>
<ref id="B39">
<label>39.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yip</surname> <given-names>SH</given-names></name> <name><surname>Wang</surname> <given-names>P</given-names></name> <name><surname>Kocher</surname> <given-names>J-PA</given-names></name> <name><surname>Sham</surname> <given-names>PC</given-names></name> <name><surname>Wang</surname> <given-names>J</given-names></name></person-group>. <article-title>Linnorm: improved statistical analysis for single cell RNA-seq expression data</article-title>. <source>Nucleic Acids Res.</source> (<year>2017</year>) <volume>45</volume>:<fpage>e179</fpage>. <pub-id pub-id-type="doi">10.1093/nar/gkx1189</pub-id><pub-id pub-id-type="pmid">29161425</pub-id></citation></ref>
<ref id="B40">
<label>40.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>D</given-names></name> <name><surname>Yi</surname> <given-names>X</given-names></name> <name><surname>Zhang</surname> <given-names>S</given-names></name> <name><surname>Zheng</surname> <given-names>Z</given-names></name> <name><surname>Wang</surname> <given-names>P</given-names></name> <name><surname>Xuan</surname> <given-names>C</given-names></name> <etal/></person-group>. <article-title>GWAS4D: multidimensional analysis of context-specific regulatory variant for human complex diseases and traits</article-title>. <source>Nucleic Acids Res.</source> (<year>2018</year>) <volume>46</volume>:<fpage>W114</fpage>&#x02013;<lpage>20</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gky407</pub-id><pub-id pub-id-type="pmid">29771388</pub-id></citation></ref>
<ref id="B41">
<label>41.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yip</surname> <given-names>SH</given-names></name> <name><surname>Sham</surname> <given-names>PC</given-names></name> <name><surname>Wang</surname> <given-names>J</given-names></name></person-group>. <article-title>Evaluation of tools for highly variable gene discovery from single-cell RNA-seq data</article-title>. <source>Brief Bioinform.</source> (<year>2019</year>) <volume>20</volume>:<fpage>1583</fpage>&#x02013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1093/bib/bby011</pub-id><pub-id pub-id-type="pmid">29481632</pub-id></citation></ref>
<ref id="B42">
<label>42.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Satija</surname> <given-names>R</given-names></name> <name><surname>Farrell</surname> <given-names>JA</given-names></name> <name><surname>Gennert</surname> <given-names>D</given-names></name> <name><surname>Schier</surname> <given-names>AF</given-names></name> <name><surname>Regev</surname> <given-names>A</given-names></name></person-group>. <article-title>Spatial reconstruction of single-cell gene expression data</article-title>. <source>Nat Biotechnol.</source> (<year>2015</year>) <volume>33</volume>:<fpage>495</fpage>&#x02013;<lpage>502</lpage>. <pub-id pub-id-type="doi">10.1038/nbt.3192</pub-id><pub-id pub-id-type="pmid">25867923</pub-id></citation></ref>
<ref id="B43">
<label>43.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Love</surname> <given-names>MI</given-names></name> <name><surname>Huber</surname> <given-names>W</given-names></name> <name><surname>Anders</surname> <given-names>S</given-names></name></person-group>. <article-title>Moderated estimation of fold change and dispersion for RNA-seq data with DESeq2</article-title>. <source>Genome Biol.</source> (<year>2014</year>) <volume>15</volume>:<fpage>550</fpage>. <pub-id pub-id-type="doi">10.1186/s13059-014-0550-8</pub-id><pub-id pub-id-type="pmid">25516281</pub-id></citation></ref>
<ref id="B44">
<label>44.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fischl</surname> <given-names>B</given-names></name> <name><surname>Sereno</surname> <given-names>MI</given-names></name> <name><surname>Dale</surname> <given-names>AM</given-names></name></person-group>. <article-title>Cortical surface-based analysis: ii: inflation, flattening, and a surface-based coordinate system</article-title>. <source>NeuroImage.</source> (<year>1999</year>) <volume>9</volume>:<fpage>195</fpage>&#x02013;<lpage>207</lpage>. <pub-id pub-id-type="doi">10.1006/nimg.1998.0396</pub-id><pub-id pub-id-type="pmid">9931269</pub-id></citation></ref>
<ref id="B45">
<label>45.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Lee</surname> <given-names>B</given-names></name> <name><surname>Wang</surname> <given-names>J</given-names></name> <name><surname>Shen</surname> <given-names>L</given-names></name></person-group>. <article-title>Identifying precision AD biomarkers with varying prognosis effects in genetics driven subpopulations</article-title>. <source>AAIC&#x00027;21: Alzheimer&#x00027;s Association Int. Conf. on Alzheimer&#x00027;s Disease, Denver</source> <publisher-loc>San Diego</publisher-loc> (<year>2021</year>).</citation>
</ref>
<ref id="B46">
<label>46.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Barbur</surname> <given-names>VA</given-names></name> <name><surname>Montgomery</surname> <given-names>DC</given-names></name> <name><surname>Peck</surname> <given-names>EA</given-names></name></person-group>. <article-title>Introduction to linear regression analysis</article-title>. <source>Statistician.</source> (<year>1994</year>) <volume>43</volume>:<fpage>339</fpage>&#x02013;<lpage>41</lpage>. <pub-id pub-id-type="doi">10.2307/2348362</pub-id></citation>
</ref>
<ref id="B47">
<label>47.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Rawlings</surname> <given-names>JO</given-names></name> <name><surname>Pantula</surname> <given-names>SG</given-names></name> <name><surname>Dickey</surname> <given-names>DA</given-names></name></person-group>. <source>Applied Regression Analysis: A Research Tool.</source> <edition>2nd ed.</edition> <publisher-loc>New York, NY</publisher-loc>: <publisher-name>Springer</publisher-name> (<year>1998</year>).</citation>
</ref>
<ref id="B48">
<label>48.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hoerl</surname> <given-names>AE</given-names></name> <name><surname>Kennard</surname> <given-names>RW</given-names></name></person-group>. <article-title>Ridge regression: biased estimation for nonorthogonal problems</article-title>. <source>Technometrics.</source> (<year>1970</year>) <volume>12</volume>:<fpage>55</fpage>&#x02013;<lpage>67</lpage>. <pub-id pub-id-type="doi">10.1080/00401706.1970.10488634</pub-id></citation>
</ref>
<ref id="B49">
<label>49.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Berkopec</surname> <given-names>A</given-names></name></person-group>. <article-title>HyperQuick algorithm for discrete hypergeometric distribution</article-title>. <source>J Discrete Algorithms.</source> (<year>2007</year>) <volume>5</volume>:<fpage>341</fpage>&#x02013;<lpage>7</lpage>. <pub-id pub-id-type="doi">10.1016/j.jda.2006.01.001</pub-id></citation>
</ref>
<ref id="B50">
<label>50.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Benjamini</surname> <given-names>Y</given-names></name> <name><surname>Hochberg</surname> <given-names>Y</given-names></name></person-group>. <article-title>Controlling the false discovery rate: a practical and powerful approach to multiple testing</article-title>. <source>J Royal Stat Soc Series B.</source> (<year>1995</year>) <volume>57</volume>:<fpage>289</fpage>&#x02013;<lpage>300</lpage>. <pub-id pub-id-type="doi">10.1111/j.2517-6161.1995.tb02031.x</pub-id></citation>
</ref>
<ref id="B51">
<label>51.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shabalin</surname> <given-names>AA</given-names></name></person-group>. <article-title>Matrix eQTL: ultra fast eQTL analysis via large matrix operations</article-title>. <source>Bioinformatics.</source> (<year>2012</year>) <volume>28</volume>:<fpage>1353</fpage>&#x02013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bts163</pub-id><pub-id pub-id-type="pmid">22492648</pub-id></citation></ref>
<ref id="B52">
<label>52.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dogru-Abbasoglu</surname> <given-names>S</given-names></name> <name><surname>Ayka&#x000E7;-Toker</surname> <given-names>G</given-names></name> <name><surname>Hanagasi</surname> <given-names>HA</given-names></name> <name><surname>G&#x000FC;rvit</surname> <given-names>H</given-names></name> <name><surname>Emre</surname> <given-names>M</given-names></name> <name><surname>Uysal</surname> <given-names>M</given-names></name></person-group>. <article-title>The Arg194Trp polymorphism in DNA repair gene XRCC1 and the risk for sporadic late-onset Alzheimer&#x00027;s disease</article-title>. <source>Neurol Sci.</source> (<year>2007</year>) <volume>28</volume>:<fpage>31</fpage>&#x02013;<lpage>4</lpage>. <pub-id pub-id-type="doi">10.1007/s10072-007-0744-x</pub-id><pub-id pub-id-type="pmid">17385092</pub-id></citation></ref>
<ref id="B53">
<label>53.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wingo</surname> <given-names>TS</given-names></name> <name><surname>Cutler</surname> <given-names>DJ</given-names></name> <name><surname>Wingo</surname> <given-names>AP</given-names></name> <name><surname>Le</surname> <given-names>N-A</given-names></name> <name><surname>Rabinovici</surname> <given-names>GD</given-names></name> <etal/></person-group>. <article-title>Association of early-onset alzheimer disease with elevated low-density lipoprotein cholesterol levels and rare genetic coding variants of APOB</article-title>. <source>JAMA Neurol.</source> (<year>2019</year>) <volume>76</volume>:<fpage>809</fpage>&#x02013;<lpage>17</lpage>. <pub-id pub-id-type="doi">10.1001/jamaneurol.2019.0648</pub-id><pub-id pub-id-type="pmid">31135820</pub-id></citation></ref>
<ref id="B54">
<label>54.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rockman</surname> <given-names>MV</given-names></name> <name><surname>Kruglyak</surname> <given-names>L</given-names></name></person-group>. <article-title>Genetics of global gene expression</article-title>. <source>Nat Rev Genet.</source> (<year>2006</year>) <volume>7</volume>:<fpage>862</fpage>&#x02013;<lpage>72</lpage>. <pub-id pub-id-type="doi">10.1038/nrg1964</pub-id><pub-id pub-id-type="pmid">17047685</pub-id></citation></ref>
<ref id="B55">
<label>55.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nica</surname> <given-names>AC</given-names></name> <name><surname>Dermitzakis</surname> <given-names>ET</given-names></name></person-group>. <article-title>Expression quantitative trait loci: present and future. Philosophical transactions of the royal society of london. Series B</article-title>. <source>Biol Sci</source>. (<year>2013</year>) <volume>368</volume>:<fpage>20120362</fpage>. <pub-id pub-id-type="doi">10.1098/rstb.2012.0362</pub-id><pub-id pub-id-type="pmid">23650636</pub-id></citation></ref>
<ref id="B56">
<label>56.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Servin</surname> <given-names>B</given-names></name> <name><surname>Stephens</surname> <given-names>M</given-names></name></person-group>. <article-title>Imputation-based analysis of association studies: candidate regions and quantitative traits</article-title>. <source>PLoS Genet.</source> (<year>2007</year>) <volume>3</volume>:<fpage>e114</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pgen.0030114</pub-id><pub-id pub-id-type="pmid">17676998</pub-id></citation></ref>
<ref id="B57">
<label>57.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sheller</surname> <given-names>MJ</given-names></name> <name><surname>Edwards</surname> <given-names>B</given-names></name> <name><surname>Reina</surname> <given-names>GA</given-names></name> <name><surname>Martin</surname> <given-names>J</given-names></name> <name><surname>Pati</surname> <given-names>S</given-names></name> <name><surname>Kotrotsou</surname> <given-names>A</given-names></name> <etal/></person-group>. <article-title>Federated learning in medicine: facilitating multi-institutional collaborations without sharing patient data</article-title>. <source>Sci Rep.</source> (<year>2020</year>) <volume>10</volume>:<fpage>12598</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-020-69250-1</pub-id><pub-id pub-id-type="pmid">32724046</pub-id></citation></ref>
<ref id="B58">
<label>58.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ng</surname> <given-names>D</given-names></name> <name><surname>Lan</surname> <given-names>X</given-names></name> <name><surname>Yao</surname> <given-names>MM-S</given-names></name> <name><surname>Chan</surname> <given-names>WP</given-names></name> <name><surname>Feng</surname> <given-names>M</given-names></name></person-group>. <article-title>Federated learning: a collaborative effort to achieve better medical imaging models for individual sites that have small labelled datasets</article-title>. <source>Quant Imaging Med Surgery.</source> (<year>2021</year>) <volume>11</volume>:<fpage>852</fpage>&#x02013;<lpage>7</lpage>. <pub-id pub-id-type="doi">10.21037/qims-20-595</pub-id><pub-id pub-id-type="pmid">33532283</pub-id></citation></ref>
<ref id="B59">
<label>59.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dong</surname> <given-names>Q</given-names></name> <name><surname>Zhang</surname> <given-names>W</given-names></name> <name><surname>Wu</surname> <given-names>J</given-names></name> <name><surname>Li</surname> <given-names>B</given-names></name> <name><surname>Schron</surname> <given-names>EH</given-names></name> <name><surname>McMahon</surname> <given-names>T</given-names></name> <etal/></person-group>. <article-title>Applying surface-based hippocampal morphometry to study APOE-E4 allele dose effects in cognitively unimpaired subjects</article-title>. <source>NeuroImage Clin.</source> (<year>2019</year>) <volume>22</volume>:<fpage>101744</fpage>. <pub-id pub-id-type="doi">10.1016/j.nicl.2019.101744</pub-id><pub-id pub-id-type="pmid">30852398</pub-id></citation></ref>
<ref id="B60">
<label>60.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yan</surname> <given-names>J</given-names></name> <name><surname>Raja</surname> <given-names>VV</given-names></name> <name><surname>Huang</surname> <given-names>Z</given-names></name> <name><surname>Amico</surname> <given-names>E</given-names></name> <name><surname>Nho</surname> <given-names>K</given-names></name> <etal/></person-group>. <article-title>Brain-wide structural connectivity alterations under the control of Alzheimer risk genes</article-title>. <source>Int J Comput Biol Drug Design.</source> (<year>2020</year>) <volume>13</volume>:<fpage>58</fpage>&#x02013;<lpage>70</lpage>. <pub-id pub-id-type="doi">10.1504/IJCBDD.2020.105098</pub-id><pub-id pub-id-type="pmid">32095160</pub-id></citation></ref>
<ref id="B61">
<label>61.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schork</surname> <given-names>NJ</given-names></name></person-group>. <article-title>Personalized medicine: time for one-person trials</article-title>. <source>Nature.</source> (<year>2015</year>) <volume>520</volume>:<fpage>609</fpage>&#x02013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1038/520609a</pub-id><pub-id pub-id-type="pmid">25925459</pub-id></citation></ref>
<ref id="B62">
<label>62.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Caselli</surname> <given-names>RJ</given-names></name> <name><surname>Dueck</surname> <given-names>AC</given-names></name> <name><surname>Osborne</surname> <given-names>D</given-names></name> <name><surname>Sabbagh</surname> <given-names>MN</given-names></name> <name><surname>Connor</surname> <given-names>DJ</given-names></name> <name><surname>Ahern</surname> <given-names>GL</given-names></name> <etal/></person-group>. <article-title>Longitudinal modeling of age-related memory decline and the APOE &#x003B5;4 effect</article-title>. <source>N Engl J Med.</source> (<year>2009</year>) <volume>361</volume>:<fpage>255</fpage>&#x02013;<lpage>63</lpage>. <pub-id pub-id-type="doi">10.1056/NEJMoa0809437</pub-id><pub-id pub-id-type="pmid">19605830</pub-id></citation></ref>
<ref id="B63">
<label>63.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cox</surname> <given-names>N</given-names></name></person-group>. <article-title>UK biobank shares the promise of big data</article-title>. <source>Nature.</source> (<year>2018</year>) <volume>562</volume>:<fpage>194</fpage>&#x02013;<lpage>5</lpage>. <pub-id pub-id-type="doi">10.1038/d41586-018-06948-3</pub-id><pub-id pub-id-type="pmid">30305754</pub-id></citation></ref>
<ref id="B64">
<label>64.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bycroft</surname> <given-names>C</given-names></name> <name><surname>Freeman</surname> <given-names>C</given-names></name> <name><surname>Petkova</surname> <given-names>D</given-names></name> <name><surname>Band</surname> <given-names>G</given-names></name> <name><surname>Elliott</surname> <given-names>LT</given-names></name> <name><surname>Sharp</surname> <given-names>K</given-names></name> <etal/></person-group>. <article-title>The UK biobank resource with deep phenotyping and genomic data</article-title>. <source>Nature.</source> (<year>2018</year>) <volume>562</volume>:<fpage>203</fpage>&#x02013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-018-0579-z</pub-id><pub-id pub-id-type="pmid">30305743</pub-id></citation></ref>
<ref id="B65">
<label>65.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Elliott</surname> <given-names>LT</given-names></name> <name><surname>Sharp</surname> <given-names>K</given-names></name> <name><surname>Alfaro-Almagro</surname> <given-names>F</given-names></name> <name><surname>Shi</surname> <given-names>S</given-names></name> <name><surname>Miller</surname> <given-names>KL</given-names></name> <name><surname>Douaud</surname> <given-names>G</given-names></name> <etal/></person-group>. <article-title>Genome-wide association studies of brain imaging phenotypes in UK biobank</article-title>. <source>Nature.</source> (<year>2018</year>) <volume>562</volume>:<fpage>210</fpage>&#x02013;<lpage>6</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-018-0571-7</pub-id><pub-id pub-id-type="pmid">33875891</pub-id></citation></ref>
<ref id="B66">
<label>66.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Y</given-names></name> <name><surname>Zhang</surname> <given-names>J</given-names></name> <name><surname>Gutman</surname> <given-names>B</given-names></name> <name><surname>Chan</surname> <given-names>TF</given-names></name> <name><surname>Becker</surname> <given-names>JT</given-names></name> <name><surname>Aizenstein</surname> <given-names>HJ</given-names></name> <etal/></person-group>. <article-title>Multivariate tensor-based morphometry on surfaces: application to mapping ventricular abnormalities in HIV/AIDS</article-title>. <source>NeuroImage.</source> (<year>2010</year>) <volume>49</volume>:<fpage>2141</fpage>&#x02013;<lpage>57</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuroimage.2009.10.086</pub-id><pub-id pub-id-type="pmid">19900560</pub-id></citation></ref>
<ref id="B67">
<label>67.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>J</given-names></name> <name><surname>Li</surname> <given-names>Q</given-names></name> <name><surname>Caselli</surname> <given-names>RJ</given-names></name> <name><surname>Thompson</surname> <given-names>PM</given-names></name> <name><surname>Ye</surname> <given-names>J</given-names></name> <name><surname>Wang</surname> <given-names>Y</given-names></name></person-group>. <article-title>Multi-source multi-target dictionary learning for prediction of cognitive decline</article-title>. <source>Inf Process Med Imaging.</source> (<year>2017</year>) <volume>10265</volume>:<fpage>184</fpage>&#x02013;<lpage>97</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-319-59050-9_15</pub-id><pub-id pub-id-type="pmid">28943731</pub-id></citation></ref>
<ref id="B68">
<label>68.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>J</given-names></name> <name><surname>Zhang</surname> <given-names>J</given-names></name> <name><surname>Shi</surname> <given-names>J</given-names></name> <name><surname>Chen</surname> <given-names>K</given-names></name> <name><surname>Caselli</surname> <given-names>RJ</given-names></name> <name><surname>Reiman</surname> <given-names>EM</given-names></name> <etal/></person-group>. <article-title>Hippocampus morphometry study on pathology-confirmed alzheimer&#x00027;s disease patients with surface multivariate morphometry statistics</article-title>. <source>Proc IEEE Int Symp BiomedImaging.</source> (<year>2018</year>) <volume>2018</volume>:<fpage>1555</fpage>&#x02013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1109/ISBI.2018.8363870</pub-id><pub-id pub-id-type="pmid">30123414</pub-id></citation></ref>
<ref id="B69">
<label>69.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>G</given-names></name> <name><surname>Dong</surname> <given-names>Q</given-names></name> <name><surname>Wu</surname> <given-names>J</given-names></name> <name><surname>Su</surname> <given-names>Y</given-names></name> <name><surname>Chen</surname> <given-names>K</given-names></name> <name><surname>Su</surname> <given-names>Q</given-names></name> <etal/></person-group>. <article-title>Developing univariate neurodegeneration biomarkers with low-rank and sparse subspace decomposition</article-title>. <source>Med Image Anal</source>. (<year>2020</year>) <volume>67</volume>:<fpage>1361</fpage>&#x02013;<lpage>8415</lpage>. <pub-id pub-id-type="doi">10.1016/j.media.2020.101877</pub-id><pub-id pub-id-type="pmid">33166772</pub-id></citation></ref>
<ref id="B70">
<label>70.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>J</given-names></name> <name><surname>Dong</surname> <given-names>Q</given-names></name> <name><surname>Zhang</surname> <given-names>J</given-names></name> <name><surname>Su</surname> <given-names>Y</given-names></name> <name><surname>Wu</surname> <given-names>T</given-names></name> <name><surname>Caselli</surname> <given-names>RJ</given-names></name> <etal/></person-group>. <article-title>Federated morphometry feature selection for hippocampal morphometry associated beta-amyloid and tau pathology</article-title>. <source>Front Neurosci.</source> (<year>2021</year>) <volume>15</volume>:<fpage>1585</fpage>. <pub-id pub-id-type="doi">10.3389/fnins.2021.762458</pub-id><pub-id pub-id-type="pmid">34899166</pub-id></citation></ref>
<ref id="B71">
<label>71.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Fury</surname> <given-names>W</given-names></name> <name><surname>Batliwalla</surname> <given-names>F</given-names></name> <name><surname>Gregersen</surname> <given-names>PK</given-names></name> <name><surname>Li</surname> <given-names>W</given-names></name></person-group>. <article-title>Overlapping probabilities of top ranking gene lists, hypergeometric distribution, and stringency of gene selection criterion</article-title>. In: <source>2006 International Conference of the IEEE Engineering in Medicine and Biology Society</source>. <publisher-loc>New York</publisher-loc> (<year>2006</year>). p. <fpage>5531</fpage>&#x02013;<lpage>4</lpage>. <pub-id pub-id-type="pmid">17947148</pub-id></citation></ref>
<ref id="B72">
<label>72.</label>
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Falcon</surname> <given-names>S</given-names></name> <name><surname>Gentleman</surname> <given-names>R</given-names></name></person-group>. <article-title>Hypergeometric testing used for gene set enrichment analysis</article-title>. In: <person-group person-group-type="editor"><name><surname>Hahne</surname> <given-names>F</given-names></name> <name><surname>Huber</surname> <given-names>W</given-names></name> <name><surname>Gentleman</surname> <given-names>R</given-names></name> <name><surname>Falcon</surname> <given-names>S</given-names></name></person-group>, editors. <source>Bioconductor Case Studies.</source> <publisher-loc>New York, NY</publisher-loc>: <publisher-name>Springer New York</publisher-name> (<year>2008</year>). p. <fpage>207</fpage>&#x02013;<lpage>20</lpage>.</citation>
</ref>
<ref id="B73">
<label>73.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Plaisier</surname> <given-names>SB</given-names></name> <name><surname>Taschereau</surname> <given-names>R</given-names></name> <name><surname>Wong</surname> <given-names>JA</given-names></name> <name><surname>Graeber</surname> <given-names>TG</given-names></name></person-group>. <article-title>Rank-rank hypergeometric overlap: identification of statistically significant overlap between gene-expression signatures</article-title>. <source>Nucleic Acids Res.</source> (<year>2010</year>) <volume>38</volume>:<fpage>e169</fpage>&#x02013;<lpage>69</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gkq636</pub-id><pub-id pub-id-type="pmid">20660011</pub-id></citation></ref>
<ref id="B74">
<label>74.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bateman</surname> <given-names>RJ</given-names></name> <name><surname>Blennow</surname> <given-names>K</given-names></name> <name><surname>Doody</surname> <given-names>R</given-names></name> <name><surname>Hendrix</surname> <given-names>S</given-names></name> <name><surname>Lovestone</surname> <given-names>S</given-names></name> <name><surname>Salloway</surname> <given-names>S</given-names></name> <etal/></person-group>. <article-title>Plasma biomarkers of AD emerging as essential tools for drug development: an EU/US CTAD task force report</article-title>. <source>J Prev Alzheimers Dis.</source> (<year>2019</year>) <volume>6</volume>:<fpage>169</fpage>&#x02013;<lpage>73</lpage>. <pub-id pub-id-type="doi">10.14283/jpad.2019.21</pub-id><pub-id pub-id-type="pmid">31062827</pub-id></citation></ref>
<ref id="B75">
<label>75.</label>
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Janelidze</surname> <given-names>S</given-names></name> <name><surname>Mattsson</surname> <given-names>N</given-names></name> <name><surname>Palmqvist</surname> <given-names>S</given-names></name> <name><surname>Smith</surname> <given-names>R</given-names></name> <name><surname>Beach</surname> <given-names>TG</given-names></name> <name><surname>Serrano</surname> <given-names>GE</given-names></name> <etal/></person-group>. <article-title>Plasma P-tau181 in Alzheimer&#x00027;s disease: relationship to other biomarkers, differential diagnosis, neuropathology and longitudinal progression to Alzheimer&#x00027;s dementia</article-title>. <source>Nat Med.</source> (<year>2020</year>) <volume>26</volume>:<fpage>379</fpage>&#x02013;<lpage>86</lpage>. <pub-id pub-id-type="doi">10.1038/s41591-020-0755-1</pub-id><pub-id pub-id-type="pmid">32123385</pub-id></citation></ref>
</ref-list> 
</back>
</article>