<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3-mathml3.dtd">
<article article-type="methods-article" dtd-version="1.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Bioinform.</journal-id>
<journal-title-group>
<journal-title>Frontiers in Bioinformatics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Bioinform.</abbrev-journal-title>
</journal-title-group>
<issn pub-type="epub">2673-7647</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1657021</article-id>
<article-id pub-id-type="doi">10.3389/fbinf.2025.1657021</article-id>
<article-version article-version-type="Version of Record" vocab="NISO-RP-8-2008"/>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Methods</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Genetic risk predictions using deep learning models with summary data</article-title>
<alt-title alt-title-type="left-running-head">Wang et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fbinf.2025.1657021">10.3389/fbinf.2025.1657021</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Angela</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x26; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/">Writing - review and editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal Analysis</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing - original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Software" vocab-term-identifier="https://credit.niso.org/contributor-roles/software/">Software</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Xiao</surname>
<given-names>Elena</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3217209"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Software" vocab-term-identifier="https://credit.niso.org/contributor-roles/software/">Software</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x26; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/">Writing - review and editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing - original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal Analysis</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Cheng</surname>
<given-names>Jason</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3274100"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing - original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Software" vocab-term-identifier="https://credit.niso.org/contributor-roles/software/">Software</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal Analysis</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x26; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/">Writing - review and editing</role>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Shen</surname>
<given-names>Xiaoxi</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1404867"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x26; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/">Writing - review and editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing - original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
</contrib>
</contrib-group>
<aff id="aff1">
<label>1</label>
<institution>University School of Milwaukee</institution>, <city>Milwaukee</city>, <state>WI</state>, <country country="US">United States</country>
</aff>
<aff id="aff2">
<label>2</label>
<institution>Department of Mathematics, Texas State University</institution>, <city>San Marcos</city>, <state>TX</state>, <country country="US">United States</country>
</aff>
<aff id="aff3">
<label>3</label>
<institution>Westwood High School</institution>, <city>Austin</city>, <state>TX</state>, <country country="US">United States</country>
</aff>
<author-notes>
<corresp id="c001">
<label>&#x2a;</label>Correspondence: Xiaoxi Shen, <email xlink:href="mailto:rcd67@txstate.edu">rcd67@txstate.edu</email>
</corresp>
</author-notes>
<pub-date publication-format="electronic" date-type="pub" iso-8601-date="2026-01-08">
<day>08</day>
<month>01</month>
<year>2026</year>
</pub-date>
<pub-date publication-format="electronic" date-type="collection">
<year>2025</year>
</pub-date>
<volume>5</volume>
<elocation-id>1657021</elocation-id>
<history>
<date date-type="received">
<day>30</day>
<month>06</month>
<year>2025</year>
</date>
<date date-type="rev-recd">
<day>12</day>
<month>12</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>15</day>
<month>12</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2026 Wang, Xiao, Cheng and Shen.</copyright-statement>
<copyright-year>2026</copyright-year>
<copyright-holder>Wang, Xiao, Cheng and Shen</copyright-holder>
<license>
<ali:license_ref start_date="2026-01-08">https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</license-p>
</license>
</permissions>
<abstract>
<sec>
<title>Background</title>
<p>As a driving force of the Fourth Industrial Revolution, deep learning methods have achieved significant success across various fields, including genetic and genomic studies. While individual-level genetic data is ideal for deep learning models, privacy concerns and data-sharing restrictions often limit its availability to researchers.</p>
</sec>
<sec>
<title>Methods</title>
<p>In this paper, we investigated the potential applications of deep learning models&#x2014;including deep neural networks, convolutional neural networks, recurrent neural networks, and transformers&#x2014;when only genetic summary data, such as linkage disequilibrium matrices, is available. The bootstrap method was used to approximate the test error. Simulation studies and real data analyses were conducted to compare the performance of deep learning methods in genetic risk prediction using individual-level genetic data versus genetic summary data.</p>
</sec>
<sec>
<title>Results</title>
<p>The test mean squared errors (MSEs) of most applied deep learning models are comparable when using individual-level data versus summary data.</p>
</sec>
<sec>
<title>Conclusion</title>
<p>Our results suggest that suitable deep learning methods could also serve as an alternative approach to predict disease related traits when only linkage disequilibrium matrices are available as input.</p>
</sec>
</abstract>
<kwd-group>
<kwd>bootstrap</kwd>
<kwd>deep neural networks</kwd>
<kwd>linkage disequilibrium</kwd>
<kwd>risk prediction</kwd>
<kwd>single nucleotide polymorphisms</kwd>
</kwd-group>
<funding-group>
<funding-statement>The author(s) declared that financial support was not received for this work and/or its publication.</funding-statement>
</funding-group>
<counts>
<fig-count count="5"/>
<table-count count="3"/>
<equation-count count="3"/>
<ref-count count="60"/>
<page-count count="00"/>
</counts>
<custom-meta-group>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Integrative Bioinformatics</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<label>1</label>
<title>Introduction</title>
<p>With the high throughput of genetic sequencing technology and the success of genome-wide association studies (GWAS), numerous disease-related single nucleotide polymorphisms (SNPs) have been identified in various studies (<xref ref-type="bibr" rid="B5">Consortium, 2007</xref>; <xref ref-type="bibr" rid="B38">Scott et al., 2007</xref>; <xref ref-type="bibr" rid="B46">Sladek et al., 2007</xref>). Additionally, in 2015, President Obama launched the Precision Medicine Initiative, which aims to revolutionize medical treatment for complex diseases (<xref ref-type="bibr" rid="B4">Collins and Varmus, 2015</xref>). One of the most critical aspects of precision medicine is customizing treatments based on the unique genetic information carried by each individual. Therefore, accurate risk prediction of disease onset and progression based on an individual&#x2019;s genetic data can enable targeted preventive treatments (<xref ref-type="bibr" rid="B21">Jostins and Barrett, 2011</xref>).</p>
<p>It is widely believed that multiple SNPs contribute to a trait, with the genetic effect arising from the cumulative impact of these genetic variants (<xref ref-type="bibr" rid="B56">Yang et al., 2010</xref>). As a result, linear mixed-effects models are conventional tools for genetic risk prediction, where the fixed effects include the clinical and demographic variables (e.g., age and gender) and the effect size of each SNP in a genetic region is treated as a random variable. The total genetic effect is calculated by aggregating the SNPs and their random effects. The Best Linear Unbiased Prediction (BLUP) is a commonly used method for predicting disease-related traits (<xref ref-type="bibr" rid="B3">Campos et al., 2013</xref>; <xref ref-type="bibr" rid="B47">Speed and Balding, 2014</xref>). However, one limitation of linear mixed-effects models is the assumption that the relationship between SNPs and the trait is strictly linear. In reality, many diseases, such as Alzheimer&#x2019;s disease (AD), have substantial genetic components and complex genetic etiologies (<xref ref-type="bibr" rid="B22">Karch et al., 2014</xref>; <xref ref-type="bibr" rid="B45">Sims et al., 2020</xref>). Therefore, it is more reasonable to assume that the relationships between SNPs and disease-related traits are nonlinear. This includes nonlinear relationships between genetic regions such as epistasis as well as those within a genetic region including compound heterozygosity (<xref ref-type="bibr" rid="B34">Nazarian et al., 2023</xref>) and allele dosage for AD (<xref ref-type="bibr" rid="B19">Hostage et al., 2013</xref>). Although such nonlinearity can be incorporated into a linear mixed-effects model through kernel methods (<xref ref-type="bibr" rid="B16">Hofmann et al., 2008</xref>), the performance of BLUP depends on the choice of kernels. In general, it is not clear which kernel is optimal for prediction. In this manuscript, the hippocampal volume is used as the phenotype in the real data analysis. The motivation for this choice stems from previous findings. First, the hippocampus plays a crucial role in memory and is particularly vulnerable to damage at the early stages of Alzheimer&#x2019;s disease (AD) (<xref ref-type="bibr" rid="B33">Mu and Gage, 2011</xref>). Changes in hippocampal volume over time can have a significant impact on AD progression (<xref ref-type="bibr" rid="B37">Schuff et al., 2009</xref>). Accurately predicting hippocampal atrophy could therefore facilitate early intervention in disease development. Moreover, the genetic influence on hippocampal volume is relatively high. For instance, in a study of a large sample of elderly twin men, (<xref ref-type="bibr" rid="B50">Sullivan et al., 2001</xref>) showed that approximately 40% of the variance in hippocampal volume is attributable to genetic factors. In addition, an exploratory GWAS of hippocampal volume using data from the Sydney Memory and Ageing Study suggested that the heritability of hippocampal volume is 62%&#x2013;65% (<xref ref-type="bibr" rid="B32">Mather et al., 2015</xref>). Second, the relationship between SNPs and hippocampal volume is potentially complex. For instance, studies have identified interactions between rs1345203 and rs1213205 that explain 1.9% of the variance in temporal lobe volume (<xref ref-type="bibr" rid="B14">Hibar et al., 2015</xref>). The ability of deep learning methods to capture nonlinear relationships may help improve predictive performance in this context. As an example illustrating the predictive performance of deep learning models based on genetic data, (<xref ref-type="bibr" rid="B31">Liu et al., 2022</xref>) used a deep neural network&#x2013;based model to predict AV45 and FDG, two tracers in positron emission tomography (PET) imaging used to model biological processes in the brain, based on genetic data. Compared with other mixed-effects model&#x2013;based methods, the deep neural network&#x2013;based models achieved better prediction accuracy.</p>
<p>Since 2010, advancements in deep learning technologies have become the driving force behind the fourth Industrial Revolution (<xref ref-type="bibr" rid="B1">Bai et al., 2020</xref>). With numerous successful applications, such as in computer vision and natural language processing, deep neural networks (DNNs) have emerged as one of the most popular research tools across various scientific fields. A major advantage of DNNs is their ability to capture complex relationships between variables, thanks to the universal approximation property (UAP) (<xref ref-type="bibr" rid="B6">Cybenko, 1989</xref>; <xref ref-type="bibr" rid="B18">Hornik et al., 1989</xref>). This makes them suitable candidates for approximating the complex relationships between genetic variants and disease-related traits. On the other hand, kernel methods have also been widely used in genetic studies to capture the nonlinear relationships (<xref ref-type="bibr" rid="B16">Hofmann et al., 2008</xref>). However, the performances of kernel methods highly depend on whether the kernel function has been chosen wisely as well as the specification of hyperparameters in the kernel function (e.g., the degree in a polynomial kernel). Compared to kernel methods, one only needs to specify the number of layers and the number of hidden units in a layer to construct a DNN, which from our point of view, is easier than selecting the correct kernel function. A lot of research has been conducted to uncover complex genotype-phenotype relationships using DNNs. For example, DNNs have been used to model Alzheimer&#x2019;s disease (AD) polygenic risk, outperforming traditional methods (<xref ref-type="bibr" rid="B59">Zhou et al., 2023</xref>). Additionally, Shen and Wang, 2024 employed deep ReLU neural networks to detect significant SNPs associated with phenotypes. Their simulation studies demonstrated that tests based on deep ReLU neural networks are more powerful at detecting nonlinear relationships compared to F-tests in linear models. We refer interested readers to <xref ref-type="bibr" rid="B42">Shen et al. (2022b)</xref> for a review of applications of deep learning models in genetic and genomic studies.</p>
<p>Although individual-level genetic data can improve the precision of predictions, such data is often difficult to obtain due to privacy concerns and data-sharing restrictions. Recently, many researchers have focused on conducting analyses using GWAS summary data, including gene- and pathway-based association tests (<xref ref-type="bibr" rid="B13">Guo and Wu, 2019</xref>; <xref ref-type="bibr" rid="B25">Kwak and Pan, 2016</xref>; <xref ref-type="bibr" rid="B51">Svishcheva et al., 2019</xref>), genetic heritability estimations (<xref ref-type="bibr" rid="B29">Li et al., 2023</xref>; <xref ref-type="bibr" rid="B49">Speed et al., 2020</xref>; <xref ref-type="bibr" rid="B48">Speed and Balding, 2019</xref>), and the detection of causal associations (<xref ref-type="bibr" rid="B55">Xue and Pan, 2020</xref>; <xref ref-type="bibr" rid="B60">Zhu et al., 2018</xref>). The goal of this paper is to explore whether deep learning methods, such as convolutional neural networks (CNNs) (<xref ref-type="bibr" rid="B26">LeCun, 1989</xref>) and long short-term memory networks (LSTMs) (<xref ref-type="bibr" rid="B15">Hochreiter and Schmidhuber, 1997</xref>), can achieve predictive performance on genetic data comparable to that obtained when applying the same model structures to individual-level data.</p>
<p>The rest of the paper is organized as follows: In the Methods section, we briefly review basic deep learning models, including deep neural networks, convolutional neural networks, and recurrent neural networks, such as LSTMs. We then propose a framework for applying deep learning models to GWAS summary data (e.g., the linkage disequilibrium (LD) matrix) and outline the approach for calculating the test error. The Results section presents simulation studies and an application predicting Alzheimer&#x2019;s disease-related traits using real data from the Alzheimer&#x2019;s Disease Neuroimaging Initiative (ADNI). Finally, we conclude with a discussion of the proposed method and its potential future improvements.</p>
</sec>
<sec sec-type="methods" id="s2">
<label>2</label>
<title>Methods</title>
<sec id="s2-1">
<label>2.1</label>
<title>Simulation data</title>
<p>To evaluate the proposed methodologies, extensive simulation studies were conducted. The simulations involved applying DNN, CNN, LSTM and Transformers to both individual-level genetic data and summary data (i.e., LD matrix). The training errors and testing errors obtained using the summary data were compared with those from using the individual-level data. The data used for simulations was generated using R and the implementation of the deep learning methods was conducted using the Keras package in python.</p>
<sec id="s2-1-1">
<label>2.1.1</label>
<title>Individual-level data</title>
<p>To mimic the real structure of genetic sequencing data, the data used for simulation were generated based on the real sequencing data from Chromosome 17: 7344328-8344327 in the 1,000 Genomes Project (<xref ref-type="bibr" rid="B52">The 1000 Genomes Project Consortium, 2010</xref>). The minor allele frequencies (MAF) of the SNPs in this region range from 0.046% to 49.954%. Since deep learning models have better performance when the signal is strong (<xref ref-type="bibr" rid="B20">James et al., 2013</xref>), including rare SNPs could deteriorate the performance. Therefore, we removed SNPs with MAF &#x3c;0.001 were removed. Each remaining SNP was scaled to have a sample mean of 0 and a sample standard deviation of <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>/</mml:mo>
<mml:msqrt>
<mml:mi>p</mml:mi>
</mml:msqrt>
</mml:mrow>
</mml:math>
</inline-formula>, where <italic>p</italic> &#x3d; 8,299 is the number of common SNPs in this region. To simulate the response variable, 30% of the common SNPs were randomly selected as causal variants, and the response variable was generated using the following equation:<disp-formula id="equ1">
<mml:math id="m2">
<mml:mrow>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>K</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>&#x3b5;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>where <italic>K</italic> is the number of causal variants; <inline-formula id="inf2">
<mml:math id="m3">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the value of the <italic>k</italic>th causal SNP for the <italic>i</italic>th individual and <inline-formula id="inf3">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mi>K</mml:mi>
</mml:msub>
<mml:mo>&#x223c;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>.</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>.</mml:mo>
<mml:mi>d</mml:mi>
<mml:mo>.</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mn>0.6</mml:mn>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf4">
<mml:math id="m5">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b5;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>&#x3b5;</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
<mml:mo>&#x223c;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>.</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>.</mml:mo>
<mml:mi>d</mml:mi>
<mml:mo>.</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. The dataset consists of <italic>n</italic> &#x3d; 1,092 individuals. When training deep learning models, 80% of the samples were randomly selected as training data and the remaining samples were used as test data.</p>
</sec>
<sec id="s2-1-2">
<label>2.1.2</label>
<title>Construction of LD matrices</title>
<p>In terms of the summary data, the LD matrices were generated as follows. Let <italic>G</italic> denote the scaled SNP data as mentioned above, which contains 1,092 individuals and 8,299 SNPs. To generate the LD matrix for training the deep learning models, 80% of the samples were randomly selected, and the SNPs of the remaining individuals were used to generate the LD matrix for testing purposes. Let <inline-formula id="inf5">
<mml:math id="m6">
<mml:mrow>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> be the design matrix of the scaled SNPs in the training data and <inline-formula id="inf6">
<mml:math id="m7">
<mml:mrow>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> be the design matrix of the scaled SNPs in the test data. In other words, <inline-formula id="inf7">
<mml:math id="m8">
<mml:mrow>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> was extracted from <italic>G</italic> by taking the rows corresponding to the training samples and <inline-formula id="inf8">
<mml:math id="m9">
<mml:mrow>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> contains the remaining rows in <italic>G</italic>. The LD matrices for training and testing were generated by calculating <inline-formula id="inf9">
<mml:math id="m10">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:msubsup>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf10">
<mml:math id="m11">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:msubsup>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, respectively. Here <inline-formula id="inf11">
<mml:math id="m12">
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>873</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> represents the number of training data and <inline-formula id="inf12">
<mml:math id="m13">
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>219</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> represents the number of test data.</p>
</sec>
</sec>
<sec id="s2-2">
<label>2.2</label>
<title>Real data</title>
<p>Alzheimer&#x2019;s disease (AD) is one of the most common neurodegenerative diseases, significantly influenced by genetic factors (<xref ref-type="bibr" rid="B22">Karch et al., 2014</xref>; <xref ref-type="bibr" rid="B45">Sims et al., 2020</xref>). Effective predictions on the development of AD based on genetic components could lead to early intervention as well as targeted treatment of the disease. In this section, we applied the proposed methods to perform genetic risk prediction using deep learning models on a real dataset from the Alzheimer&#x2019;s Disease Neuroimaging Initiative (ADNI (<ext-link ext-link-type="uri" xlink:href="https://adni.loni.usc.edu/">https://adni.loni.usc.edu/</ext-link>)). The ADNI study is a multisite, longitudinal observational study aimed at improving the scientific understanding of Alzheimer&#x2019;s disease (AD). Data from phases 1 and 2 of the ADNI study were used in the analysis. A total of 3,108 participants were included across these two phases, with about 89% of the participants identifying as white and an average baseline age of 72.76 years. More demographic information is summarized in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Demographics of participants in ADNI1 and ADNI2 studies.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Demographic variable</th>
<th align="right">Mean (standard deiviation)/Relative frequency</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Age</td>
<td align="right">72.76 (7.7)</td>
</tr>
<tr>
<td align="left">Years of education</td>
<td align="right">15.4 (3.99)</td>
</tr>
<tr>
<td colspan="2" align="left">Gender</td>
</tr>
<tr>
<td align="left">    Male</td>
<td align="right">53.15%</td>
</tr>
<tr>
<td align="left">    Female</td>
<td align="right">43.98%</td>
</tr>
<tr>
<td colspan="2" align="left">Self identified race</td>
</tr>
<tr>
<td align="left">    American indian or alaskan native</td>
<td align="right">0.06%</td>
</tr>
<tr>
<td align="left">    Asian</td>
<td align="right">1.54%</td>
</tr>
<tr>
<td align="left">    Native Hawaiian or other pacific islander</td>
<td align="right">0.06%</td>
</tr>
<tr>
<td align="left">    Black or african american</td>
<td align="right">4.63%</td>
</tr>
<tr>
<td align="left">    White</td>
<td align="right">89.25%</td>
</tr>
<tr>
<td align="left">    More than one race</td>
<td align="right">0.68%</td>
</tr>
<tr>
<td align="left">    Unknown</td>
<td align="right">0.35%</td>
</tr>
<tr>
<td colspan="2" align="left">Self identified ethnicity</td>
</tr>
<tr>
<td align="left">    Hispanic or latino</td>
<td align="right">3.09%</td>
</tr>
<tr>
<td align="left">    Not hispanic or latino</td>
<td align="right">92.82%</td>
</tr>
<tr>
<td align="left">    Unknown</td>
<td align="right">0.61%</td>
</tr>
<tr>
<td colspan="2" align="left">Participant&#x2019;s primary language</td>
</tr>
<tr>
<td align="left">    English</td>
<td align="right">93.56%</td>
</tr>
<tr>
<td align="left">    Spanish</td>
<td align="right">1.51%</td>
</tr>
<tr>
<td align="left">    Other</td>
<td align="right">1.58%</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>A total of 808 samples at the screening and baseline of the ADNI1 and ADNI2 studies have the whole genome sequencing data, and we used SNPs from the <italic>APOE</italic> gene located on Chromosome 19: 45409005-45412652. Variants with a call rate &#x3c;99% or a Hardy-Weinberg equilibrium p-value &#x3c; 1e-6 were removed. The final SNP dataset included 780 individuals and 168 SNPs. Regarding the choice of the response variable, the hippocampus region was selected because it plays a vital role in memory (<xref ref-type="bibr" rid="B33">Mu and Gage, 2011</xref>), and shrinkage in the hippocampal volume is an early symptom of AD (<xref ref-type="bibr" rid="B37">Schuff et al., 2009</xref>). Thus, the volume of the hippocampus region was chosen as a potential response variable. In this dataset, the hippocampal volume has mean 6778.89 with a standard deviation of 1179.07. We first took the logarithm of the hippocampal volume such that it has an approximately normal shape. To remove some confounding effects, we regressed the logarithm of hippocampal volume onto important predictors, including age (mean: 73.46, sd: 7.01), gender (Male: 35.98%, Female: 63.94%), and number of years in education (mean: 16.08, sd: 2.78). The residuals obtained from this regression were used as the response variable to train the deep learning models.</p>
<p>Similar to the simulation studies, 80% of the data were randomly selected as the training data, and the remining data was served as the test data to evaluate model performances. The LD matrices for training and testing were obtained using the matrix inner product of the SNP matrix corresponding to the <italic>APOE</italic> gene in the training set and test set, respectively. The formula of generating the LD matrices is the same as those described in the simulation data. The training data consists of 624 individuals while the test data has 156 individuals.</p>
</sec>
<sec id="s2-3">
<label>2.3</label>
<title>Deep neural networks (DNNs)</title>
<p>The development of deep neural networks can be traced back to 1950s when a mathematical model, known as the perceptron (<xref ref-type="bibr" rid="B36">Rosenblatt, 1958</xref>), was proposed to model the functionality of neurons in a human brain. Around the 1990s, multiple perceptrons were stacked to create artificial neural networks. An artificial neural network consists of three layers: the input layer contains all the features from the data used to make predictions; the hidden layer contains several units to further extract useful information from the inputs, and the output layer produces the output of the artificial neural network to make predictions. Deep neural networks are obtained by including multiple hidden layers. One of the major advantages discovered for DNN is their ability to capture these complex relationships due to their famous universal approximation property (UAP) (<xref ref-type="bibr" rid="B6">Cybenko, 1989</xref>; <xref ref-type="bibr" rid="B18">Hornik et al., 1989</xref>; <xref ref-type="bibr" rid="B57">Yarotsky, 2017</xref>; <xref ref-type="bibr" rid="B58">Yarotsky and Zhevnerchuk, 2020</xref>). This suggests that DNNs may be suitable candidates for approximating the complex structures present in such research problems. Because of the UAP, deep neural networks are popular tools in predictive analyses nowadays.</p>
<p>Since SNP data and LD matrices often contain nonlinear, high-dimensional relationships that are difficult to capture with classical models. Deep neural networks are well suited because stacked nonlinear layers can learn complex interaction patterns without requiring specifying the functional form of underlying relationship explicitly. The flexibility of DNNs makes them effective when the relevant predictive structure is distributed across many genetic regions. In the application of DNNs to genetic risk prediction, each input unit represents a SNP in a genetic region (e.g., a gene). Throughout the remainder of the paper, additive coding was used for the genotypes (i.e., 0 for genotype AA, 1 for genotype Aa, and 2 for genotype aa) in the raw genetic data. In additive coding, the alleles assigned to &#x2018;A&#x2019; and &#x2018;a&#x2019; are not arbitrary. &#x2018;A&#x2019; always represents the allele with a higher frequency in the population (the major allele), and &#x2018;a&#x2019; represents the allele with a lower frequency in the population (the minor allele). In other words, the numerical value used for encoding a genotype in additive coding corresponds to the number of minor alleles in the genotype. Additive coding has been commonly used in the statistical genetics literature (<xref ref-type="bibr" rid="B54">Wu et al., 2011</xref>; <xref ref-type="bibr" rid="B27">Li et al., 2014</xref>; <xref ref-type="bibr" rid="B43">Shen et al., 2022c</xref>). As mentioned in the Simulated Data and Real Data sections, each column of the genotype matrix, which represents a SNP, will be standardized so that the column has mean 0 and standard deviation <inline-formula id="inf13">
<mml:math id="m14">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>/</mml:mo>
<mml:msqrt>
<mml:mi>p</mml:mi>
</mml:msqrt>
</mml:mrow>
</mml:math>
</inline-formula> with <italic>p</italic> being the number of SNPs. In other words, the actual SNP coding used in the analysis are <inline-formula id="inf14">
<mml:math id="m15">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:msqrt>
<mml:mi>p</mml:mi>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:msqrt>
<mml:mi>p</mml:mi>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf15">
<mml:math id="m16">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:msqrt>
<mml:mi>p</mml:mi>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</inline-formula> with <italic>m</italic> being the column mean and <italic>s</italic> being the column standard deviation. The genetic information then passes through hidden layers to extract important features from the data. The output layer, which contains a single unit with a linear activation function, produces the predicted value of the trait. <xref ref-type="fig" rid="F1">Figure 1</xref> provides a graphical illustration of the structure of a deep neural network.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Architecture of a deep neural network applied on genetic data.</p>
</caption>
<graphic xlink:href="fbinf-05-1657021-g001.tif">
<alt-text content-type="machine-generated">Diagram depicting a neural network model for genetic prediction. DNA strand on the left shows several Single Nucleotide Polymorphisms (SNPs) labeled SNP 1 to SNP p. These SNPs act as inputs to interconnected network layers, represented by blue and orange nodes, leading to a single output.</alt-text>
</graphic>
</fig>
<p>When applying DNNs to the simulated and the real individual level SNP dataset, each SNP was used as an input. In other words, the input dimension is 8,299 for the simulated data and is 168 for the real data. We used a similar structure as described in <xref ref-type="bibr" rid="B59">Zhou et al. (2023)</xref>, with the network comprising four hidden layers containing 231, 77, 22, and 5 hidden units, respectively. Additionally, one dropout layer with dropout rate 0.2 was added after the first hidden layer and another dropout layer with dropout rate 0.5 was added after the third hidden layer. When the DNNs were trained using the adaptive moment estimation (ADAM) (<xref ref-type="bibr" rid="B23">Kingma and Ba, 2017</xref>), we set the number of epochs to 100, the batch size to 256, and the learning rate to 0.001 with a decay rate of 0.96. These hyperparameters were selected based on validation errors from a predefined set of candidate values.</p>
</sec>
<sec id="s2-4">
<label>2.4</label>
<title>Convolutional neural networks (CNNs)</title>
<p>CNNs are a class of neural networks that use filters to process multidimensional data, such as images, by extracting relevant features. The core building blocks of CNN include convolutional layers and pooling layers, which work together to refine feature representations (<xref ref-type="bibr" rid="B28">Li et al., 2022</xref>).</p>
<p>Each convolutional layer consists of several filters, with each filter acting as a sliding window that applies a nonlinear activation function (such as ReLU) to the linear combination of filter entries and outputs from the previous layer, producing feature maps. Pooling layers reduce the size of the representation, accelerating computations and enhancing the robustness of detected features. A commonly used pooling technique is max pooling, which extracts the maximum value within a sliding window. After passing through multiple convolutional and pooling layers, the extracted features are flattened into a vector and fed into a fully connected neural network to make final predictions. <xref ref-type="fig" rid="F2">Figure 2</xref> illustrates the basic structure of a convolutional neural network.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Structure of a convolutional neural network. The input is a 2-dimensional array (in our applications, it could be a SNP matrix or a LD matrix). The blue cuboid and the yellow cuboid represent convolutional layers and max-pooling layers respectively. After the last max pooling layer, the features were fed to a fully connected deep neural network to make predictions.</p>
</caption>
<graphic xlink:href="fbinf-05-1657021-g002.tif">
<alt-text content-type="machine-generated">Diagram of a convolutional neural network architecture. It starts with input dimensions of 168x168x1. Layers include multiple 2D convolutional and max pooling stages, reducing sizes progressively: 164x164x32, 82x82x32, 80x80x64, 40x40x64, 38x38x64, 19x19x64, 17x17x64, and 8x8x64. A flatten layer leads to dense layers with dimensions of 4096, 2048, 256, and 96, and ends with an output of 4 for prediction.</alt-text>
</graphic>
</fig>
<p>Additionally, the weights in the filters of convolutional layers and in the fully connected neural network are trained using backpropagation, with the learning rate controlling the rate of parameter updates. However, CNNs may encounter issues such as vanishing or exploding gradients during training. To improve generalization and reduce overfitting to the training data, dropout layers can be employed. These layers randomly deactivate nodes during training, preventing the model from relying too heavily on specific neurons. This encourages the development of more general and robust models. While CNNs excel at capturing local patterns, they may struggle with broader patterns, which require additional and larger-size filters. However, increasing filter sizes and quantity could significantly raises computational cost and training time.</p>
<p>Due to the fact that CNNs can capture local patterns, it makes CNNs a natural fit when the SNPs exhibit spatial structure along the genome. In particular, adjacent SNPs tend to be correlated due to LD, so a CNN&#x2019;s convolutional filters can detect local patterns or LD blocks similarly to how they detect edges or textures in images. Moreover, weight sharing dramatically reduces the number of parameters, helping the model generalize even when training data are limited. This makes CNNs especially useful for modeling local genetic architecture and short-range dependencies. When applying CNNs to simulated and real individual level SNP data, the inputs are the values of each SNP and 1-dimensional filters were used (i.e., the input dimension is 8,299 for the simulated data and is 168 for the real data). The rationale for using 1-dimensional filters for individual-level data is that individuals are typically considered independent observations, a common assumption in machine learning theory. As a result, local information is only present within the observations of the SNPs. Based on validation errors from a predefined set of candidate hyperparameter values using 1-dimensional filters, two CNN structures outperformed the others. The first structure consists of one convolutional layer with 50 filters, each of size 500, and one hidden neural network layer with 50 hidden units. The second structure includes one convolutional layer with 50 filters, each of size 500, and five hidden neural network layers, each with 50 hidden units. In all cases, each hidden layer used a ReLU activation function. Training was conducted over 200 epochs with a batch size of 32, an initial learning rate of 0.1, and a decay rate of 0.98.</p>
</sec>
<sec id="s2-5">
<label>2.5</label>
<title>Recurrent neural networks (RNNs)</title>
<p>RNNs are a class of neural networks extended to include feedback connections. This allows them to capture temporal patterns in sequential data. The long-short term memory (LSTM) is a special type of RNN aimed at learning long-range dependencies by mitigating the vanishing gradient problem that traditional RNNs struggle with (<xref ref-type="bibr" rid="B2">Bengio et al., 1994</xref>). An LSTM unit uses input, forget, and output gates to control the flow of information into and out of its memory cell. This structure allows the network to retain relevant information across longer time spans. <xref ref-type="fig" rid="F3">Figure 3a</xref> provides an illustration of the structure of an LSTM unit.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>
<bold>(a)</bold> Structure of an LSTM unit where <inline-formula id="inf16">
<mml:math id="m17">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the current input of a LSTM unit; <inline-formula id="inf17">
<mml:math id="m18">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf18">
<mml:math id="m19">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the outputs from the previous LSTM unit and the current LSTM unit respectively; <inline-formula id="inf19">
<mml:math id="m20">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf20">
<mml:math id="m21">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represent the memories from the previous LSTM units and the updated memory after the current LSTM unit, respectively. Additionally, <inline-formula id="inf21">
<mml:math id="m22">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and tanh are the sigmoid and hyperbolic tangent activation function. The unit consists of three main gates: the forget gate, the input gate, and the output gate. The forget gate uses a dense layer with inputs <inline-formula id="inf22">
<mml:math id="m23">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf23">
<mml:math id="m24">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> to determine what proportion of previous information should be discarded. The input gate consists of two parts: the first is a dense layer with a sigmoid activation function, and the second is another dense layer with a hyperbolic tangent activation function. Both dense layers take <inline-formula id="inf24">
<mml:math id="m25">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf25">
<mml:math id="m26">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> as input. The first part determines how much new information <inline-formula id="inf26">
<mml:math id="m27">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> contributes to the cell and the second part provides a candidate cell state. The output gate, which also consists of a dense layer, controls what information from the cell state is used to compute the final output. <bold>(b)</bold> A BiLSTM contains both forward and backward loops to improve the model for capturing long-rage dependencies. In the figure each box of LSTM represents a single LSTM unit with the structure shown in <bold>(a)</bold> Here <inline-formula id="inf27">
<mml:math id="m28">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> form a segment of input sequence and <inline-formula id="inf28">
<mml:math id="m29">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the corresponding output sequence.</p>
</caption>
<graphic xlink:href="fbinf-05-1657021-g003.tif">
<alt-text content-type="machine-generated">Diagram illustrating two components of Long Short-Term Memory (LSTM) networks. (a) Shows the internal structure including forget, input, and output gates, utilizing sigma and tanh functions. (b) Displays a sequence of LSTM units processing data over time steps, with inputs \(x_{t-1}\), \(x_t\), and \(x_{t&#x2b;1}\), resulting in outputs \(y_{t-1}\), \(y_t\), and \(y_{t&#x2b;1}\).</alt-text>
</graphic>
</fig>
<p>Due to linkage disequilibrium, SNPs are often correlated, making LSTM an ideal model for capturing sequential dependencies in genetic data. In addition, the physical positions of SNPs reflect the underlying linkage disequilibrium (LD) structure and local genomic context, as nearby SNPs often exhibit correlated variation due to shared inheritance. By leveraging its memory cells and gating mechanisms, LSTM can be trained on numerical SNP sequences, which are the ordered sequences of SNP genotypes encoded as 0, 1 and 2, representing the number of minor alleles at each locus, to learn complex temporal and spatial dependencies in genomic structures. This capability makes it particularly effective for identifying nonlinear relationships between correlated genetic features and response variables. In our project, we numerically encoded SNPs as inputs for models composed of multiple LSTM units and trained these models to predict disease-related traits.</p>
<p>To further improve the models&#x2019; ability to capture long-range dependencies, we also implemented a bidirectional approach. Unlike unidirectional LSTMs, which only process information in the forward direction, bidirectional LSTM (BiLSTM) networks traverse the input data both forwards and backwards. This allows them to produce outputs based on later context, while LSTM relies only on previous context. As a result, BiLSTM networks generally outperform LSTM (<xref ref-type="bibr" rid="B12">Graves and Schmidhuber, 2005</xref>; <xref ref-type="bibr" rid="B44">Siami-Namini et al., 2019</xref>). The structure of a BiLSTM is described in <xref ref-type="fig" rid="F3">Figure 3b</xref>.</p>
<p>Similar to the DNN and CNN models, the inputs for the LSTM and BiLSTM models are the SNP values in the data. Hence, the input dimensions for the simulated data and the real data are 8,299 and 168, respectively. When applying LSTM to individual-level SNP data, we used a five-layer LSTM, with each layer containing 10 LSTM units, followed by a dense layer. Training was conducted over 10 epochs with a batch size of 256. For the BiLSTM, we used a two-layer architecture, with each layer containing 10 BiLSTM units, followed by a dense layer. The model was trained for 5 epochs with a batch size of 256. For both LSTM and BiLSTM models, the ADAM optimizer was used with an initial learning rate of 0.001 and a decay rate of 0.96. These hyperparameters were determined based on validation errors from a predefined set of candidate values.</p>
</sec>
<sec id="s2-6">
<label>2.6</label>
<title>Transformers</title>
<p>The transformer architecture (<xref ref-type="bibr" rid="B53">Vaswani et al., 2017</xref>) brings deep learning into a modern era. Although the original work focused on English&#x2013;German machine translation, transformers have since been widely applied to a broad range of tasks, including applications in genetics and genomics (<xref ref-type="bibr" rid="B11">Gra&#xe7;a et al., 2024</xref>; <xref ref-type="bibr" rid="B30">Li et al., 2025</xref>). A transformer model consists of an encoder and a decoder. In our application, only the encoder component was used. The structure of the transformer encoder is illustrated in <xref ref-type="fig" rid="F4">Figure 4</xref>. There are four main components in a transformer encoder:<list list-type="order">
<list-item>
<p>Input embedding converts categorical inputs (e.g., each word or symbol in the vocabulary) into numerical vectors.</p>
</list-item>
<list-item>
<p>Positional encoding allows the transformer to keep track of the order of words in a sequence.</p>
</list-item>
<list-item>
<p>Multi-head attention computes the relationships between each word and all the words in the sentence, including itself.</p>
</list-item>
<list-item>
<p>Residual connections provide shortcut paths that stabilize and speed up the training of deep networks.</p>
</list-item>
</list>
</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Structure of a transformer encoder. There are four main components in the structure. Input embedding converts categorical inputs (e.g., each word or symbol in the vocabulary) into numerical vectors. Positional encoding allows the transformer to keep track of the order of words in a sequence. Multi-head attention computes the relationships between each word and all other words in the sentence, including itself and the residual connections provide shortcut paths that help stabilize and speed up the training of deep networks.</p>
</caption>
<graphic xlink:href="fbinf-05-1657021-g004.tif">
<alt-text content-type="machine-generated">Flowchart of a transformer encoder. Flowchart of a neural network architecture, starting with &#x22;Inputs&#x22; leading to &#x22;Input Embedding,&#x22; then &#x22;Position Encoding.&#x22; Followed by &#x22;Multi-head Attention,&#x22; &#x22;Add &#x26; Norm,&#x22; &#x22;Feedforward Neural Network,&#x22; and another &#x22;Add &#x26; Norm,&#x22; with arrows indicating data flow and feedback loops.</alt-text>
</graphic>
</fig>
<p>Since transformers use self-attention to model relationships between all pairs of positions in a sequence simultaneously, it makes them highly effective for genetic data where both local and long-range LD patterns matter. Instead of processing the sequence step by step, transformers directly learn how each SNP is related to others. In addition, interactions across the genetic region or LD matrix are modeled more flexibly, making transformers particularly powerful for discovering complex relationships across the genetic regions. On the other hand, because of the high dimensionality of SNP data (e.g., the simulated dataset contains 8,299 SNPs), feeding all SNPs into a transformer simultaneously would result in excessive memory usage and computational cost. To address this issue, we applied a sliding window of size 1,000 with a stride of 500, generating 15 overlapping SNP sequences of length 1,000. Each of these SNP sequences was used as input to the transformer, and the outputs from the multi-head attention units were averaged to form the input to the final feedforward neural network. The number of multi-head attention units was set to 6, and the number of hidden units in the final feedforward neural network was 64. The ADAM optimizer was used with a batch size of 8 and 10 training epochs.</p>
</sec>
<sec id="s2-7">
<label>2.7</label>
<title>Applications to summary data</title>
<p>Although deep learning models are expected to perform better on individual-level genetic data, such data are not always accessible due to privacy concerns and data-sharing restrictions. In contrast, genetic summary data are more readily available, making it worthwhile to investigate the performance of deep learning models on this type of data. Additionally, it is important to assess whether prediction accuracy based on genetic summary data is comparable to that achieved with individual-level data. In this paper, we utilized the LD matrix as our summary data. Although other GWAS summary statistics can also help address privacy concerns, using LD matrices as summary data offers additional benefits. GWAS summary statistics, such as polygenic risk scores (PRS), are calculated based on the marginal linear effects of SNPs. As a result, information about the interactions and correlations among SNPs is not taken into account, whereas LD matrices preserve these correlations and can improve predictive performance.</p>
<p>A significant challenge in applying deep learning models to genetic summary data is evaluating the prediction error on both the training and, more importantly, the test data. Our proposed approach is illustrated in <xref ref-type="fig" rid="F5">Figure 5a</xref>. During the training phase, the LD matrix of a genetic region, derived from the training data, is used as input. Deep learning models such as DNN, CNN, or LSTM are then applied to this input LD matrix. Unlike the traditional approach, where the number of units in the output layer matches the dimension of the response variable, this framework sets the number of output units equal to the number of training samples. Let <inline-formula id="inf29">
<mml:math id="m30">
<mml:mrow>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> be the quantitative response variables in the training set and let <inline-formula id="inf30">
<mml:math id="m31">
<mml:mrow>
<mml:msub>
<mml:mi>O</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>O</mml:mi>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> be the outputs from a deep learning model, the parameters in a deep learning model were trained via ADAM to minimize the loss function<disp-formula id="equ2">
<mml:math id="m32">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="&#x7c;">
<mml:mrow>
<mml:msub>
<mml:mi>O</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Illustration of application of deep learning models to genetic summary data. <bold>(a)</bold> In the training phase, the LD matrix of a genetic region, derived from the training data, is used as input. Deep learning models such as DNN, CNN, or LSTM are then applied to this input LD matrix. The number of output units of these deep learning methods is set to be equal to the number of training samples. In the testing phase, the LD matrix from the test data will first be fed to the trained deep learning model. <inline-formula id="inf31">
<mml:math id="m33">
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> outputs are then randomly selected with replacement from the <inline-formula id="inf32">
<mml:math id="m34">
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> outputs. Denote the test error based on this random sample as <inline-formula id="inf33">
<mml:math id="m35">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. This process is then repeated <italic>B</italic> times to obtain a set of bootstrapped test error samples <inline-formula id="inf34">
<mml:math id="m36">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>T</mml:mi>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mi>B</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. The approximated test error will be the mean of these <italic>B</italic> test errors. <bold>(b)</bold> In the situation of large LD matrix, to reduce computational burden, the LD matrix will be broken into smaller submatrix along the diagonal (as shown in the orange box). These smaller LD matrices will then be used as inputs to train deep learning models.</p>
</caption>
<graphic xlink:href="fbinf-05-1657021-g005.tif">
<alt-text content-type="machine-generated">Diagram showing a machine learning model workflow. (a) In the training phase, an LD matrix of SNPs is input into a deep learning model (e.g., DNN, CNN, or BiLSTM), resulting in an output layer with \( n_{tr} \) units. In the testing phase, \( n_{te} \) units undergo sampling with replacement. (b) A matrix displays correlation values between pairs of SNPs, with specific values highlighted in orange.</alt-text>
</graphic>
</fig>
<p>To evaluate the test error, we adopted the resampling approach from the well-known bootstrap method (<xref ref-type="bibr" rid="B8">Efron and Tibshirani, 1994</xref>). First, the LD matrix from the test data will be fed to the trained deep learning model. Once the <inline-formula id="inf35">
<mml:math id="m37">
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> outputs are obtained, we randomly select <inline-formula id="inf36">
<mml:math id="m38">
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> outputs with replacement from these outputs. Denote the test error based on this random sample as <inline-formula id="inf37">
<mml:math id="m39">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. This process is then repeated <italic>B</italic> times to obtain a set of bootstrapped test error samples <inline-formula id="inf38">
<mml:math id="m40">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>T</mml:mi>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mi>B</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. The approximated test error will be the mean of these <italic>B</italic> test errors:<disp-formula id="equ3">
<mml:math id="m41">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:mrow>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>B</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mi>T</mml:mi>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>The following network structures were used when applying the deep learning models to LD matrices:<list list-type="bullet">
<list-item>
<p>
<italic>DNN</italic>: The elements in the upper triangle of the LD matrix were used as inputs for the DNN. The network structure was the same as that used for the individual-level SNP data. The ADAM optimizer was applied, and we set the number of epochs to 100, the batch size to 256, and the learning rate to 0.001 with a decay rate of 0.96.</p>
</list-item>
<list-item>
<p>
<italic>CNN</italic>: Since LD matrices are inherently two-dimensional, and the inputs used in CNNs are typically images, we treated the LD matrices as &#x201c;images&#x201d; when applying CNNs. Unlike the CNN models for individual-level SNP data, two-dimensional filters were applied when the inputs were LD matrices. We then used the same CNN architecture as that described for the individual-level SNP data. Due to the structure of CNNs, the size of the two-dimensional filters was scaled down by a factor of 10 compared to their one-dimensional counterparts. In other words, two-dimensional filters of size 50 &#xd7; 50 were used instead. This scaling was necessary because most CNN models with two-dimensional filters could not complete training within a reasonable time frame. Training was conducted over 200 epochs with a batch size of 32, an initial learning rate of 0.1, and a decay rate of 0.98 per epoch.</p>
</list-item>
</list>
</p>
<p>For the real data, since the LD matrix based on the <italic>APOE</italic> gene is significantly smaller than the one used in the simulation studies, we modified the convolutional layers in the two CNN structures. The first structure consists of one convolutional layer with 32 filters, each of size 5, followed by a pooling layer with a pooling size of 2. The second structure includes four convolutional layers: the first with 32 filters of size 5, followed by three layers, each with 64 filters of size 3. Each convolutional layer is followed by a max pooling layer with a pooling size of 2. For all configurations, each hidden layer employed a ReLU activation function. The same CNN architectures were then applied to the summary-level data using two-dimensional (2D) convolutional filters with sizes 5 &#xd7; 5 and 3 &#xd7; 3, along with max pooling layers having filters of size 2 &#xd7; 2.<list list-type="bullet">
<list-item>
<p>
<italic>LSTM/BiLSTM</italic>: When the LSTM or BiLSTM models were applied to the LD matrices, the entire LD matrix was used as input. In other words, the input dimensions were 8,299 &#xd7; 8,299 for the simulated data and 168 &#xd7; 168 for the real data. The same LSTM and BiLSTM architectures used for the individual-level SNP data were also applied to the LD matrices. In addition, the ADAM optimizer was used with 10 epochs and a batch size of 256 for the LSTM, and with 5 epochs and a batch size of 256 for the BiLSTM. For both models, the initial learning rate was set to 0.001, with a decay rate of 0.96 per epoch during training.</p>
</list-item>
<list-item>
<p>
<italic>Transformer</italic>: Due to the large size of the LD matrix in the simulated data, the transformer input consisted of diagonal blocks extracted from the original LD matrix, as described in the following paragraph. Without this operation, training a transformer would be infeasible because of the tremendous memory requirements. For the real data, we directly used the 168 &#xd7; 168 LD matrix as the input. The input data were first flattened and embedded into a vector of dimension 128. Positional encoding was then applied, followed by two transformer blocks, each containing six multi-head attention units, a dropout layer with a dropout rate of 0.1, and layer normalization. Finally, two dense hidden layers, each with 128 hidden units, were applied, followed by another dropout layer (dropout rate 0.1) and layer normalization.</p>
</list-item>
</list>
</p>
<p>Training a deep learning model, especially a CNN or a transformer with an LD matrix as input, can involve a large number of parameters. In addition, when the dimension of the LD matrix is large, storing such a large LD matrix requires substantial memory. Combining with the number of parameters to train in a deep learning model, it could lead to prohibitively long computation time. To reduce the computational burden in cases with large LD matrices, the matrix is divided into smaller block matrices along the diagonal, and these smaller LD matrices are used as inputs instead. This idea is originated from the fact that LD is largely local due to the haplotype block structure of SNPs (<xref ref-type="bibr" rid="B10">Gabriel et al., 2002</xref>) and is illustrated in <xref ref-type="fig" rid="F5">Figure 5b</xref>. Although dividing the LD matrix into smaller blocks may lead to the loss of some long-range SNP relationships, doing so helps mitigate substantial computational challenges. Moreover, when multiple small LD blocks are fed into a deep learning model, such as a convolutional neural network (CNN), the model can still capture correlations between neighbouring blocks, which may partially recover the missing LD information. In our application, a block size of 193 was used so that 8,299 SNPs resulted in exactly 43 smaller blocks, and these blocks of LD matrices were used as inputs when training the CNN and the transformer.</p>
<p>When applying deep learning models with LD matrices as inputs, there is an inherent trade-off between retaining genomic information and managing computational cost. For the DNN and LSTM models, we used the entire LD matrix as input, as the computational costs of DNN and LSTM are less demanding. In DNN applications, we used the upper triangular portion of the LD matrix, which does not result in information loss because the LD matrix is symmetric. In contrast, for the CNN and transformer models, which are more computationally intensive, it should be noted that we used the same block LD matrices along the diagonal as inputs to reduce computational burden.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<label>3</label>
<title>Result</title>
<p>
<xref ref-type="table" rid="T2">Table 2</xref> summarizes the training and test errors of the deep learning models applied to individual-level SNP data as well as to LD matrices. The results were obtained from 500 independent repetitions. Each cell contains the sample mean of the training/test errors based on the 500 runs, with the sample standard deviation provided in parentheses. As shown in <xref ref-type="table" rid="T2">Table 2</xref>, when individual-level SNP data is available, the training errors from the DNN and CNN (structure 1) are relatively small, but their test errors are larger. Notably, the test errors of CNNs are significantly higher than those of DNNs, suggesting that CNNs may be less effective for SNP data. In contrast, both LSTM and BiLSTM models exhibit smaller test errors compared to DNNs, indicating that LSTM-based models may be more suitable for SNP data. In addition, the transformer models achieved a performance comparable to that of the LSTM and BiLSTM on the individual-level SNP data, but the test error was higher when only the LD matrix was available. A key observation from <xref ref-type="table" rid="T2">Table 2</xref> is that when only genetic summary data (LD matrices) is available, deep learning models can still achieve performance comparable to that obtained with individual-level SNP data.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Comparisons between training/test errors of deep learning models on individual-level SNP data and on genetic summary data based on simulated data.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" colspan="2" align="left">Method</th>
<th colspan="2" align="center">Individual-level SNP data</th>
<th colspan="2" align="center">Genetic summary data</th>
</tr>
<tr>
<th align="center">Training error</th>
<th align="center">Test error</th>
<th align="center">Training error</th>
<th align="center">Test error</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td colspan="2" align="left">DNN</td>
<td align="left">0.564 (3.720e-01)</td>
<td align="left">1.233 (8.589e-02)</td>
<td align="left">0.119 (9.714e-02)</td>
<td align="left">1.822 (1.758e-01)</td>
</tr>
<tr>
<td colspan="2" align="left">CNN</td>
<td align="left"/>
<td align="left"/>
<td align="left"/>
<td align="left"/>
</tr>
<tr>
<td align="left"/>
<td align="left">Structure 1</td>
<td align="left">0.356 (3.237e-01)</td>
<td align="left">1.466 (2.198e-01)</td>
<td align="left">0.694 (2.190e-02)</td>
<td align="left">2.303 (6.369e-04)</td>
</tr>
<tr>
<td align="left"/>
<td align="left">Structure 2</td>
<td align="left">1.338 (5.434e-01)</td>
<td align="left">1.367 (6.085e-01)</td>
<td align="left">0.720 (1.149e-02)</td>
<td align="left">2.303 (6.375e-04)</td>
</tr>
<tr>
<td colspan="2" align="left">LSTM</td>
<td align="left">1.119 (1.034e-04)</td>
<td align="left">1.134 (4.835e-03)</td>
<td align="left">1.109 (1.945e-03)</td>
<td align="left">1.153 (4.573e-04)</td>
</tr>
<tr>
<td colspan="2" align="left">BiLSTM</td>
<td align="left">1.119 (1.471e-04)</td>
<td align="left">1.136 (5.407e-03)</td>
<td align="left">1.101 (2.720e-04)</td>
<td align="left">1.153 (2.171e-04)</td>
</tr>
<tr>
<td colspan="2" align="left">Transformer</td>
<td align="left">1.120 (2.358e-02)</td>
<td align="left">1.118 (9.446e-02)</td>
<td align="left">0.182 (1.320e-02)</td>
<td align="left">1.894 (2.225e-02)</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>
<xref ref-type="table" rid="T3">Table 3</xref> summarizes the results obtained from the real data analysis. Similar to the simulation studies, 500 independent runs were performed on different random initialization and the training/test errors have a very similar pattern as in <xref ref-type="table" rid="T2">Table 2</xref>. The performances of all deep learning models applied on LD matrices are similar to those obtained from using individual-level SNP data. On the other hand, both BiLSTM and CNN perform better compared to that of DNN&#x2019;s and CNN performs the best. We hypothesize that SNPs within the <italic>APOE</italic> gene are more spatially correlated compared to the ones generated in the simulation studies. It is a little bit surprise to see that transformers did not perform very well in this case as the test error is significantly larger than other statistical models, which could potentially be due to the small sample size in the ADNI data.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Comparisons between training/test errors of deep learning models on individual-level SNP data and on genetic summary data based on ADNI data. The unit for the response variable (logarithm of hippocampal volume) is the natural logarithm of cubic millimeters.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" colspan="2" align="left">Method</th>
<th colspan="2" align="center">Individual-level SNP data</th>
<th colspan="2" align="center">Genetic summary data</th>
</tr>
<tr>
<th align="center">Training error</th>
<th align="center">Test error</th>
<th align="center">Training error</th>
<th align="center">Test error</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td colspan="2" align="left">DNN</td>
<td align="left">2.188e-02 (1.923e-03)</td>
<td align="left">2.118e-02 (1.118e-03)</td>
<td align="left">3.494e-03 (1.272e-03)</td>
<td align="left">1.694e-02 (2.237e-04)</td>
</tr>
<tr>
<td colspan="2" align="left">CNN</td>
<td align="left"/>
<td align="left"/>
<td align="left"/>
<td align="left"/>
</tr>
<tr>
<td align="left"/>
<td align="left">Structure 1</td>
<td align="left">1.389e-02 (8.903e-4)</td>
<td align="left">1.397e-02 (9.313e-4)</td>
<td align="left">1.430e-02 (3.564e-3)</td>
<td align="left">1.531-e02 (2.412e-04)</td>
</tr>
<tr>
<td align="left"/>
<td align="left">Structure 2</td>
<td align="left">1.389e-02 (4.448e-4)</td>
<td align="left">1.391e-02 (3.936e-4)</td>
<td align="left">1.356e-02 (5.1131e-4)</td>
<td align="left">1.389-e02 (1.863e-04)</td>
</tr>
<tr>
<td colspan="2" align="left">LSTM</td>
<td align="left">2.345e-02 (6.690e-05)</td>
<td align="left">2.063e-02 (2.266e-04)</td>
<td align="left">2.265e-02 (3.367e-04)</td>
<td align="left">2.265e-02 (4.052e-04)</td>
</tr>
<tr>
<td colspan="2" align="left">BiLSTM</td>
<td align="left">2.364e-02 (4.794e-04)</td>
<td align="left">2.059e-02 (4.054e-04)</td>
<td align="left">1.407e-02 (1.537e-04)</td>
<td align="left">1.392e-02 (7.873e-06)</td>
</tr>
<tr>
<td colspan="2" align="left">Transformer</td>
<td align="left">1.532e-02 (4.477e-06)</td>
<td align="left">1.394e-02 (4.304e-06)</td>
<td align="left">2.392e-02 (2.663e-03)</td>
<td align="left">5.495e-02 (2.959e-03)</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>As a comparison, we also applied the best linear unbiased predictor (BLUP) from linear mixed-effects models to predict the traits based on the simulated individual-level SNP data, which serves as a benchmark. Based on 500 independent runs, for the simulated data, the mean and standard deviation of the training error of BLUP were 0.849 and 3.499e-01, respectively, and the mean and standard deviation of the test error were 1.087 and 9.984e-02, respectively. Although BLUP appears to outperform all the deep learning models due to its smaller test error, it should be noted that the underlying relationship between the SNPs and the trait is linear. Therefore, BLUP is expected to be the best predictor in terms of mean squared error.</p>
</sec>
<sec id="s4">
<label>4</label>
<title>Discussions and conclusion</title>
<p>Deep learning methods have achieved significant success in genetic and genomic predictive analyses. However, their application to genetic summary data has not been fully explored. In this paper, we propose an approach for training and evaluating deep learning models using genetic summary data as inputs, with the test error approximated through the bootstrap method. Through simulation studies and real data analyses, we find that deep learning methods based on LD matrices can achieve prediction accuracies comparable to those obtained using individual-level data. This finding broadens the potential applications of deep learning methods to genetic risk prediction based on summary data.</p>
<p>In practice, the performance of deep learning methods heavily depends on the choice of hyperparameters (such as the number of hidden units and hidden layers in a DNN) as well as the learning algorithms. To determine these hyperparameters, we created a pool of deep learning models with varying configurations, selecting those with the best validation prediction accuracy. Developing effective strategies for choosing hyperparameters to ensure optimal predictive performance will be a focus of our future work. Additionally, while CNNs can capture local information, it is noteworthy that their performances vary a lot in the simulation studies and the real data analyses. We conjecture that the performance of CNNs relies heavily on the noises and spatial correlations of SNPs in the genetic data, which could potentially limit their feature extraction capabilities. Further analyses on different simulated and real genetic datasets are needed to verify this conjecture, and this will be another work of our future research. On the other hand, the performances of LSTM or BiLSTM are more consistent and can produce slightly better test error compared to DNNs, which may suggest these models are more suitable for genetic and genomic applications.</p>
<p>Computational costs and limited sample sizes are two major bottlenecks of the proposed method. To enable a deep learning model to flexibly capture complex relationships, larger network architectures and greater sample sizes are preferred. However, due to limited sample sizes and computational resources, we were only able to test deep learning models with relatively small architectures. Even with such small structures, it remains infeasible to use a genome-wide LD matrix as input. In the paper, we proposed to use nonoverlap block matrices along the diagonal to address the issue. To avoid further information loss, a potential solution could be using overlapped block matrices along the diagonal. However, having more input LD matrices will results in more parameters in deep learning models to train. Therefore, how to keep the balance between reducing potential information loss and how to develop strategies to efficiently handle genome-wide LD matrices as input will be one of our future research directions. Furthermore, to address the limited sample size problem, one potential approach is to use AI-based tools [e.g., TabDDPM (<xref ref-type="bibr" rid="B24">Kotelnikov et al., 2024</xref>)] to generate synthetic tabular data. Nevertheless, such approaches require rigorous validation before they can be widely applied.</p>
<p>While Alzheimer&#x2019;s disease is a polygenic disorder involving numerous loci across the genome, our real data analysis focused on SNPs adjacent to the APOE gene. This region was chosen because of its well-established and strong association with AD, as well as to reduce the computational burden of training deep learning models on genome-wide data. As a result, the analysis serves primarily as a proof-of-concept, demonstrating the model&#x2019;s ability to capture nonlinear SNP&#x2013;phenotype relationships within a biologically relevant region. Future work will extend this approach to genome-wide analyses, which will provide a more comprehensive assessment of the model&#x2019;s predictive ability for polygenic traits. Such extensions will also enable a direct comparison of predictive performance and computational efficiency with polygenic risk scores and other nonlinear models, including kernel-based approaches.</p>
<p>Although the focus of this paper is to compare the performance of deep learning models when only LD matrices are available as inputs with that when individual-level SNP data are available, we would like to briefly discuss the difference between our approach and PRS-based methods, since PRS also relies on genetic summary data. In general, a polygenic risk score for a disease is obtained by aggregating the effects of SNPs across the genome, where the effect sizes of individual SNPs are estimated from genome-wide association studies (GWAS). Therefore, an implicit assumption underlying PRS is that SNP effects are additive and linear, which prevents PRS from capturing nonlinear genetic effects or SNP&#x2013;SNP interactions (<xref ref-type="bibr" rid="B9">Elgart et al., 2022</xref>). In contrast, the proposed deep learning framework learns nonlinear, high-dimensional mappings from SNPs or LD matrices to the phenotype directly, enabling it to capture complex genetic architectures such as SNP&#x2013;SNP interactions, local LD structure, and potential non-additive effects that PRS cannot model.</p>
<p>Recently, there has been extensive research aimed at opening the black box and making deep learning models more interpretable. According to the survey by <xref ref-type="bibr" rid="B35">R&#xe4;uker et al. (2023)</xref>, interpretability techniques can be classified into intrinsic and post hoc approaches. Intrinsic techniques involve training models that are inherently more interpretable or possess natural explanations, whereas post hoc methods aim to interpret a model after it has been trained. From a statistical perspective, one post hoc approach to understand a deep learning model is to use the trained model for statistical inference, such as hypothesis testing or variable selection. Many recent studies have explored such possibilities. For instance, multiple hypothesis testing procedures based on neural networks have been proposed in recent years, with some applied to detecting significant disease-related genes (<xref ref-type="bibr" rid="B17">Horel and Giesecke, 2020</xref>; <xref ref-type="bibr" rid="B40">Shen et al., 2021</xref>; <xref ref-type="bibr" rid="B41">Shen et al., 2022a</xref>; <xref ref-type="bibr" rid="B39">Shen and Wang, 2024</xref>; <xref ref-type="bibr" rid="B7">Dai et al., 2024</xref>). Although this paper mainly focuses on the predictive performance of deep learning models, the explainability of the proposed methods is also an important topic and will be considered in future research.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec sec-type="ethics-statement" id="s6">
<title>Ethics statement</title>
<p>Ethical approval was not required for the study involving humans in accordance with the local legislation and institutional requirements. Written informed consent to participate in this study was not required from the participants or the participants&#x2019; legal guardians/next of kin in accordance with the national legislation and the institutional requirements.</p>
</sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>AW: Writing &#x2013; review and editing, Formal Analysis, Writing &#x2013; original draft, Software. EX: Software, Writing &#x2013; review and editing, Writing &#x2013; original draft, Formal Analysis. JC: Writing &#x2013; original draft, Software, Formal Analysis, Writing &#x2013; review and editing. XS: Methodology, Conceptualization, Writing &#x2013; review and editing, Writing &#x2013; original draft, Supervision.</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>The author(s) declared that this work was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s10">
<title>Generative AI statement</title>
<p>The author(s) declared that generative AI was used in the creation of this manuscript. ChatGPT 4o was used to correct grammatical mistakes.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<fn-group>
<fn fn-type="custom" custom-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/498015/overview">Chaoyang Zhang</ext-link>, University of Southern Mississippi, United States</p>
</fn>
<fn fn-type="custom" custom-type="reviewed-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1462719/overview">Jung Hae-un</ext-link>, Kyung Hee University, Republic of Korea</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3194124/overview">Aleksandar Ilic</ext-link>, University of Lisbon, Portugal</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3209499/overview">Johan Zvrskovec</ext-link>, King&#x2019;s College London, United Kingdom</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bai</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Dallasega</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Orzes</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Sarkis</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Industry 4.0 technologies assessment: a sustainability perspective</article-title>. <source>Int. J. Prod. Econ.</source> <volume>229</volume>, <fpage>107776</fpage>. <pub-id pub-id-type="doi">10.1016/j.ijpe.2020.107776</pub-id>
</mixed-citation>
</ref>
<ref id="B2">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bengio</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Simard</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Frasconi</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>1994</year>). <article-title>Learning long-term dependencies with gradient descent is difficult</article-title>. <source>IEEE Trans. Neural Netw.</source> <volume>5</volume>, <fpage>157</fpage>&#x2013;<lpage>166</lpage>. <pub-id pub-id-type="doi">10.1109/72.279181</pub-id>
<pub-id pub-id-type="pmid">18267787</pub-id>
</mixed-citation>
</ref>
<ref id="B3">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Campos</surname>
<given-names>G. de los</given-names>
</name>
<name>
<surname>Vazquez</surname>
<given-names>A. I.</given-names>
</name>
<name>
<surname>Fernando</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Klimentidis</surname>
<given-names>Y. C.</given-names>
</name>
<name>
<surname>Sorensen</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Prediction of complex human traits using the genomic best linear unbiased predictor</article-title>. <source>PLOS Genet.</source> <volume>9</volume>, <fpage>e1003608</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pgen.1003608</pub-id>
<pub-id pub-id-type="pmid">23874214</pub-id>
</mixed-citation>
</ref>
<ref id="B4">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Collins</surname>
<given-names>F. S.</given-names>
</name>
<name>
<surname>Varmus</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>A new initiative on precision medicine</article-title>. <source>N. Engl. J. Med.</source> <volume>372</volume>, <fpage>793</fpage>&#x2013;<lpage>795</lpage>. <pub-id pub-id-type="doi">10.1056/NEJMp1500523</pub-id>
<pub-id pub-id-type="pmid">25635347</pub-id>
</mixed-citation>
</ref>
<ref id="B5">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Consortium</surname>
<given-names>W. T. C. C.</given-names>
</name>
<name>
<surname>Clayton</surname>
<given-names>D. G.</given-names>
</name>
<name>
<surname>Cardon</surname>
<given-names>L. R.</given-names>
</name>
<name>
<surname>Craddock</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Deloukas</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Duncanson</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2007</year>). <article-title>Genome-wide association study of 14,000 cases of seven common diseases and 3,000 shared controls</article-title>. <source>Nature</source> <volume>447</volume>, <fpage>661</fpage>&#x2013;<lpage>678</lpage>. <pub-id pub-id-type="doi">10.1038/nature05911</pub-id>
<pub-id pub-id-type="pmid">17554300</pub-id>
</mixed-citation>
</ref>
<ref id="B6">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cybenko</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>1989</year>). <article-title>Approximation by superpositions of a sigmoidal function</article-title>. <source>Math. Control Signal Syst.</source> <volume>2</volume>, <fpage>303</fpage>&#x2013;<lpage>314</lpage>. <pub-id pub-id-type="doi">10.1007/BF02551274</pub-id>
</mixed-citation>
</ref>
<ref id="B7">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dai</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Significance tests of feature relevance for a black-box learner</article-title>. <source>IEEE Trans. Neural Netw. Learn. Syst.</source> <volume>35</volume>, <fpage>1898</fpage>&#x2013;<lpage>1911</lpage>. <pub-id pub-id-type="doi">10.1109/TNNLS.2022.3185742</pub-id>
<pub-id pub-id-type="pmid">35771783</pub-id>
</mixed-citation>
</ref>
<ref id="B8">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Efron</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Tibshirani</surname>
<given-names>R. J.</given-names>
</name>
</person-group> (<year>1994</year>). <source>An introduction to the bootstrap</source>. <publisher-loc>New York</publisher-loc>: <publisher-name>Chapman and Hall/CRC</publisher-name>. <pub-id pub-id-type="doi">10.1201/9780429246593</pub-id>
</mixed-citation>
</ref>
<ref id="B9">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Elgart</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lyons</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Romero-Brufau</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kurniansyah</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Brody</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Non-linear machine learning models incorporating SNPs and PRS improve polygenic prediction in diverse human populations</article-title>. <source>Commun. Biol.</source> <volume>5</volume>, <fpage>856</fpage>. <pub-id pub-id-type="doi">10.1038/s42003-022-03812-z</pub-id>
<pub-id pub-id-type="pmid">35995843</pub-id>
</mixed-citation>
</ref>
<ref id="B10">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gabriel</surname>
<given-names>S. B.</given-names>
</name>
<name>
<surname>Schaffner</surname>
<given-names>S. F.</given-names>
</name>
<name>
<surname>Nguyen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Moore</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Roy</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Blumenstiel</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2002</year>). <article-title>The structure of haplotype blocks in the human genome</article-title>. <source>Science</source> <volume>296</volume>, <fpage>2225</fpage>&#x2013;<lpage>2229</lpage>. <pub-id pub-id-type="doi">10.1126/science.1069424</pub-id>
<pub-id pub-id-type="pmid">12029063</pub-id>
</mixed-citation>
</ref>
<ref id="B11">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gra&#xe7;a</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Nobre</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Sousa</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Ilic</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Distributed transformer for high order epistasis detection in large-scale datasets</article-title>. <source>Sci. Rep.</source> <volume>14</volume>, <fpage>14579</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-024-65317-5</pub-id>
<pub-id pub-id-type="pmid">38918413</pub-id>
</mixed-citation>
</ref>
<ref id="B12">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Graves</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Schmidhuber</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Framewise phoneme classification with bidirectional LSTM and other neural network architectures</article-title>. <source>Neural Netw. IJCNN</source> <volume>18</volume>, <fpage>602</fpage>&#x2013;<lpage>610</lpage>. <pub-id pub-id-type="doi">10.1016/j.neunet.2005.06.042</pub-id>
<pub-id pub-id-type="pmid">16112549</pub-id>
</mixed-citation>
</ref>
<ref id="B13">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guo</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Powerful and efficient SNP-Set association tests across multiple phenotypes using GWAS summary data</article-title>. <source>Bioinformatics</source> <volume>35</volume>, <fpage>1366</fpage>&#x2013;<lpage>1372</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty811</pub-id>
<pub-id pub-id-type="pmid">30239606</pub-id>
</mixed-citation>
</ref>
<ref id="B14">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hibar</surname>
<given-names>D. P.</given-names>
</name>
<name>
<surname>Stein</surname>
<given-names>J. L.</given-names>
</name>
<name>
<surname>Jahanshad</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Kohannim</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Hua</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Toga</surname>
<given-names>A. W.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Genome-wide interaction analysis reveals replicated epistatic effects on brain structure</article-title>. <source>Neurobiol. Aging</source> <volume>36</volume>, <fpage>S151</fpage>&#x2013;<lpage>S158</lpage>. <pub-id pub-id-type="doi">10.1016/j.neurobiolaging.2014.02.033</pub-id>
<pub-id pub-id-type="pmid">25264344</pub-id>
</mixed-citation>
</ref>
<ref id="B15">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hochreiter</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Schmidhuber</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>1997</year>). <article-title>Long short-term memory</article-title>. <source>Neural Comput.</source> <volume>9</volume>, <fpage>1735</fpage>&#x2013;<lpage>1780</lpage>. <pub-id pub-id-type="doi">10.1162/neco.1997.9.8.1735</pub-id>
<pub-id pub-id-type="pmid">9377276</pub-id>
</mixed-citation>
</ref>
<ref id="B16">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hofmann</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Sch&#xf6;lkopf</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Smola</surname>
<given-names>A. J.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Kernel methods in machine learning</article-title>. <source>Ann. Statistics</source> <volume>36</volume>, <fpage>1171</fpage>&#x2013;<lpage>1220</lpage>. <pub-id pub-id-type="doi">10.1214/009053607000000677</pub-id>
</mixed-citation>
</ref>
<ref id="B17">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Horel</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Giesecke</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Significance tests for neural networks</article-title>. <source>J. Mach. Learn. Res.</source> <volume>21</volume>, <fpage>1</fpage>&#x2013;<lpage>29</lpage>.<pub-id pub-id-type="pmid">34305477</pub-id>
</mixed-citation>
</ref>
<ref id="B18">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hornik</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Stinchcombe</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>White</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>1989</year>). <article-title>Multilayer feedforward networks are universal approximators</article-title>. <source>Neural Networks</source> <volume>2</volume>, <fpage>359</fpage>&#x2013;<lpage>366</lpage>. <pub-id pub-id-type="doi">10.1016/0893-6080(89)90020-8</pub-id>
</mixed-citation>
</ref>
<ref id="B19">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hostage</surname>
<given-names>C. A.</given-names>
</name>
<name>
<surname>Choudhury</surname>
<given-names>K. R.</given-names>
</name>
<name>
<surname>Doraiswamy</surname>
<given-names>P. M.</given-names>
</name>
<name>
<surname>Petrella</surname>
<given-names>J. R.</given-names>
</name>
<name>
<surname>Initiative</surname>
<given-names>for the A. D. N.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Dissecting the gene dose-effects of the APOE &#x3b5;4 and &#x3b5;2 alleles on hippocampal volumes in aging and alzheimer&#x2019;s disease</article-title>. <source>PLOS ONE</source> <volume>8</volume>, <fpage>e54483</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0054483</pub-id>
<pub-id pub-id-type="pmid">23405083</pub-id>
</mixed-citation>
</ref>
<ref id="B20">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>James</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Witten</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Hastie</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Tibshirani</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2013</year>). <source>An introduction to statistical learning, springer texts in statistics</source>. <publisher-loc>New York, NY</publisher-loc>: <publisher-name>Springer</publisher-name>. <pub-id pub-id-type="doi">10.1007/978-1-4614-7138-7</pub-id>
</mixed-citation>
</ref>
<ref id="B21">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jostins</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Barrett</surname>
<given-names>J. C.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Genetic risk prediction in complex disease</article-title>. <source>Hum. Mol. Genet.</source> <volume>20</volume>, <fpage>R182</fpage>&#x2013;<lpage>R188</lpage>. <pub-id pub-id-type="doi">10.1093/hmg/ddr378</pub-id>
<pub-id pub-id-type="pmid">21873261</pub-id>
</mixed-citation>
</ref>
<ref id="B22">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Karch</surname>
<given-names>C. M.</given-names>
</name>
<name>
<surname>Cruchaga</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Goate</surname>
<given-names>A. M.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Alzheimer&#x2019;s disease genetics: from the bench to the clinic</article-title>. <source>Neuron</source> <volume>83</volume>, <fpage>11</fpage>&#x2013;<lpage>26</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuron.2014.05.041</pub-id>
<pub-id pub-id-type="pmid">24991952</pub-id>
</mixed-citation>
</ref>
<ref id="B23">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kingma</surname>
<given-names>D. P.</given-names>
</name>
<name>
<surname>Ba</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Adam: a method for stochastic optimization</article-title>. <pub-id pub-id-type="doi">10.48550/arXiv.1412.6980</pub-id>
</mixed-citation>
</ref>
<ref id="B24">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kotelnikov</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Baranchuk</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Rubachev</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Babenko</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>TabDDPM: modelling tabular data with diffusion models</article-title>. <pub-id pub-id-type="doi">10.5555/3618408.3619133</pub-id>
</mixed-citation>
</ref>
<ref id="B25">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kwak</surname>
<given-names>I.-Y.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Adaptive gene- and pathway-trait association testing with GWAS summary statistics</article-title>. <source>Bioinformatics</source> <volume>32</volume>, <fpage>1178</fpage>&#x2013;<lpage>1184</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btv719</pub-id>
<pub-id pub-id-type="pmid">26656570</pub-id>
</mixed-citation>
</ref>
<ref id="B26">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>LeCun</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>1989</year>). <source>Generalization and network design strategies</source>. In Editor <person-group person-group-type="editor">
<name>
<surname>Pfeifer</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Schreter</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Fogelman</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Steels</surname>
<given-names>L.</given-names>
</name>
</person-group> (<publisher-name>Connectionism in perspective Elsevier</publisher-name>).</mixed-citation>
</ref>
<ref id="B27">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Elston</surname>
<given-names>R. C.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>A generalized genetic random field method for the genetic association analysis of sequencing data</article-title>. <source>Genet. Epidemiol.</source> <volume>38</volume>, <fpage>242</fpage>&#x2013;<lpage>253</lpage>. <pub-id pub-id-type="doi">10.1002/gepi.21790</pub-id>
<pub-id pub-id-type="pmid">24482034</pub-id>
</mixed-citation>
</ref>
<ref id="B28">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A survey of convolutional neural networks: analysis, applications, and prospects</article-title>. <source>IEEE Trans. Neural Netw. Learn. Syst.</source> <volume>33</volume>, <fpage>6999</fpage>&#x2013;<lpage>7019</lpage>. <pub-id pub-id-type="doi">10.1109/TNNLS.2021.3084827</pub-id>
<pub-id pub-id-type="pmid">34111009</pub-id>
</mixed-citation>
</ref>
<ref id="B29">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Mazumder</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Accurate and efficient estimation of local heritability using summary statistics and the linkage disequilibrium matrix</article-title>. <source>Nat. Commun.</source> <volume>14</volume>, <fpage>7954</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-023-43565-9</pub-id>
<pub-id pub-id-type="pmid">38040712</pub-id>
</mixed-citation>
</ref>
<ref id="B30">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Arora</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Attaoua</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Hamet</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Tremblay</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Bihlo</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>Leveraging hierarchical structures for genetic block interaction studies using the hierarchical transformer</article-title>. <source>medRxiv.</source> <volume>11</volume>, <fpage>24317486</fpage>. <pub-id pub-id-type="doi">10.1101/2024.11.18.24317486</pub-id>
<pub-id pub-id-type="pmid">39606365</pub-id>
</mixed-citation>
</ref>
<ref id="B31">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Meng</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Weng</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Wen</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Explainable deep transfer learning model for disease risk prediction using high-dimensional genomic data</article-title>. <source>PLOS Comput. Biol.</source> <volume>18</volume>, <fpage>e1010328</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1010328</pub-id>
<pub-id pub-id-type="pmid">35839250</pub-id>
</mixed-citation>
</ref>
<ref id="B32">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mather</surname>
<given-names>K. A.</given-names>
</name>
<name>
<surname>Armstrong</surname>
<given-names>N. J.</given-names>
</name>
<name>
<surname>Wen</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Kwok</surname>
<given-names>J. B.</given-names>
</name>
<name>
<surname>Assareh</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Thalamuthu</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Investigating the genetics of hippocampal volume in older adults without dementia</article-title>. <source>PLOS ONE</source> <volume>10</volume>, <fpage>e0116920</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0116920</pub-id>
<pub-id pub-id-type="pmid">25625606</pub-id>
</mixed-citation>
</ref>
<ref id="B33">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Gage</surname>
<given-names>F. H.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Adult hippocampal neurogenesis and its role in Alzheimer&#x2019;s disease</article-title>. <source>Mol. Neurodegener.</source> <volume>6</volume>, <fpage>85</fpage>. <pub-id pub-id-type="doi">10.1186/1750-1326-6-85</pub-id>
<pub-id pub-id-type="pmid">22192775</pub-id>
</mixed-citation>
</ref>
<ref id="B34">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nazarian</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Cook</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Morado</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kulminski</surname>
<given-names>A. M.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Interaction analysis reveals complex genetic associations with alzheimer&#x2019;s disease in the CLU and ABCA7 gene regions</article-title>. <source>Genes</source> <volume>14</volume>, <fpage>1666</fpage>. <pub-id pub-id-type="doi">10.3390/genes14091666</pub-id>
<pub-id pub-id-type="pmid">37761806</pub-id>
</mixed-citation>
</ref>
<ref id="B35">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>R&#xe4;uker</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Ho</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Casper</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hadfield-Menell</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Toward transparent AI: a survey on interpreting the inner structures of deep neural networks</article-title>,&#x201d; in <conf-name>Presented at the 2023 IEEE conference on secure and trustworthy machine learning (SaTML)</conf-name>. <publisher-name>IEEE Computer Society</publisher-name>, <fpage>464</fpage>&#x2013;<lpage>483</lpage>. <pub-id pub-id-type="doi">10.1109/SaTML54575.2023.00039</pub-id>
</mixed-citation>
</ref>
<ref id="B36">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rosenblatt</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>1958</year>). <article-title>The perceptron: a probabilistic model for information storage and organization in the brain</article-title>. <source>Psychol. Review</source> <volume>65</volume>, <fpage>386</fpage>&#x2013;<lpage>408</lpage>. <pub-id pub-id-type="doi">10.1037/h0042519</pub-id>
<pub-id pub-id-type="pmid">13602029</pub-id>
</mixed-citation>
</ref>
<ref id="B37">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schuff</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Woerner</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Boreta</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Kornfield</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Shaw</surname>
<given-names>L. M.</given-names>
</name>
<name>
<surname>Trojanowski</surname>
<given-names>J. Q.</given-names>
</name>
<etal/>
</person-group> (<year>2009</year>). <article-title>MRI of hippocampal volume loss in early Alzheimer&#x2019;s disease in relation to ApoE genotype and biomarkers</article-title>. <source>Brain</source> <volume>132</volume>, <fpage>1067</fpage>&#x2013;<lpage>1077</lpage>. <pub-id pub-id-type="doi">10.1093/brain/awp007</pub-id>
<pub-id pub-id-type="pmid">19251758</pub-id>
</mixed-citation>
</ref>
<ref id="B38">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Scott</surname>
<given-names>L. J.</given-names>
</name>
<name>
<surname>Mohlke</surname>
<given-names>K. L.</given-names>
</name>
<name>
<surname>Bonnycastle</surname>
<given-names>L. L.</given-names>
</name>
<name>
<surname>Willer</surname>
<given-names>C. J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Duren</surname>
<given-names>W. L.</given-names>
</name>
<etal/>
</person-group> (<year>2007</year>). <article-title>A genome-wide association study of type 2 diabetes in finns detects multiple susceptibility variants</article-title>. <source>Science</source> <volume>316</volume>, <fpage>1341</fpage>&#x2013;<lpage>1345</lpage>. <pub-id pub-id-type="doi">10.1126/science.1142382</pub-id>
<pub-id pub-id-type="pmid">17463248</pub-id>
</mixed-citation>
</ref>
<ref id="B39">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>An exploration of testing genetic associations using goodness-of-fit statistics based on deep ReLU neural networks</article-title>. <source>Front. Syst. Biol.</source> <volume>4</volume>, <fpage>1460369</fpage>. <pub-id pub-id-type="doi">10.3389/fsysb.2024.1460369</pub-id>
<pub-id pub-id-type="pmid">40809144</pub-id>
</mixed-citation>
</ref>
<ref id="B40">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Sakhanenko</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A goodness-of-fit test based on neural network sieve estimators</article-title>. <source>Statistics and Probability Letters</source> <volume>174</volume>, <fpage>109100</fpage>. <pub-id pub-id-type="doi">10.1016/j.spl.2021.109100</pub-id>
<pub-id pub-id-type="pmid">35665309</pub-id>
</mixed-citation>
</ref>
<ref id="B41">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Sakhanenko</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2022a</year>). <article-title>A sieve quasi-likelihood ratio test for neural networks with applications to genetic association studies</article-title>. <pub-id pub-id-type="doi">10.48550/arXiv.2212.08255</pub-id>
</mixed-citation>
</ref>
<ref id="B42">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2022b</year>). <article-title>A brief review on deep learning applications in genomic studies</article-title>. <source>Front. Syst. Biol.</source> <volume>2</volume>, <fpage>877717</fpage>. <pub-id pub-id-type="doi">10.3389/fsysb.2022.877717</pub-id>
</mixed-citation>
</ref>
<ref id="B43">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Cui</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2022c</year>). <article-title>A conditional autoregressive model for genetic association analysis accounting for genetic heterogeneity</article-title>. <source>Stat. Med.</source> <volume>41</volume>, <fpage>517</fpage>&#x2013;<lpage>542</lpage>. <pub-id pub-id-type="doi">10.1002/sim.9257</pub-id>
<pub-id pub-id-type="pmid">34811777</pub-id>
</mixed-citation>
</ref>
<ref id="B44">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Siami-Namini</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Tavakoli</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Namin</surname>
<given-names>A. S.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>The performance of LSTM and BiLSTM in forecasting time series</article-title>,&#x201d; in <source>2019 IEEE international conference on big data (big data). Presented at the 2019 IEEE international conference on big data (big data)</source>, <fpage>3285</fpage>&#x2013;<lpage>3292</lpage>. <pub-id pub-id-type="doi">10.1109/BigData47090.2019.9005997</pub-id>
</mixed-citation>
</ref>
<ref id="B45">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sims</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Hill</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Williams</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>The multiplex model of the genetics of Alzheimer&#x2019;s disease</article-title>. <source>Nat. Neurosci.</source> <volume>23</volume>, <fpage>311</fpage>&#x2013;<lpage>322</lpage>. <pub-id pub-id-type="doi">10.1038/s41593-020-0599-5</pub-id>
<pub-id pub-id-type="pmid">32112059</pub-id>
</mixed-citation>
</ref>
<ref id="B46">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sladek</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Rocheleau</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Rung</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Dina</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Serre</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2007</year>). <article-title>A genome-wide association study identifies novel risk loci for type 2 diabetes</article-title>. <source>Nature</source> <volume>445</volume>, <fpage>881</fpage>&#x2013;<lpage>885</lpage>. <pub-id pub-id-type="doi">10.1038/nature05616</pub-id>
<pub-id pub-id-type="pmid">17293876</pub-id>
</mixed-citation>
</ref>
<ref id="B47">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Speed</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Balding</surname>
<given-names>D. J.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>MultiBLUP: improved SNP-Based prediction for complex traits</article-title>. <source>Genome Res.</source> <volume>24</volume>, <fpage>1550</fpage>&#x2013;<lpage>1557</lpage>. <pub-id pub-id-type="doi">10.1101/gr.169375.113</pub-id>
<pub-id pub-id-type="pmid">24963154</pub-id>
</mixed-citation>
</ref>
<ref id="B48">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Speed</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Balding</surname>
<given-names>D. J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>SumHer better estimates the SNP heritability of complex traits from summary statistics</article-title>. <source>Nat. Genet.</source> <volume>51</volume>, <fpage>277</fpage>&#x2013;<lpage>284</lpage>. <pub-id pub-id-type="doi">10.1038/s41588-018-0279-5</pub-id>
<pub-id pub-id-type="pmid">30510236</pub-id>
</mixed-citation>
</ref>
<ref id="B49">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Speed</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Holmes</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Balding</surname>
<given-names>D. J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Evaluating and improving heritability models using summary statistics</article-title>. <source>Nat. Genet.</source> <volume>52</volume>, <fpage>458</fpage>&#x2013;<lpage>462</lpage>. <pub-id pub-id-type="doi">10.1038/s41588-020-0600-y</pub-id>
<pub-id pub-id-type="pmid">32203469</pub-id>
</mixed-citation>
</ref>
<ref id="B50">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sullivan</surname>
<given-names>E. V.</given-names>
</name>
<name>
<surname>Pfefferbaum</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Swan</surname>
<given-names>G. E.</given-names>
</name>
<name>
<surname>Carmelli</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Heritability of hippocampal size in elderly twin men: equivalent influence from genes and environment</article-title>. <source>Hippocampus</source> <volume>11</volume>, <fpage>754</fpage>&#x2013;<lpage>762</lpage>. <pub-id pub-id-type="doi">10.1002/hipo.1091</pub-id>
<pub-id pub-id-type="pmid">11811670</pub-id>
</mixed-citation>
</ref>
<ref id="B51">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Svishcheva</surname>
<given-names>G. R.</given-names>
</name>
<name>
<surname>Belonogova</surname>
<given-names>N. M.</given-names>
</name>
<name>
<surname>Zorkoltseva</surname>
<given-names>I. V.</given-names>
</name>
<name>
<surname>Kirichenko</surname>
<given-names>A. V.</given-names>
</name>
<name>
<surname>Axenovich</surname>
<given-names>T. I.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Gene-based association tests using GWAS summary statistics</article-title>. <source>Bioinformatics</source> <volume>35</volume>, <fpage>3701</fpage>&#x2013;<lpage>3708</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btz172</pub-id>
<pub-id pub-id-type="pmid">30860568</pub-id>
</mixed-citation>
</ref>
<ref id="B52">
<mixed-citation publication-type="journal">
<collab>The 1000 Genomes Project Consortium</collab>
<person-group person-group-type="author">
<name>
<surname>Abecasis</surname>
<given-names>G. R.</given-names>
</name>
<name>
<surname>Altshuler</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Auton</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Brooks</surname>
<given-names>L. D.</given-names>
</name>
<name>
<surname>Durbin</surname>
<given-names>R. M.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>A map of human genome variation from population scale sequencing</article-title>. <source>Nature</source> <volume>467</volume>, <fpage>1061</fpage>&#x2013;<lpage>1073</lpage>. <pub-id pub-id-type="doi">10.1038/nature09534</pub-id>
<pub-id pub-id-type="pmid">20981092</pub-id>
</mixed-citation>
</ref>
<ref id="B53">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Vaswani</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Shazeer</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Parmar</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Uszkoreit</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jones</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Gomez</surname>
<given-names>A. N.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). &#x201c;<article-title>Attention is all you need</article-title>,&#x201d; in <conf-name>Advances in Neural Information Processing Systems 30: Annual Conferenceon Neural Information Processing Systems 2017</conf-name>, <conf-loc>Long Beach, CA</conf-loc>, <conf-date>December 4-9, 2017</conf-date>. Editor <person-group person-group-type="editor">
<name>
<surname>Guyon</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>von Luxburg</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Bengio</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wallach</surname>
<given-names>H. M.</given-names>
</name>
<name>
<surname>Fergus</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Vishwanathan</surname>
<given-names>S. V. N.</given-names>
</name>
<name>
<surname>Garnett</surname>
<given-names>R.</given-names>
</name>
</person-group>, <fpage>5998</fpage>&#x2013;<lpage>6008</lpage>.</mixed-citation>
</ref>
<ref id="B54">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>M. C.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Boehnke</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Rare-variant association testing for sequencing data with the sequence kernel association test</article-title>. <source>Am. J. Hum. Genet.</source> <volume>89</volume>, <fpage>82</fpage>&#x2013;<lpage>93</lpage>. <pub-id pub-id-type="doi">10.1016/j.ajhg.2011.05.029</pub-id>
<pub-id pub-id-type="pmid">21737059</pub-id>
</mixed-citation>
</ref>
<ref id="B55">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xue</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Inferring causal direction between two traits in the presence of horizontal pleiotropy with GWAS summary data</article-title>. <source>PLOS Genet.</source> <volume>16</volume>, <fpage>e1009105</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pgen.1009105</pub-id>
<pub-id pub-id-type="pmid">33137120</pub-id>
</mixed-citation>
</ref>
<ref id="B56">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Benyamin</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>McEvoy</surname>
<given-names>B. P.</given-names>
</name>
<name>
<surname>Gordon</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Henders</surname>
<given-names>A. K.</given-names>
</name>
<name>
<surname>Nyholt</surname>
<given-names>D. R.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>Common SNPs explain a large proportion of the heritability for human height</article-title>. <source>Nat. Genet.</source> <volume>42</volume>, <fpage>565</fpage>&#x2013;<lpage>569</lpage>. <pub-id pub-id-type="doi">10.1038/ng.608</pub-id>
<pub-id pub-id-type="pmid">20562875</pub-id>
</mixed-citation>
</ref>
<ref id="B57">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yarotsky</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Error bounds for approximations with deep ReLU networks</article-title>. <source>Neural Netw.</source> <volume>94</volume>, <fpage>103</fpage>&#x2013;<lpage>114</lpage>. <pub-id pub-id-type="doi">10.1016/j.neunet.2017.07.002</pub-id>
<pub-id pub-id-type="pmid">28756334</pub-id>
</mixed-citation>
</ref>
<ref id="B58">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Yarotsky</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zhevnerchuk</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>The phase diagram of approximation rates for deep neural networks,&#x201d; in Proceedings of the 34th International Conference on Neural Information Processing Systems (NIPS '20)</article-title> (<publisher-loc>Red Hook, NY</publisher-loc>: <publisher-name>Curran Associates, Inc.</publisher-name>), <fpage>13005</fpage>&#x2013;<lpage>13015</lpage>.</mixed-citation>
</ref>
<ref id="B59">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Yu</given-names>
</name>
<name>
<surname>Ip</surname>
<given-names>F. C. F.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Lv</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Deep learning-based polygenic risk analysis for Alzheimer&#x2019;s disease prediction</article-title>. <source>Commun. Med.</source> <volume>3</volume>, <fpage>1</fpage>&#x2013;<lpage>20</lpage>. <pub-id pub-id-type="doi">10.1038/s43856-023-00269-x</pub-id>
<pub-id pub-id-type="pmid">37024668</pub-id>
</mixed-citation>
</ref>
<ref id="B60">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Trzaskowski</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Maier</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Causal associations between risk factors and common diseases inferred from GWAS summary data</article-title>. <source>Nat. Commun.</source> <volume>9</volume>, <fpage>224</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-017-02317-2</pub-id>
<pub-id pub-id-type="pmid">29335400</pub-id>
</mixed-citation>
</ref>
</ref-list>
</back>
</article>