<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1617504</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2025.1617504</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Mining for gene-environment and gene-gene interactions: parametric and non-parametric tests for detecting variance quantitative trait loci</article-title>
<alt-title alt-title-type="left-running-head">Lin</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fgene.2025.1617504">10.3389/fgene.2025.1617504</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Lin</surname>
<given-names>Wan-Yu</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<xref ref-type="author-notes" rid="fn1">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/33939/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Institute of Health Data Analytics and Statistics, College of Public Health, National Taiwan University</institution>, <addr-line>Taipei</addr-line>, <country>Taiwan</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Master of Public Health Program, College of Public Health, National Taiwan University</institution>, <addr-line>Taipei</addr-line>, <country>Taiwan</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/78720/overview">Ruzong Fan</ext-link>, Georgetown University Medical Center, United States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/658747/overview">Cen Wu</ext-link>, Kansas State University, United States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1535052/overview">Chi Zhang</ext-link>, Yale University, United States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Wan-Yu Lin, <email>linwy@ntu.edu.tw</email>
</corresp>
<fn fn-type="other" id="fn1">
<label>
<sup>&#x2020;</sup>
</label>
<p>ORCID: Wan-Yu Lin, <ext-link ext-link-type="uri" xlink:href="https://orcid.org/0000-0002-3385-4702">orcid.org/0000-0002-3385-4702</ext-link>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>01</day>
<month>09</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1617504</elocation-id>
<history>
<date date-type="received">
<day>24</day>
<month>04</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>04</day>
<month>08</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Lin.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Lin</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Detection of variance quantitative trait loci (vQTL) can facilitate the discovery of gene-environment (GxE) and gene-gene interactions (GxG). Identifying vQTLs before direct GxE and GxG analyses can considerably reduce the number of tests and the multiple-testing penalty.</p>
</sec>
<sec>
<title>Methods</title>
<p>Despite some methods proposed for vQTL detection, few studies have performed a head-to-head comparison simultaneously concerning false positive rates (FPRs), power, and computational time. This work compares three parametric and two non-parametric vQTL tests.</p>
</sec>
<sec>
<title>Results</title>
<p>Simulation studies show that the deviation regression model (DRM) and Kruskal-Wallis test (KW) are the most recommended parametric and non-parametric tests, respectively. The quantile integral linear model (QUAIL, non-parametric) appropriately preserves the FPR under normally or non-normally distributed traits. However, its power is never among the optimal choices, and its computational time is much longer than that of competitors. The Brown-Forsythe test (BF, parametric) can suffer from severe inflation in FPR when SNP&#x2019;s minor allele frequencies &#x3c;0.2. The double generalized linear model (DGLM, parametric) is not valid for non-normally distributed traits, although it is the most powerful method for normally distributed traits.</p>
</sec>
<sec>
<title>Discussion</title>
<p>Considering the robustness (to outliers) and computation time, I chose KW to analyze four lipid traits in the Taiwan Biobank. I further showed that GxE and GxG were enriched among 30 vQTLs identified from the four lipid traits.</p>
</sec>
</abstract>
<kwd-group>
<kwd>cholesterol</kwd>
<kwd>dyslipidemia</kwd>
<kwd>triglyceride</kwd>
<kwd>variance quantitative trait locus</kwd>
<kwd>Taiwan Biobank</kwd>
</kwd-group>
<contract-num rid="cn001">112-2628-B-002-024-MY3</contract-num>
<contract-sponsor id="cn001">National Science and Technology Council<named-content content-type="fundref-id">10.13039/100020595</named-content>
</contract-sponsor>
<counts>
<page-count count="12"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Statistical Genetics and Methodology</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Omitting essential predictors from a regression model can lead to heteroscedasticity (<xref ref-type="bibr" rid="B3">Breusch and Pagan, 1979</xref>). In genetic analyses, loci with unequal phenotypic variances across different genotype groups are called &#x201c;variance quantitative trait loci&#x201d; (vQTLs), which can be caused by omitting gene-environment interaction (GxE) or gene-gene interaction (GxG) in the regression model (<xref ref-type="bibr" rid="B22">Pare et al., 2010</xref>; <xref ref-type="bibr" rid="B26">Ronnegard and Valdar, 2012</xref>; <xref ref-type="bibr" rid="B27">Shi, 2022</xref>; <xref ref-type="bibr" rid="B32">Wang et al., 2019</xref>). For example, Wang et al. performed a genome-wide vQTL analysis of 5.6 million variants on &#x223c;350,000 unrelated individuals of European ancestry for 13 quantitative traits. They identified 75 significant vQTLs for nine traits, especially those related to obesity. Pervasive GxE effects for obesity-related traits were further explored through direct GxE analyses. Wang et al.&#x2018;s study has demonstrated the detection of GxE without environmental data (<xref ref-type="bibr" rid="B32">Wang et al., 2019</xref>).</p>
<p>Methods for detecting vQTLs are categorized as parametric or non-parametric. The development of parametric vQTL tests usually depends on the assumption of normally distributed traits (<xref ref-type="bibr" rid="B18">Marderstein et al., 2021</xref>; <xref ref-type="bibr" rid="B36">Young et al., 2018</xref>). Contrastingly, non-parametric vQTL tests are generally more robust to trait distributions (<xref ref-type="bibr" rid="B21">Miao et al., 2022</xref>).</p>
<p>For every single-nucleotide polymorphism (SNP), we can use the Brown-Forsythe (BF) test (<xref ref-type="bibr" rid="B4">Brown and Forsythe, 1974</xref>) to compute the dispersion within three genotype groups. Compared with the Levene&#x2019;s test (<xref ref-type="bibr" rid="B13">Levene, 1960</xref>), the BF test is more robust to outliers by choosing the median to replace the mean as the center of each genotype group. Moreover, a &#x201c;deviation regression model&#x201d; (DRM) has been proposed to allow continuous predictors such as minor allele dosages (<xref ref-type="bibr" rid="B18">Marderstein et al., 2021</xref>). Therefore, the predictor is not limited to minor allele counts of SNPs (i.e., 0, 1, and 2). Furthermore, a double generalized linear model (DGLM) was also developed to identify loci associated with trait variability and to detect interactions in genome-wide association studies (GWAS) (<xref ref-type="bibr" rid="B25">Ronnegard and Valdar, 2011</xref>; <xref ref-type="bibr" rid="B28">Smyth, 1989</xref>). BF, DRM, and DGLM are all the so-called parametric vQTL methods.</p>
<p>Non-parametric methods for vQTL identification include the Kruskal-Wallis test (KW) (<xref ref-type="bibr" rid="B11">Kruskal and Wallis, 1952</xref>) and a quantile integral linear model (QUAIL) (<xref ref-type="bibr" rid="B21">Miao et al., 2022</xref>). As the non-parametric counterpart of one-way analysis of variance (ANOVA), KW is used to compare the median difference between multiple groups. As long as we first calculate the deviation between trait values and within-group medians, KW can also be applied to test phenotypic homoscedasticity among the three genotype groups. Another non-parametric method, QUAIL, assesses genetic effects on phenotypic variability based on the quantile regression framework. It allows covariate adjustment, non-normally distributed traits, and continuous predictors (i.e., not limited to three genotype categories) (<xref ref-type="bibr" rid="B21">Miao et al., 2022</xref>). However, to integrate information from <italic>K</italic> quantiles (usually <italic>K</italic> &#x3d; 100), QUAIL&#x2019;s computation time is much longer than that of its competitors (BF, DRM, DGLM, and KW). Implementing QUAIL for genome-wide vQTL analysis is computationally challenging.</p>
<p>This work evaluates the performance of the abovementioned parametric or non-parametric methods in detecting vQTLs through Monte Carlo simulations. I performed a head-to-head comparison to assess the false positive rates (FPRs), power, and computation time across several popular vQTL methods. Furthermore, I apply these approaches to four lipid traits in the Taiwan Biobank (TWB) data, including high-density lipoprotein cholesterol (HDL), low-density lipoprotein cholesterol (LDL), triglyceride (TG), and total cholesterol (TCHO). After identifying vQTLs for these four traits, I perform direct GxE and GxG analyses to demonstrate enriched interaction effects among vQTLs.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>2 Materials and methods</title>
<sec id="s2-1">
<title>2.1 Covariate-adjusted residuals</title>
<p>Let <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> be the continuous trait of the <italic>j</italic>th individual, <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> be two indicator variables categorizing three genotype groups of a SNP, and <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mi mathvariant="bold-italic">j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> be the <italic>C</italic>-length covariate vector of individual <italic>j</italic> (assuming <italic>C</italic> covariates should be adjusted). <inline-formula id="inf5">
<mml:math id="m5">
<mml:mrow>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf6">
<mml:math id="m6">
<mml:mrow>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> can be coded as (0, 0), (1, 0), and (0, 1) for 0, 1, and 2 copies of a SNP&#x2019;s specific allele. According to previous studies, the FPR of detecting vQTLs may be inflated in the presence of a large SNP&#x2019;s main effect (<xref ref-type="bibr" rid="B25">Ronnegard and Valdar, 2011</xref>). Therefore, I remove SNP&#x2019;s main effect and covariates&#x2019; effects from <inline-formula id="inf7">
<mml:math id="m7">
<mml:mrow>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> to ensure that the identification of vQTLs will not be confounded by other factors. Specifically, I extract the residuals of regressing <inline-formula id="inf8">
<mml:math id="m8">
<mml:mrow>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> on <inline-formula id="inf9">
<mml:math id="m9">
<mml:mrow>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf10">
<mml:math id="m10">
<mml:mrow>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> , and <inline-formula id="inf11">
<mml:math id="m11">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:mi mathvariant="bold-italic">j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. With this preliminary procedure, all parametric or non-parametric vQTL tests have the potential to adjust for covariates such as age, sex, and ancestry principal components (PCs).</p>
</sec>
<sec id="s2-2">
<title>2.2 Non-parametric vQTL methods</title>
<p>Non-parametric vQTL methods are developed without the assumption of normally distributed traits. Therefore, they are generally more robust to trait distributions (<xref ref-type="bibr" rid="B21">Miao et al., 2022</xref>). Like many non-parametric statistical tests, they may be less powerful than parametric vQTL methods when the traits indeed follow normal distributions (<xref ref-type="bibr" rid="B26">Ronnegard and Valdar, 2012</xref>). However, they can have more valid results even when the traits are not normally distributed. In the following, I introduce two non-parametric methods, including the Kruskal-Wallis test (KW) and the quantile integral linear model (QUAIL).</p>
<sec id="s2-2-1">
<title>2.2.1 Kruskal-Wallis test (KW)</title>
<p>KW is a non-parametric statistical test used to assess whether there is a significant difference between the medians of two or more independent groups. Because the KW test is not originally designed to detect differences in variances, I have to convert the data into a measure of &#x201c;dispersion&#x201d; before using KW. Let <inline-formula id="inf12">
<mml:math id="m12">
<mml:mrow>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> be the &#x201c;covariate-adjusted residual&#x201d; (obtained from <xref ref-type="sec" rid="s2-1">Section 2.1</xref>) of individual <italic>j</italic> with genotype <italic>i</italic>, <inline-formula id="inf13">
<mml:math id="m13">
<mml:mrow>
<mml:mover accent="true">
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x223c;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> be the median of &#x201c;covariate-adjusted residuals&#x201d; within genotype group <italic>i</italic>. The deviation between individual <italic>j</italic>&#x2019;s covariate-adjusted residual and his/her group median is <inline-formula id="inf14">
<mml:math id="m14">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x223c;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> (a measure of dispersion), where <inline-formula id="inf15">
<mml:math id="m15">
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and usually <inline-formula id="inf16">
<mml:math id="m16">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> (subscript <italic>j</italic> is sufficient to distinguish all <italic>N</italic> individuals; subscript <italic>i</italic> is supplementary to indicate the genotype group). Suppose <inline-formula id="inf17">
<mml:math id="m17">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the rank of <inline-formula id="inf18">
<mml:math id="m18">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> counting from all individuals across three genotype groups. The KW test statistic (<xref ref-type="bibr" rid="B11">Kruskal and Wallis, 1952</xref>) is listed in <xref ref-type="disp-formula" rid="e1">Equation 1</xref> as follows,<disp-formula id="e1">
<mml:math id="m19">
<mml:mrow>
<mml:mtext>KW</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>M</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mover accent="true">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:msub>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>r</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>M</mml:mi>
</mml:msubsup>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:msubsup>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>r</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where <italic>N</italic> is the total sample size, <inline-formula id="inf19">
<mml:math id="m20">
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the number of individuals in genotype group <italic>i</italic>, <italic>M</italic> is the number of genotype groups (usually <italic>M</italic> &#x3d; 3), <inline-formula id="inf20">
<mml:math id="m21">
<mml:mrow>
<mml:mover accent="true">
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:msub>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:msubsup>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the average of ranks in genotype group <italic>i</italic>, <inline-formula id="inf21">
<mml:math id="m22">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>r</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>M</mml:mi>
</mml:msubsup>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:msubsup>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:mi>N</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the average of ranks among all individuals. The KW test statistic (1) follows the chi-square distribution with the degrees of freedom <inline-formula id="inf22">
<mml:math id="m23">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>. All analyses in this work were conducted using R (version 4.3.1). The KW test was implemented with the R function &#x201c;kruskal.test&#x201d; in the R environment.</p>
</sec>
<sec id="s2-2-2">
<title>2.2.2 Quantile integral linear model (QUAIL)</title>
<p>Although I have adjusted the covariates&#x2019; effects in <xref ref-type="sec" rid="s2-1">Section 2.1</xref>, I still present the statistical model shown in QUAIL&#x2019;s paper as follows (<xref ref-type="bibr" rid="B21">Miao et al., 2022</xref>). Through this model, people may get a complete understanding of Miao et al.&#x2019;s methodology. Let the conditional quantile function (<italic>Q</italic>) of the trait <italic>Y</italic> at a SNP be <xref ref-type="disp-formula" rid="e2">Equation 2</xref> as follows <disp-formula id="e2">
<mml:math id="m24">
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mi>Y</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mo>&#x7c;</mml:mo>
<mml:mi>g</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>y</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold-italic">X</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mi>&#x3c4;</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>i</mml:mi>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mi>&#x3c4;</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="bold-italic">X</mml:mi>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3b1;</mml:mi>
<mml:mi mathvariant="bold-italic">&#x3c4;</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>where <inline-formula id="inf23">
<mml:math id="m25">
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the quantile ranging from 0 to 0.5, <inline-formula id="inf24">
<mml:math id="m26">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf25">
<mml:math id="m27">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mi>&#x3c4;</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the intercept, <inline-formula id="inf26">
<mml:math id="m28">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mi>&#x3c4;</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the coefficient of the quantile regression, <bold>
<italic>X</italic>
</bold> is the <inline-formula id="inf27">
<mml:math id="m29">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> matrix for <italic>C</italic> covariates, and <inline-formula id="inf28">
<mml:math id="m30">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="bold-italic">&#x3b1;</mml:mi>
<mml:mi mathvariant="bold-italic">&#x3c4;</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the <italic>C</italic>-length vector for covariate effects. QUAIL measures <inline-formula id="inf29">
<mml:math id="m31">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mi>&#x3c4;</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> given <inline-formula id="inf30">
<mml:math id="m32">
<mml:mrow>
<mml:mi mathvariant="normal">&#x3c4;</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.5</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, i.e., the difference between the regression coefficients of the <inline-formula id="inf31">
<mml:math id="m33">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>
<sup>th</sup> and <inline-formula id="inf32">
<mml:math id="m34">
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
<sup>th</sup> quantiles. QUAIL tests the hypothesis <inline-formula id="inf33">
<mml:math id="m35">
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>:</mml:mo>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> vs <inline-formula id="inf34">
<mml:math id="m36">
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>:</mml:mo>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2260;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf35">
<mml:math id="m37">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> in <xref ref-type="disp-formula" rid="e3">Equation 3</xref> is the quantile-integrated effect, i.e.,<disp-formula id="e3">
<mml:math id="m38">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mo>&#x222b;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mn>0.5</mml:mn>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mi>&#x3c4;</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
<inline-formula id="inf36">
<mml:math id="m39">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the integral of <inline-formula id="inf37">
<mml:math id="m40">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mi>&#x3c4;</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> from <inline-formula id="inf38">
<mml:math id="m41">
<mml:mrow>
<mml:mi mathvariant="normal">&#x3c4;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> to <inline-formula id="inf39">
<mml:math id="m42">
<mml:mrow>
<mml:mi mathvariant="normal">&#x3c4;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.5</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, which can be estimated by summing the areas of <italic>K</italic> rectangles with a length of <inline-formula id="inf40">
<mml:math id="m43">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mover accent="true">
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x3c4;</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:msub>
<mml:mi>&#x3c4;</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:msub>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> (where <italic>k</italic> &#x3d; 1, &#x2026; , <italic>K</italic>) and a width of <inline-formula id="inf41">
<mml:math id="m44">
<mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>0.5</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>/</mml:mo>
<mml:mn>2</mml:mn>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. That is, <inline-formula id="inf42">
<mml:math id="m45">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> can be estimated through <inline-formula id="inf43">
<mml:math id="m46">
<mml:mrow>
<mml:mover accent="true">
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>K</mml:mi>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mover accent="true">
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x3c4;</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:msub>
<mml:mi>&#x3c4;</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:msub>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:mn>2</mml:mn>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. To have a more accurate estimation for <inline-formula id="inf44">
<mml:math id="m47">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <italic>K</italic> is required to be a large number. In the QUAIL R code, Miao et al. used <italic>K</italic> &#x3d; 100 as the default setting (<xref ref-type="bibr" rid="B21">Miao et al., 2022</xref>). The QUAIL R code was downloaded from GitHub (<ext-link ext-link-type="uri" xlink:href="https://github.com/qlu-lab/QUAIL">https://github.com/qlu-lab/QUAIL</ext-link>).</p>
</sec>
</sec>
<sec id="s2-3">
<title>2.3 Parametric vQTL methods</title>
<p>Parametric vQTL methods are usually developed with the assumption of normally distributed traits (<xref ref-type="bibr" rid="B18">Marderstein et al., 2021</xref>; <xref ref-type="bibr" rid="B36">Young et al., 2018</xref>). Therefore, their performances are generally more sensitive to trait distributions (<xref ref-type="bibr" rid="B21">Miao et al., 2022</xref>). However, if the trait really follows the normal distribution, parametric tests can be more powerful than their non-parametric counterparts. In the following, I describe three parametric vQTL methods, including the deviation regression model (DRM), Brown-Forsythe test (BF), and the double generalized linear model (DGLM).</p>
<sec id="s2-3-1">
<title>2.3.1 Deviation regression model (DRM)</title>
<p>To search for SNPs that are associated with phenotypic variability, Marderstein et al. regressed <inline-formula id="inf45">
<mml:math id="m48">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> on the minor allele count at each SNP (i.e., 0, 1, or 2) (<xref ref-type="bibr" rid="B18">Marderstein et al., 2021</xref>). The regression model is <inline-formula id="inf46">
<mml:math id="m49">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#xb7;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>&#x3b5;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, where the predictor <inline-formula id="inf47">
<mml:math id="m50">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> is the minor allele count (i.e., 0, 1, or 2) and the random error term <inline-formula id="inf48">
<mml:math id="m51">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b5;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> follows a normal distribution with a mean of 0 and a variance of <inline-formula id="inf49">
<mml:math id="m52">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>&#x3b5;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>. Because <inline-formula id="inf50">
<mml:math id="m53">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the deviation between individuals&#x2019; covariate-adjusted residuals and their group medians, this approach is called the deviation regression model (DRM). DRM tests whether the regression coefficient <inline-formula id="inf51">
<mml:math id="m54">
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is statistically significant, i.e., <inline-formula id="inf52">
<mml:math id="m55">
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>:</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> vs <inline-formula id="inf53">
<mml:math id="m56">
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>:</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x2260;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. The DRM R code was downloaded from GitHub (<ext-link ext-link-type="uri" xlink:href="https://github.com/drewmard/DRM">https://github.com/drewmard/DRM</ext-link>).</p>
</sec>
<sec id="s2-3-2">
<title>2.3.2 Brown-Forsythe test (BF)</title>
<p>The BF test statistic is listed in <xref ref-type="disp-formula" rid="e4">Equation 4</xref> as follows, <disp-formula id="e4">
<mml:math id="m57">
<mml:mrow>
<mml:mtext>BF</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>M</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mover accent="true">
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:msub>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>D</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>M</mml:mi>
</mml:msubsup>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:msubsup>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:msub>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>where <inline-formula id="inf54">
<mml:math id="m58">
<mml:mrow>
<mml:mover accent="true">
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:msub>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:msubsup>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the average of deviations in genotype group <italic>i</italic>, <inline-formula id="inf55">
<mml:math id="m59">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>D</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>M</mml:mi>
</mml:msubsup>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:msubsup>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the average of deviations in all individuals. The BF test statistic (4) follows the <italic>F</italic> distribution with the numerator degrees of freedom <inline-formula id="inf56">
<mml:math id="m60">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>, and the denominator degrees of freedom <inline-formula id="inf57">
<mml:math id="m61">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> (<xref ref-type="bibr" rid="B4">Brown and Forsythe, 1974</xref>). The BF test is equivalent to the ANOVA <italic>F</italic> test for regressing <inline-formula id="inf58">
<mml:math id="m62">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> on two indicator variables that separate the three genotype groups of an SNP. Therefore, BF treats genotypes as a categorical scale (by using two indicator variables), while DRM regards genotypes as a continuous scale (by coding genotypes as 0, 1, or 2).</p>
<p>To implement the BF test, I used the R function &#x201c;leveneTest&#x201d; while specifying &#x201c;center &#x3d; median&#x201d; in the &#x201c;car&#x201d; package (version 3.1&#x2013;3) (<xref ref-type="bibr" rid="B7">Fox and Weisberg, 2018</xref>). The Levene&#x2019;s and the BF tests are both used to test the homoscedasticity across different groups. The BF test is more robust to outliers than Levene&#x2019;s test because it uses deviations from the group median, while Levene&#x2019;s test uses deviations from the group mean.</p>
</sec>
<sec id="s2-3-3">
<title>2.3.3 Double generalized linear model (DGLM)</title>
<p>DGLM fits two generalized linear models (GLM) that model genetic effects on the mean and dispersion of the covariate-adjusted residuals (<xref ref-type="bibr" rid="B37">Zhang and Bell, 2024</xref>). It regresses the covariate-adjusted residual <inline-formula id="inf59">
<mml:math id="m63">
<mml:mrow>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> on the minor allele count at each SNP (i.e., 0, 1, or 2) while allowing the dispersion of <inline-formula id="inf60">
<mml:math id="m64">
<mml:mrow>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> varies with the minor allele count. The regression model is <inline-formula id="inf61">
<mml:math id="m65">
<mml:mrow>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#xb7;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>&#x3b5;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, where the random error term <inline-formula id="inf62">
<mml:math id="m66">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b5;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> follows a normal distribution with a mean of 0 and a variance of <inline-formula id="inf63">
<mml:math id="m67">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>i</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> (allowing each genotype group has its variance). The DGLM tests whether the variance of <inline-formula id="inf64">
<mml:math id="m68">
<mml:mrow>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> varies with the predictor (i.e., the minor allele count at each SNP). The DGLM method was implemented with the &#x201c;<italic>dglm</italic>&#x201d; R package (version: 1.8.6).</p>
</sec>
</sec>
<sec id="s2-4">
<title>2.4 Data from the Taiwan Biobank (TWB)</title>
<p>As of February 2024, 147,836 individuals aged <italic>30 to 70 years</italic> have been genotyped whole-genome. The TWB performed genotype imputation with the IMPUTE2 software (v2.3.1) (<xref ref-type="bibr" rid="B5">Delaneau et al., 2013</xref>; <xref ref-type="bibr" rid="B8">Howie et al., 2009</xref>). To improve the imputation accuracy, the TWB combined 1,445 TWB individuals with whole-genome sequence data and 504 East Asians (EAS) from the 1,000 Genomes Phase 3 v5 as the reference panel (<xref ref-type="bibr" rid="B33">Wei et al., 2021</xref>). After completing the imputation, the TWB excluded variants with missing rates &#x3e;5%, minor allele frequencies (MAFs) &#x3c; 0.01%, or imputation information scores &#x3c;0.3, where 0.3 was usually adopted as the acceptable threshold for imputation quality (<xref ref-type="bibr" rid="B9">Kosugi et al., 2023</xref>). After this quality control filtering, we had 9,814,944 autosomal variants for analysis.</p>
</sec>
<sec id="s2-5">
<title>2.5 Simulation studies</title>
<p>I performed simulations to evaluate the type I error rates and power of the five abovementioned methods. I here focused on the simulations of detecting SNP1-by-SNP2 interactions. Nonetheless, the results can be generalized to GxE identification. I randomly selected four SNPs as SNP1, including rs34625133 on chromosome (chr.) 1, rs7870809 on chr. 9, rs7982209 on chr. 13, and rs140100 on chr. 22. The MAFs of these four SNPs were 0.1, 0.2, 0.3, and 0.4, respectively. They were, in turn (one by one), treated as SNP1. I aimed to evaluate the power of detecting SNP1-by-SNP2 interactions under various MAFs of SNP1.</p>
<p>I used 20,000 common SNPs (MAFs <inline-formula id="inf65">
<mml:math id="m69">
<mml:mrow>
<mml:mo>&#x2265;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 0.05) on chr. 1 as SNP2 and generated traits as <xref ref-type="disp-formula" rid="e5">Equation 5</xref>,<disp-formula id="e5">
<mml:math id="m70">
<mml:mrow>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mi mathvariant="italic">INT</mml:mi>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>&#x3b5;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mn>30000</mml:mn>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mtext>or&#x2009;</mml:mtext>
<mml:mn>147836</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>where <inline-formula id="inf66">
<mml:math id="m71">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, 1, or 2 represented individual <italic>j</italic>&#x2019;s minor allele count at SNP1, and <inline-formula id="inf67">
<mml:math id="m72">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, 1, or 2 was his/her minor allele count at SNP2. I considered two levels of sample size: <italic>N</italic> &#x3d; 30,000 or 147,836, in which 147,836 was the number of TWB individuals with genotyping data. The coefficients <inline-formula id="inf68">
<mml:math id="m73">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf69">
<mml:math id="m74">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are main effects of SNP1 and SNP2, respectively. I investigated two situations: (1) with SNP main effects (<inline-formula id="inf70">
<mml:math id="m75">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.3</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>); and (2) without SNP main effects (<inline-formula id="inf71">
<mml:math id="m76">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>). Given an individual&#x2019;s genotypes, <inline-formula id="inf72">
<mml:math id="m77">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mi mathvariant="italic">INT</mml:mi>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the deterministic part and <inline-formula id="inf73">
<mml:math id="m78">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b5;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the stochastic part of the trait. Following Miao et al. (<xref ref-type="bibr" rid="B21">Miao et al., 2022</xref>), I considered three settings for the random error term <inline-formula id="inf74">
<mml:math id="m79">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b5;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>: (1) a standard normal distribution; (2) a <italic>t</italic> distribution with the degrees of freedom 3 (a kurtotic distribution); (3) a chi-square distribution with the degrees of freedom 6 (a skewed distribution). Because <inline-formula id="inf75">
<mml:math id="m80">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b5;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the only stochastic part of the trait, the distribution of the trait is determined by the distribution of <inline-formula id="inf76">
<mml:math id="m81">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b5;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. A kurtotic distribution such as the <italic>t</italic> distribution has heavier tails than the normal distribution. This phenomenon mimics the fact that more extreme values exist than the normal distribution. The chi-square distribution is right-skewed like the pattern of many phenotypes, such as cholesterol (<xref ref-type="bibr" rid="B30">Tharu and Tsokos, 2017</xref>) and body weight (<xref ref-type="bibr" rid="B10">Kozlowski and Gawelczyk, 2002</xref>). To fix the phenotypic variance explained by SNPs across different distributions of <inline-formula id="inf77">
<mml:math id="m82">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b5;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, I followed Miao et al. (<xref ref-type="bibr" rid="B21">Miao et al., 2022</xref>) to perform the <italic>z</italic>-score transformation on <inline-formula id="inf78">
<mml:math id="m83">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b5;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>By specifying <inline-formula id="inf79">
<mml:math id="m84">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mi mathvariant="italic">INT</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, I plotted the QQ plot to examine the p-values under the null hypothesis. On the other hand, <inline-formula id="inf80">
<mml:math id="m85">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mi mathvariant="italic">INT</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.3</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> was assumed when evaluating the power of the vQTLs methods. To sum up, 12 scenarios (two levels of sample size <inline-formula id="inf81">
<mml:math id="m86">
<mml:mrow>
<mml:mo>&#xd7;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> two situations of main effects <inline-formula id="inf82">
<mml:math id="m87">
<mml:mrow>
<mml:mo>&#xd7;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> three kinds of trait distributions) were simulated to evaluate the performance of the five vQTL methods. Each scenario was simulated 20,000 times. The percentage of the variance explained by the interaction effect was approximately 0.3%, 0.6%, 0.9%, and 1.2% when the MAF of SNP1 was 0.1, 0.2, 0.3, and 0.4, respectively. The coefficients in <xref ref-type="disp-formula" rid="e5">Equation (5)</xref> (i.e., <inline-formula id="inf83">
<mml:math id="m88">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf84">
<mml:math id="m89">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf85">
<mml:math id="m90">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mi mathvariant="italic">INT</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>) were used to generate phenotype <italic>Y</italic>. Instead of testing any coefficient, in simulations, I tested whether phenotypic variance varied with genotype groups.</p>
<p>Although <xref ref-type="disp-formula" rid="e5">Equation (5)</xref> is originally designed to test for GxG interactions, the results can be generalized to GxE identification. Without loss of generality, I can replace <inline-formula id="inf86">
<mml:math id="m91">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> in <xref ref-type="disp-formula" rid="e5">Equation (5)</xref> with an environmental factor having three possible levels: 0, 1, and 2. Therefore, the simulation results can be generalized to GxE identification.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>3 Results</title>
<sec id="s3-1">
<title>3.1 Simulation studies</title>
<sec id="s3-1-1">
<title>3.1.1 False positive rates</title>
<p>
<xref ref-type="fig" rid="F1">Figure 1</xref> presents the QQ plots (the left column) and power under &#x3b1; &#x3d; 5E-8 (columns 2&#x2013;5) for five vQTL tests (<italic>N</italic> &#x3d; 30,000; with SNP main effects). <xref ref-type="sec" rid="s12">Supplementary Figure S1</xref> demonstrates the results given <italic>N</italic> &#x3d; 30,000, and SNP main effects do not exist. <xref ref-type="sec" rid="s12">Supplementary Figure S2</xref> (with SNP main effects) and S3 (without SNP main effects) show the results when <italic>N</italic> &#x3d; 147,836. The rank-based inverse-normal transformation (INT) can convert data to a normal distribution. However, previous vQTL studies found that this INT transformation led to inflation in false positive rates (FPR) (<xref ref-type="bibr" rid="B18">Marderstein et al., 2021</xref>; <xref ref-type="bibr" rid="B21">Miao et al., 2022</xref>; <xref ref-type="bibr" rid="B32">Wang et al., 2019</xref>). The FPR was calculated by dividing the incorrectly classified negatives by the total negatives. To evaluate INT&#x2019;s performance in our simulation setting, we equipped it with DGLM.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>QQ plots (the left column) and power (<inline-formula id="inf87">
<mml:math id="m92">
<mml:mrow>
<mml:mi mathvariant="bold">&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 5E-8; columns 2&#x2013;5) for vQTL tests (<italic>N</italic> &#x3d; 30,000; with SNP main effects). The distribution for the error term: (top row) a standard normal distribution; (middle row) a <italic>t</italic> distribution with the degrees of freedom 3; (bottom row) a chi-square distribution with the degrees of freedom 6. (A) Normal dist., QQ plot. (B) Normal dist., MAF of SNP1 &#x003d; 0.1. (C) Normal dist., MAF of SNP1 &#x003d; 0.2. (D) Normal dist., MAF of SNP1 &#x003d; 0.3. (E) Normal dist., MAF of SNP1 &#x003d; 0.4. (F) t dist., QQ plot. (G) t dist., MAF o SNP1 &#x003d; 0.1. (H) t dist., MAF o SNP1 &#x003d; 0.2. (I) t dist., MAF o SNP1 &#x003d; 0.3. (J) t dist., MAF o SNP1 &#x003d; 0.4. (K) chi-square dist., QQ plot. (L) chi-square dist., MAF of SNP1 &#x003d; 0.1. (M) chi-square dist., MAF of SNP1 &#x003d; 0.2. (N) chi-square dist., MAF of SNP1 &#x003d; 0.3. (O) chi-square dist., MAF of SNP1 &#x003d; 0.4.</p>
</caption>
<graphic xlink:href="fgene-16-1617504-g001.tif">
<alt-text content-type="machine-generated">Fifteen-panel figure comparing statistical methods for genetic analysis. Panels A, F, and K show QQ plots of observed versus expected -log10 p-values. Panels B to E, G to J, and L to O show power versus minor allele frequency (MAF) of SNP2. Each line represents a different method: KW, QUAL, DRM, BF, DGLM, and DGLM_INT. Panels E, J, and O show almost complete power across methods with increasing MAF. The rest illustrate varying power, with some methods outperforming others as MAF increases.</alt-text>
</graphic>
</fig>
<sec id="s3-1-1-1">
<title>3.1.1.1 Parametric tests</title>
<p>The results show that DGLM has inflated FPR when the phenotypes are not normally distributed (<xref ref-type="fig" rid="F1">Figure 1</xref>; <xref ref-type="sec" rid="s12">Supplementary Figures S1&#x2013;S3F, K</xref>). Taking the INT transformation helps to adjust FPR for kurtotic phenotypes (<xref ref-type="fig" rid="F1">Figure 1</xref>; <xref ref-type="sec" rid="s12">Supplementary Figures S1&#x2013;S3F</xref>) but not for skewed phenotypes (<xref ref-type="fig" rid="F1">Figure 1</xref>; <xref ref-type="sec" rid="s12">Supplementary Figures S1&#x2013;S3K</xref>).</p>
<p>The BF test has inflated FPR given kurtotic phenotypes (<xref ref-type="fig" rid="F1">Figure 1</xref>; <xref ref-type="sec" rid="s12">Supplementary Figures S1&#x2013;S3F</xref>), and the situation is getting worse for smaller sample sizes (<xref ref-type="fig" rid="F1">Figure 1</xref>; <xref ref-type="sec" rid="s12">Supplementary Figure S1F</xref>). <xref ref-type="sec" rid="s12">Supplementary Figures S4&#x2013;S15</xref> demonstrate the QQ plots stratified by the MAF range. For the BF test, the inflation in FPR is especially severe for kurtotic phenotypes when <italic>N</italic> &#x3d; 30,000 and MAF &#x3c;0.2 (<xref ref-type="sec" rid="s12">Supplementary Figures S5, S8</xref>, (A) (B) (C)). Given <italic>N</italic> &#x3d; 30,000 and a small MAF, a genotype group may only contain a few observations, and false positives may come with the issue of data sparsity (<xref ref-type="bibr" rid="B15">Lin, 2024</xref>).</p>
<p>DRM is similar to BF, except that DRM treats genotypes as a continuous scale (0, 1, or 2). DRM does not categorize the three genotypes into three groups. Therefore, the sparsity within genotype groups is not a critical problem for DRM, and the inflation in FPR is not critical for DRM compared with BF (<xref ref-type="sec" rid="s12">Supplementary Figures S5, S8A&#x2013;C</xref>).</p>
</sec>
<sec id="s3-1-1-2">
<title>3.1.1.2 Non-parametric tests</title>
<p>QUAIL preserves an appropriate FPR across three trait distributions. KW maintains a suitable FPR under the kurtotic distribution (<xref ref-type="fig" rid="F1">Figure 1</xref>; <xref ref-type="sec" rid="s12">Supplementary Figure S1&#x2013;S3F</xref>) but has a slight inflation in FPR given the skewed distribution (<xref ref-type="fig" rid="F1">Figure 1</xref>; <xref ref-type="sec" rid="s12">Supplementary Figure S1&#x2013;S3K</xref>). Under a kurtotic distribution, more extreme values can occur compared to the normal distribution. Nonetheless, by transforming the extreme values into ranks, KW can still preserve an appropriate FPR. By contrast, previous research has found that KW tends to have inflated FPR for heteroscedastic cases (<xref ref-type="bibr" rid="B20">McDonald, 2014</xref>). Under a skewed distribution, unequal variances in ranks across the three genotype groups may still exist even if there is no GxG or GxE. KW&#x2019;s strategy to transform values into ranks cannot entirely address the heteroscedastic problem.</p>
</sec>
</sec>
<sec id="s3-1-2">
<title>3.1.2 Statistical power</title>
<p>The statistical power was calculated by dividing the correctly classified positives by the total positives (each scenario was simulated 20,000 times). When the phenotype is normally distributed, DGLM and DGLM_INT were the most powerful methods (<xref ref-type="fig" rid="F1">Figure 1</xref>; <xref ref-type="sec" rid="s12">Supplementary Figure S1&#x2013;S3C&#x2013;E</xref>). Among the two non-parametric tests, QUAIL was superior to KW for normally distributed phenotypes (<xref ref-type="fig" rid="F1">Figure 1</xref>; <xref ref-type="sec" rid="s12">Supplementary Figure S1&#x2013;S3C&#x2013;E</xref>). When the phenotype is a kurtotic distribution, DGLM and BF had inflation in FPR (<xref ref-type="fig" rid="F1">Figure 1</xref>; <xref ref-type="sec" rid="s12">Supplementary Figure S1&#x2013;S3F</xref>). KW became the most outstanding test among the four valid methods that preserve a suitable FPR (<xref ref-type="fig" rid="F1">Figure 1</xref>; <xref ref-type="sec" rid="s12">Supplementary Figure S1&#x2013;S3F&#x2013;J</xref>). When the phenotype is skewed, KW had a slight inflation in FPR, whereas DGLM and DGLM_INT had a severe inflation in FPR (<xref ref-type="fig" rid="F1">Figure 1</xref>; <xref ref-type="sec" rid="s12">Supplementary Figure S1&#x2013;S3K</xref>). The other three valid methods (QUAIL, DRM, and BF) had similar power (<xref ref-type="fig" rid="F1">Figure 1</xref>; <xref ref-type="sec" rid="s12">Supplementary Figure S1&#x2013;S3L&#x2013;O</xref>).</p>
<p>By comparing the results of different simulation scenarios, I concluded three points: (1) When the MAF for SNP1 or SNP2 was larger, the percentage of the variance explained by the interaction effect increased, and the power of each method was enhanced. (2) When the sample size was enlarged from <italic>N</italic> &#x3d; 30,000 to <italic>N</italic> &#x3d; 147,836, the power performances of all methods were greatly improved. (3) Lastly, by comparing <xref ref-type="fig" rid="F1">Figure 1</xref> with <xref ref-type="sec" rid="s12">Supplementary Figure S1, S2</xref> with <xref ref-type="sec" rid="s12">Supplementary Figure S3</xref>, the power of each method was higher in the presence of SNP main effects than in the absence of SNP main effects.</p>
</sec>
<sec id="s3-1-3">
<title>3.1.3 Computational time</title>
<p>
<xref ref-type="table" rid="T1">Table 1</xref> shows the computation time (in seconds) for each simulation replicate. I measured the execution time in R (version 4.3.1) on a Linux system running at 3.6&#x2009;GHz and 32&#xa0;GB of RAM. Although QUAIL is a valid test with appropriate FPR under three trait distributions, it takes much more computation time than other methods. The other non-parametric test, KW, requires only 1/70&#x2013;1/90 of QUAIL&#x2019;s execution time. The computation time of the four parametric tests is also reasonable. The time of computation needed for these tests is ordered as DRM &#x3c; BF &#x3c; KW &#x3c; DGLM <inline-formula id="inf88">
<mml:math id="m93">
<mml:mrow>
<mml:mo>&#x2248;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> DGLM_INT &#x3c;&#x3c;&#x3c;&#x3c;&#x3c; QUAIL. <xref ref-type="table" rid="T2">Table 2</xref> summarizes the performance of each method in the simulation on FPR, power, and computational time under different scenarios.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Computation time (in seconds) for each simulation replicate The execution time was measured in R (version 4.3.1) on a Linux system running at 3.6&#x2009;GHz and 32&#xa0;GB of RAM.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Phenotype distribution</th>
<th align="center">
<italic>N</italic>
</th>
<th align="center">KW</th>
<th align="center">QUAIL</th>
<th align="center">DRM</th>
<th align="center">BF</th>
<th align="center">DGLM</th>
<th align="center">DGLM_INT</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Normal</td>
<td align="center">30,000</td>
<td align="center">0.076</td>
<td align="center">5.372</td>
<td align="center">0.018</td>
<td align="center">0.020</td>
<td align="center">0.151</td>
<td align="center">0.162</td>
</tr>
<tr>
<td align="center">T</td>
<td align="center">30,000</td>
<td align="center">0.092</td>
<td align="center">8.252</td>
<td align="center">0.024</td>
<td align="center">0.028</td>
<td align="center">0.273</td>
<td align="center">0.214</td>
</tr>
<tr>
<td align="center">Chi-square</td>
<td align="center">30,000</td>
<td align="center">0.075</td>
<td align="center">5.485</td>
<td align="center">0.018</td>
<td align="center">0.021</td>
<td align="center">0.163</td>
<td align="center">0.168</td>
</tr>
<tr>
<td align="center">Normal</td>
<td align="center">147,836</td>
<td align="center">0.402</td>
<td align="center">33.464</td>
<td align="center">0.096</td>
<td align="center">0.105</td>
<td align="center">0.824</td>
<td align="center">0.891</td>
</tr>
<tr>
<td align="center">T</td>
<td align="center">147,836</td>
<td align="center">0.399</td>
<td align="center">33.950</td>
<td align="center">0.094</td>
<td align="center">0.103</td>
<td align="center">1.090</td>
<td align="center">0.875</td>
</tr>
<tr>
<td align="center">Chi-square</td>
<td align="center">147,836</td>
<td align="center">0.411</td>
<td align="center">35.913</td>
<td align="center">0.098</td>
<td align="center">0.106</td>
<td align="center">0.884</td>
<td align="center">0.929</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Summary of the simulation results.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left"/>
<th align="left">KW</th>
<th align="left">QUAIL</th>
<th align="left">DRM</th>
<th align="left">BF</th>
<th align="left">DGLM</th>
<th align="left">DGLM_INT</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">False positive rate (FPR)</td>
<td align="left">Slight inflation in FPR given skewed distributions</td>
<td align="left">Valid</td>
<td align="left">Inflation in FPR is not critical for DRM compared with BF.</td>
<td align="left">Inflation in FPR is especially severe for kurtotic traits when N &#x3d; 30,000 and MAF &#x3c;0.2</td>
<td align="left">Valid only when the trait is normally distributed</td>
<td align="left">Severe inflation in FPR given skewed distributions</td>
</tr>
<tr>
<td align="left">Statistical power</td>
<td align="left">Optimal for kurtotic distributions</td>
<td align="left">Never among the optimal choices</td>
<td align="left">Similar to BF.</td>
<td align="left">Similar to DRM.</td>
<td align="left">Optimal for normally distributed traits</td>
<td align="left">Optimal for normally distributed traits</td>
</tr>
<tr>
<td align="left">Computational time (ranking from shortest to longest)</td>
<td align="left">3</td>
<td align="left">6 (longest)</td>
<td align="left">1 (shortest)</td>
<td align="left">2</td>
<td align="left">4</td>
<td align="left">4</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s3-2">
<title>3.2 Real data analysis</title>
<sec id="s3-2-1">
<title>3.2.1 30 vQTLs of four lipid traits</title>
<p>When conducting actual genome-wide data analysis, investigators had to exclude cryptic relatedness among individuals (<xref ref-type="bibr" rid="B19">Marees et al., 2018</xref>). The TWB investigated cryptic relatedness among participants with the KING (Kinship-based INference for GWAS) software (<xref ref-type="bibr" rid="B17">Manichaikul et al., 2010</xref>). I removed the person with higher missing genotype rates for each first- or second-degree relative pair. This step excluded 28,928 from the 147,836 TWB individuals, and 118,908 remained in the vQTL analysis.</p>
<p>I analyzed four lipid traits, including high-density lipoprotein cholesterol (HDL), low-density lipoprotein cholesterol (LDL), triglyceride (TG), and total cholesterol (TCHO). <xref ref-type="sec" rid="s12">Supplementary Figure S16</xref> shows the histograms of the four lipid traits. All four traits are right-skewed with positive skewness values, especially TG (skewness &#x3d; 2.16). I first removed SNP&#x2019;s main effect and covariates&#x2019; effects from each phenotype. Specifically, I extracted the residuals of regressing each trait on <inline-formula id="inf89">
<mml:math id="m94">
<mml:mrow>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> , <inline-formula id="inf90">
<mml:math id="m95">
<mml:mrow>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> (two indicator variables separating the three genotype groups of a SNP), and 17 covariates, including sex (male vs female), age (in years), BMI (in kg/m<sup>2</sup>), current smoking status (yes vs no), current drinking status (yes vs no), performing physical exercise (yes vs no), educational attainment (integer ranging from 1 to 7), and 10 ancestry PCs. Current smoking was defined as &#x201c;having smoked cigarettes for at least 6 months when joining the TWB&#x201d;. Drinking indicated &#x201c;having a weekly intake of more than 150&#xa0;mL of alcoholic beverages for at least 6 months when joining the TWB&#x201d;. Regular exercise meant &#x201c;performing exercise lasting for 30&#xa0;min thrice a week&#x201d;. Educational attainment was coded as an integer ranging from 1 to 7: 1 (illiterate), 2 (no formal education but literate), 3 (primary school graduate), 4 (junior high school graduate), 5 (senior high school graduate), 6 (college graduate), and 7 (Master&#x2019;s or higher degree).</p>
<p>Sex, age, and BMI significantly influence people&#x2019;s lipid profiles (<xref ref-type="bibr" rid="B1">Beyene et al., 2020</xref>). Smoking impairs lipid metabolism by decreasing LDL receptor expression (<xref ref-type="bibr" rid="B16">Ma et al., 2020</xref>). Alcohol consumption disturbs lipid metabolism by increasing adipose tissue lipolysis and causing ectopic fat deposition in the liver (<xref ref-type="bibr" rid="B29">Steiner and Lang, 2017</xref>). Regular physical activity has been found to improve lipoprotein-lipid profiles (<xref ref-type="bibr" rid="B14">Lin, 2021</xref>). Moreover, lower educational attainment is usually linked to worse lipid profiles (<xref ref-type="bibr" rid="B12">Lara and Amigo, 2018</xref>). Because these seven covariates are associated with lipid traits, I adjusted them in all vQTL methods. Furthermore, the first 10 ancestry PCs are adjusted in genetic analyses to avoid population stratification (<xref ref-type="bibr" rid="B31">Uffelmann et al., 2021</xref>).</p>
<p>Based on the simulation results, I chose the KW test as the primary vQTL method. This non-parametric approach is robust to outliers by transforming trait values into ranks (<xref ref-type="fig" rid="F1">Figure 1</xref>; <xref ref-type="sec" rid="s12">Supplementary Figure S1&#x2013;S3F</xref>). Moreover, although KW has a slight inflation in FPR for skewed trait distributions (<xref ref-type="fig" rid="F1">Figure 1</xref>; <xref ref-type="sec" rid="s12">Supplementary Figure S1&#x2013;S3K</xref>), investigators can perform a follow-up regression analysis to check whether GxE and GxG exist (i.e., the so-called &#x201c;direct GxE or GxG analysis&#x201d;). Furthermore, KW needs only 1/70&#x2013;1/90 of the other non-parametric competitor&#x2019;s (QUAIL) execution time. Because SNPs with MAFs &#x3c;0.05 are difficult to replicate in GxG or GxE findings (<xref ref-type="bibr" rid="B15">Lin, 2024</xref>; <xref ref-type="bibr" rid="B32">Wang et al., 2019</xref>), I only analyzed 2,580,790 common SNPs with MAFs <inline-formula id="inf91">
<mml:math id="m96">
<mml:mrow>
<mml:mo>&#x2265;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 0.05.</p>
<p>With the PLINK clumping procedure (<xref ref-type="bibr" rid="B24">Purcell et al., 2007</xref>), I detected 10 independent HDL-vQTLs (KW p-value &#x3c; 5E-8) with linkage disequilibrium (LD) measure <italic>r</italic>
<sup>
<italic>2</italic>
</sup> &#x3c; 0.01. Moreover, 6, 12, and 6 independent vQTLs were identified from LDL, TG, and TCHO, respectively. Four SNPs were found to be vQTLs of multiple lipid traits, including rs4704208 (near <italic>HMGCR</italic>), rs11748027 (in <italic>ANKDD1B</italic>), rs483082 (in <italic>APOC1</italic>), and rs662799 (in <italic>APOA5</italic>). <xref ref-type="table" rid="T3">Table 3</xref> summarizes the 30 (&#x3d;10 &#x2b; 6&#x2b;12&#x2b;6&#x2013;4) unique vQTLs of the four lipid traits.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Thirty vQTLs (KW p-value &#x3c; 5E-8 and linkage disequilibrium measure <italic>r</italic>
<sup>
<italic>2</italic>
</sup> &#x3c; 0.01) of the four lipid traits.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Trait (number of vQTLs)</th>
<th align="left">Chromosome</th>
<th align="left">Base pair</th>
<th align="left">vQTL</th>
<th align="left">Gene</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">HDL (10)</td>
<td align="left">9</td>
<td align="left">104827463</td>
<td align="left">rs4149307</td>
<td align="left">
<italic>ABCA1</italic>
</td>
</tr>
<tr>
<td align="left"/>
<td align="left">9</td>
<td align="left">104902020</td>
<td align="left">rs1883025</td>
<td align="left">
<italic>ABCA1</italic>
</td>
</tr>
<tr>
<td align="left"/>
<td align="left">15</td>
<td align="left">58431476</td>
<td align="left">rs1800588</td>
<td align="left">
<italic>LIPC</italic>
</td>
</tr>
<tr>
<td align="left"/>
<td align="left">16</td>
<td align="left">56957451</td>
<td align="left">rs183130</td>
<td align="left">Near <italic>CETP</italic>
</td>
</tr>
<tr>
<td align="left"/>
<td align="left">16</td>
<td align="left">67998971</td>
<td align="left">rs113731015</td>
<td align="left">
<italic>DPEP2</italic>
</td>
</tr>
<tr>
<td align="left"/>
<td align="left">16</td>
<td align="left">68256297</td>
<td align="left">rs12446418</td>
<td align="left">
<italic>PLA2G15</italic>
</td>
</tr>
<tr>
<td align="left"/>
<td align="left">18</td>
<td align="left">49627658</td>
<td align="left">rs80208964</td>
<td align="left">
<italic>LOC105372112</italic>
</td>
</tr>
<tr>
<td align="left"/>
<td align="left">19</td>
<td align="left">11236981</td>
<td align="left">rs737338</td>
<td align="left">
<italic>DOCK6</italic>
</td>
</tr>
<tr>
<td align="left"/>
<td align="left">19</td>
<td align="left">44912383</td>
<td align="left">rs445925</td>
<td align="left">
<italic>APOC1</italic>
</td>
</tr>
<tr>
<td align="left"/>
<td align="left">20</td>
<td align="left">45907572</td>
<td align="left">rs148753678</td>
<td align="left">
<italic>PLTP</italic>
</td>
</tr>
<tr>
<td align="left">LDL (6)</td>
<td align="left">2</td>
<td align="left">21024193</td>
<td align="left">rs57825321</td>
<td align="left">
<italic>APOB</italic>
</td>
</tr>
<tr>
<td align="left"/>
<td align="left">5</td>
<td align="left">75325846</td>
<td align="left">rs4704208<xref ref-type="table-fn" rid="Tfn1">
<sup>a</sup>
</xref>
</td>
<td align="left">Near <italic>HMGCR</italic>
</td>
</tr>
<tr>
<td align="left"/>
<td align="left">5</td>
<td align="left">75614147</td>
<td align="left">rs11748027<xref ref-type="table-fn" rid="Tfn2">
<sup>b</sup>
</xref>
</td>
<td align="left">
<italic>ANKDD1B</italic>
</td>
</tr>
<tr>
<td align="left"/>
<td align="left">11</td>
<td align="left">116786951</td>
<td align="left">rs3741297</td>
<td align="left">
<italic>ZPR1</italic>
</td>
</tr>
<tr>
<td align="left"/>
<td align="left">19</td>
<td align="left">11131631</td>
<td align="left">rs2738464</td>
<td align="left">
<italic>LDLR</italic>
</td>
</tr>
<tr>
<td align="left"/>
<td align="left">19</td>
<td align="left">44912921</td>
<td align="left">rs483082<xref ref-type="table-fn" rid="Tfn3">
<sup>c</sup>
</xref>
</td>
<td align="left">
<italic>APOC1</italic>
</td>
</tr>
<tr>
<td align="left">TG (12)</td>
<td align="left">1</td>
<td align="left">62436136</td>
<td align="left">rs631106</td>
<td align="left">
<italic>USP1</italic>
</td>
</tr>
<tr>
<td align="left"/>
<td align="left">1</td>
<td align="left">62690372</td>
<td align="left">rs1168114</td>
<td align="left">
<italic>DOCK7</italic>
</td>
</tr>
<tr>
<td align="left"/>
<td align="left">2</td>
<td align="left">27518370</td>
<td align="left">rs780094</td>
<td align="left">
<italic>GCKR</italic>
</td>
</tr>
<tr>
<td align="left"/>
<td align="left">2</td>
<td align="left">27798099</td>
<td align="left">rs3935148</td>
<td align="left">
<italic>RBKS</italic>
</td>
</tr>
<tr>
<td align="left"/>
<td align="left">7</td>
<td align="left">73606007</td>
<td align="left">rs3812316</td>
<td align="left">
<italic>MLXIPL</italic>
</td>
</tr>
<tr>
<td align="left"/>
<td align="left">8</td>
<td align="left">19966163</td>
<td align="left">rs1803924</td>
<td align="left">
<italic>LPL</italic>
</td>
</tr>
<tr>
<td align="left"/>
<td align="left">8</td>
<td align="left">125465736</td>
<td align="left">rs2001945</td>
<td align="left">Near <italic>TRIB1</italic>
</td>
</tr>
<tr>
<td align="left"/>
<td align="left">11</td>
<td align="left">61815236</td>
<td align="left">rs174561</td>
<td align="left">
<italic>FADS1</italic>
</td>
</tr>
<tr>
<td align="left"/>
<td align="left">11</td>
<td align="left">116792991</td>
<td align="left">rs662799<xref ref-type="table-fn" rid="Tfn4">
<sup>d</sup>
</xref>
</td>
<td align="left">
<italic>APOA5</italic>
</td>
</tr>
<tr>
<td align="left"/>
<td align="left">11</td>
<td align="left">117050674</td>
<td align="left">rs1815786</td>
<td align="left">
<italic>SIK3</italic>
</td>
</tr>
<tr>
<td align="left"/>
<td align="left">17</td>
<td align="left">67966122</td>
<td align="left">rs10445361</td>
<td align="left">
<italic>BPTF</italic>
</td>
</tr>
<tr>
<td align="left"/>
<td align="left">19</td>
<td align="left">44913484</td>
<td align="left">rs438811</td>
<td align="left">
<italic>APOC1</italic>
</td>
</tr>
<tr>
<td align="left">TCHO (6)</td>
<td align="left">1</td>
<td align="left">109275536</td>
<td align="left">rs3832016</td>
<td align="left">
<italic>CELSR2</italic>
</td>
</tr>
<tr>
<td align="left"/>
<td align="left">2</td>
<td align="left">21029662</td>
<td align="left">rs13306194</td>
<td align="left">
<italic>APOB</italic>
</td>
</tr>
<tr>
<td align="left"/>
<td align="left">5</td>
<td align="left">75325846</td>
<td align="left">rs4704208<xref ref-type="table-fn" rid="Tfn1">
<sup>a</sup>
</xref>
</td>
<td align="left">Near <italic>HMGCR</italic>
</td>
</tr>
<tr>
<td align="left"/>
<td align="left">5</td>
<td align="left">75614147</td>
<td align="left">rs11748027<xref ref-type="table-fn" rid="Tfn2">
<sup>b</sup>
</xref>
</td>
<td align="left">
<italic>ANKDD1B</italic>
</td>
</tr>
<tr>
<td align="left"/>
<td align="left">11</td>
<td align="left">116792991</td>
<td align="left">rs662799<xref ref-type="table-fn" rid="Tfn4">
<sup>d</sup>
</xref>
</td>
<td align="left">
<italic>APOA5</italic>
</td>
</tr>
<tr>
<td align="left"/>
<td align="left">19</td>
<td align="left">44912921</td>
<td align="left">rs483082<xref ref-type="table-fn" rid="Tfn3">
<sup>c</sup>
</xref>
</td>
<td align="left">
<italic>APOC1</italic>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="Tfn1">
<label>
<sup>a</sup>
</label>
<p>rs4704208 (near <italic>HMGCR</italic>) is a vQTL of LDL and TCHO.</p>
</fn>
<fn id="Tfn2">
<label>
<sup>b</sup>
</label>
<p>rs11748027 (in <italic>ANKDD1B</italic>) is a vQTL of LDL and TCHO.</p>
</fn>
<fn id="Tfn3">
<label>
<sup>c</sup>
</label>
<p>rs483082 (in <italic>APOC1</italic>) is a vQTL of LDL and TCHO.</p>
</fn>
<fn id="Tfn4">
<label>
<sup>d</sup>
</label>
<p>rs662799 (in <italic>APOA5</italic>) is a vQTL of TG and TCHO.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3-2-2">
<title>3.2.2 Direct GxE analysis</title>
<p>From the regression (or variable selection) perspective, it is important to keep the hierarchical structure between main and interaction effects (<xref ref-type="bibr" rid="B38">Zhou et al., 2021</xref>). Following previous vQTL research (<xref ref-type="bibr" rid="B32">Wang et al., 2019</xref>), after identifying vQTLs, I performed a &#x201c;direct GxE analysis&#x201d; by regressing the trait on the minor allele count of each vQTL (0, 1, or 2), seven environmental factors (Es), and the product (interaction) term between the minor allele count and one of the seven&#xa0;Es. To remove population stratification, I also adjusted the top 10 ancestry PCs in this regression. The seven&#xa0;Es are listed as the horizontal axis of <xref ref-type="fig" rid="F2">Figure 2</xref>, including SEX (female vs male), SPO (performing regular exercise, yes vs no), EDU (educational attainment, an integer from 1 to 7), AGE (chronological age, in years), BMI (in kg/m<sup>2</sup>), DRK (alcohol consumption, yes vs no), and SMK (cigarette smoking status, yes vs no). As Westerman et al. (<xref ref-type="bibr" rid="B34">Westerman et al., 2022</xref>) indicated, some vQTLs were &#x201c;pleiotropic&#x201d; concerning phenotypic variability. <xref ref-type="table" rid="T2">Table 2</xref> also shows that four loci are vQTLs shared by multiple lipid traits. Instead of performing the direct GxE and GxG analysis for respective vQTLs of each lipid trait (i.e., 10 vQTLs for HDL, 6 vQTLs for LDL, 12 vQTLs for TG, and 6 vQTLs for TCHO), I analyzed all 30 vQTLs to have a more comprehensive picture of the four lipid traits.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>The phylogenetic heat map of the gene-environment interaction analysis for triglyceride (TG). The magnitude of the value represents&#x2013;log10 (two-sided p-value of the SNP-E interaction), which is always positive. However, I deliberately added a positive/negative sign before the magnitude. A positive sign indicates that the environmental factor (E) exacerbates the vQTLs&#x2019; effects. In contrast, a negative sign suggests that the E attenuates the vQTLs&#x2019; effects. The <italic>x</italic>-axis lists the 7&#xa0;Es, including SEX (female vs. male), SPO (performing regular exercise, yes vs no), EDU (educational attainment, an integer from 1 to 7), AGE (chronological age, in years), BMI (in kg/m<sup>2</sup>), DRK (alcohol consumption, yes vs no), and SMK (cigarette smoking status, yes vs no).</p>
</caption>
<graphic xlink:href="fgene-16-1617504-g002.tif">
<alt-text content-type="machine-generated">Heatmap displaying genetic markers and their association with various traits such as SEX, SPO, EDU, AGE, BMI, DRK, and SMK. Colors range from blue, indicating low values, to red, indicating high values. A dendrogram on the x-axis and y-axis shows clustering of traits and markers. A color key in the top left explains the value range from -8 to 8.</alt-text>
</graphic>
</fig>
<p>Because TG had more vQTLs compared with the other three traits, I present the phylogenetic heat map of the direct GxE analysis for TG in <xref ref-type="fig" rid="F2">Figure 2</xref>. The <italic>x</italic>-axis of <xref ref-type="fig" rid="F2">Figure 2</xref> lists the 7&#xa0;Es. The magnitude of the value in <xref ref-type="fig" rid="F2">Figure 2</xref> represents&#x2013;log10 (two-sided p-value of the SNP-E interaction), which is always positive. However, I deliberately added a positive/negative sign before the magnitude. A positive sign indicates that E exacerbates the vQTLs&#x2019; effects. In contrast, a negative sign suggests that the E attenuates the vQTLs&#x2019; effects. Females have more attenuated TG genetic effects than males (blue color in SEX column), whereas higher BMI, alcohol consumption, and cigarette smoking lead to more substantial TG genetic effects (red color in BMI, DRK, and SMK columns). Moreover, the phylogenetic heat maps for the remaining three traits are demonstrated in <xref ref-type="sec" rid="s12">Supplementary Figures S17&#x2013;S19</xref>.</p>
</sec>
<sec id="s3-2-3">
<title>3.2.3 Direct GxG analysis</title>
<p>I further performed a &#x201c;direct GxG analysis&#x201d; by regressing the trait on the minor allele counts (0, 1, or 2) of two SNPs, the seven&#xa0;Es mentioned above, and the product (interaction) term of the minor allele counts of the two SNPs. Similarly, I adjusted the top 10 ancestry PCs in this regression. <xref ref-type="fig" rid="F3">Figure 3</xref> shows the phylogenetic heat map of the direct GxG analysis for TG. The maps for the other three lipid traits can be found in <xref ref-type="sec" rid="s12">Supplementary Figures S20&#x2013;S22</xref>.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>The phylogenetic heat map of the gene-gene interaction analysis for triglyceride (TG). The magnitude of the value represents&#x2013;log10 (two-sided p-value of the SNP-SNP interaction), which is always positive. The most significant SNP-SNP interaction is between <italic>rs438811</italic> (chr. 19, in <italic>APOC1</italic>) and <italic>rs662799</italic> (chr. 11, in <italic>APOA5</italic>), where p &#x3d; 1.3E-15. SNP <italic>rs483082</italic> is highly correlated with <italic>rs438811</italic> (<italic>r</italic>
<sup>
<italic>2</italic>
</sup> &#x3d; 0.99), and <italic>rs3741297</italic> is in linkage disequilibrium with <italic>rs662799</italic> (<italic>r</italic>
<sup>
<italic>2</italic>
</sup> &#x3d; 0.19). Therefore, despite four black cells indicating four significant SNP-SNP interactions, I only highlighted <italic>rs438811</italic>-<italic>rs662799</italic> interaction.</p>
</caption>
<graphic xlink:href="fgene-16-1617504-g003.tif">
<alt-text content-type="machine-generated">Heatmap depicting genetic data with hierarchical clustering. Rows and columns represent genetic markers, labeled with rs numbers. Color intensity ranges from yellow to red to black, indicating value differences, as shown in the color key.</alt-text>
</graphic>
</fig>
<p>Regarding TG, the interaction between <italic>rs438811</italic> (chr. 19, in <italic>APOC1</italic>) and <italic>rs662799</italic> (chr. 11, in <italic>APOA5</italic>) is the most significant among the <inline-formula id="inf92">
<mml:math id="m97">
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:mn>30</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>2</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>435</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> GxG tests constructed by the 30 vQTLs (p &#x3d; 1.3E-15). The magnitude of the value in <xref ref-type="fig" rid="F3">Figure 3</xref> represents&#x2013;log10 (two-sided p-value of the SNP-SNP interaction), which is always positive. The most significant SNP-SNP interaction is between <italic>rs438811</italic> (chr. 19, in <italic>APOC1</italic>) and <italic>rs662799</italic> (chr. 11, in <italic>APOA5</italic>), where p &#x3d; 1.3E-15. SNP <italic>rs483082</italic> is highly correlated with <italic>rs438811</italic> (<italic>r</italic>
<sup>
<italic>2</italic>
</sup> &#x3d; 0.99), and <italic>rs3741297</italic> is in obvious linkage disequilibrium with <italic>rs662799</italic> (<italic>r</italic>
<sup>
<italic>2</italic>
</sup> &#x3d; 0.19). Therefore, despite four black cells in <xref ref-type="fig" rid="F3">Figure 3</xref> indicating four significant SNP-SNP interactions, I only highlighted <italic>rs438811</italic>-<italic>rs662799</italic> interaction. As analyzed by the direct GxG regression model, each T allele at <italic>rs438811</italic> is associated with 10.8&#xa0;mg/dL (p &#x3d; 7.0E-117), and each G allele at <italic>rs662799</italic> is associated with 26.4&#xa0;mg/dL (p &#x223c; 0, extremely low p-value). Possible values of the interaction (product) term between <italic>rs438811</italic> and <italic>rs662799</italic> are 0, 1 (&#x3d;<inline-formula id="inf93">
<mml:math id="m98">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>), 2 (&#x3d;<inline-formula id="inf94">
<mml:math id="m99">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> or <inline-formula id="inf95">
<mml:math id="m100">
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>), and 4 (&#x3d;<inline-formula id="inf96">
<mml:math id="m101">
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>). Each point of the interaction term is associated with 5.9&#xa0;mg/dL (p &#x3d; 1.3E-15). This result indicates a synergistic interaction between <italic>rs438811</italic> and <italic>rs662799</italic>: T allele at <italic>rs438811</italic> and G allele at <italic>rs662799</italic> working together exerts a more substantial impact on TG than they would have on their own.</p>
</sec>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>4 Discussion</title>
<p>Searching for vQTLs greatly facilitates the mining of GxE and GxG. For a GWAS incorporating one million SNPs and seven&#xa0;Es, seven million GxE tests and <inline-formula id="inf97">
<mml:math id="m102">
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:mn>1000000</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>2</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>499999500000</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> GxG tests are required to investigate GxE and GxG. In the data analysis of this work, 30 vQTLs were identified from four lipid traits. I only needed to perform 210 (<inline-formula id="inf98">
<mml:math id="m103">
<mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>30</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>7</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>) GxE and 435 (<inline-formula id="inf99">
<mml:math id="m104">
<mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:mn>30</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>2</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>) GxG tests for each trait. The number of tests has dramatically decreased, and the penalty for multiple tests can be attenuated with a preliminary vQTL search.</p>
<p>This work evaluates the performance of three parametric (DRM, BF, and DGLM) and two non-parametric (KW and QUAIL) methods in detecting vQTLs. Although DGLM is the most powerful method for normally distributed traits (<xref ref-type="fig" rid="F1">Figure 1</xref>; <xref ref-type="sec" rid="s12">Supplementary Figure S1&#x2013;S3C&#x2013;E</xref>), it has severe inflation in FPR for traits following other distributions (<xref ref-type="fig" rid="F1">Figure 1</xref>; <xref ref-type="sec" rid="s12">Supplementary Figure S1&#x2013;S3F, K</xref>). The random error term of DGLM is assumed to follow the normal distribution. This is why it fails to control for FPR when the trait is non-normally distributed. Therefore, DGLM should not be adopted for vQTL detection when the trait distribution is unclear. Taking INT transformation helps adjust FPR for kurtotic traits (<xref ref-type="fig" rid="F1">Figure 1</xref>; <xref ref-type="sec" rid="s12">Supplementary Figure S1&#x2013;S3F</xref>) but not for skewed traits (<xref ref-type="fig" rid="F1">Figure 1</xref>; <xref ref-type="sec" rid="s12">Supplementary Figure S1&#x2013;S3K</xref>). Transforming trait values into ranks like INT and KW can address the outlier problem in kurtotic distributions. However, these two rank-based methods, especially DGLM_INT, cannot fully solve the heteroscedastic problem under skewed traits (<xref ref-type="fig" rid="F1">Figure 1</xref>; <xref ref-type="sec" rid="s12">Supplementary Figure S1&#x2013;S3K</xref>).</p>
<p>BF has inflated FPR under kurtotic traits when MAF &#x3c;0.2 (<xref ref-type="sec" rid="s12">Supplementary Figures S5, S8A&#x2013;C</xref>). The BF test is equivalent to the ANOVA <italic>F</italic> test for regressing <inline-formula id="inf100">
<mml:math id="m105">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> on the two indicator variables differentiating the three genotype groups of an SNP. Therefore, BF is not a valid test given kurtotic traits and unbalanced sample sizes across three genotype groups (i.e., only a few observations in a particular genotype group, which is a common situation under small MAF). This unbalanced issue can be alleviated given a larger total sample size (<italic>N</italic>). As we can see, the problem of inflated FPR in BF is not so severe when <italic>N</italic> increases to 147,836 (<xref ref-type="sec" rid="s12">Supplementary Figures S11, S14A&#x2013;C</xref>).</p>
<p>DRM has a power performance similar to that of BF. These two tests are based on a similar strategy. DRM regresses <inline-formula id="inf101">
<mml:math id="m106">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> (the deviation between individual <italic>j</italic>&#x2019;s covariate-adjusted residual and his/her group median) on the minor allele count at each SNP (i.e., 0, 1, or 2), whereas BF regresses <inline-formula id="inf102">
<mml:math id="m107">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> on the two indicator variables of genotypes. The only difference between the two methods is that DRM regards genotypes as a continuous scale (coding as 0, 1, and 2), whereas BF treats genotypes as a categorical scale. It is unsurprising that DRM and BF perform similarly in power. However, DRM controls FPR more appropriately than BF, because unbalanced sample size is a more critical issue when separating genotypes into three groups.</p>
<p>Two non-parametric tests, QUAIL and KW, were evaluated in this work. Although QUAIL maintains an appropriate FPR under normally or non-normally distributed traits (column 1 of <xref ref-type="fig" rid="F1">Figure 1</xref>; <xref ref-type="sec" rid="s12">Supplementary Figure S1&#x2013;S3</xref>), its statistical power is never among the optimal choices under any situation (columns 2&#x2013;5 of <xref ref-type="fig" rid="F1">Figure 1</xref>; <xref ref-type="sec" rid="s12">Supplementary Figure S1&#x2013;S3</xref>). Besides, the power of QUAIL can be compromised by quantile crossing, which is a well-known issue when estimating multiple quantiles simultaneously (<xref ref-type="bibr" rid="B2">Bondell et al., 2010</xref>). However, the QUAIL R code does not evaluate whether an analysis suffers from quantile crossing. Moreover, QUAIL requires much longer computation time than the other competitors. Therefore, I would not recommend using QUAIL to perform a genome-wide vQTL search. On the other hand, KW is the most powerful vQTL test for kurtotic phenotypes (<xref ref-type="fig" rid="F1">Figure 1</xref>; <xref ref-type="sec" rid="s12">Supplementary Figure S1&#x2013;S3G&#x2013;J</xref>). When the traits are skewed, KW presents a slight inflation in FPR (<xref ref-type="fig" rid="F1">Figure 1</xref>; <xref ref-type="sec" rid="s12">Supplementary Figure S1&#x2013;S3K</xref>). However, the subsequent direct GxE or GxG analysis can further assess the significance of GxE and GxG. Therefore, after considering the computation time, I chose KW to analyze the four lipid traits in the TWB data.</p>
<p>With the comprehensive simulations in this work, DRM and KW are the most recommended parametric and non-parametric vQTL tests, respectively. Unlike DRM (<xref ref-type="bibr" rid="B18">Marderstein et al., 2021</xref>), KW has no specific paper to introduce its implementation in detecting vQTLs. Therefore, as a demonstration, I here used KW to analyze the four TWB lipid traits. Other investigators may also choose DRM to perform genome-wide vQTL searches because of its adequate control of FPR and shorter computational time (<xref ref-type="table" rid="T1">Table 1</xref>). If one applies two or more methods to the same data set, he/she may explore the superiority of each method by comparing the replication rates of different techniques. That is, one can first separate the data set into a discovery set and a replication set, then calculate the replication rates of different methods.</p>
<p>Same with previous vQTL works (<xref ref-type="bibr" rid="B15">Lin, 2024</xref>; <xref ref-type="bibr" rid="B32">Wang et al., 2019</xref>), I only analyzed common SNPs with MAFs <inline-formula id="inf103">
<mml:math id="m108">
<mml:mrow>
<mml:mo>&#x2265;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 0.05. The reason is that GxE is the joint distribution between a SNP and E, and GxG is the joint distribution between two SNPs. Interaction studies are difficult to replicate if an SNP has a small MAF (<xref ref-type="bibr" rid="B15">Lin, 2024</xref>). <xref ref-type="fig" rid="F2">Figures 2</xref>, <xref ref-type="fig" rid="F3">3</xref> show that GxE and GxG are enriched among vQTLs of the four lipid traits. With more vQTLs identified from TG, it is not surprising that more GxE and GxG can be explored from this trait (GxE: comparing <xref ref-type="fig" rid="F2">Figure 2</xref> with <xref ref-type="sec" rid="s12">Supplementary Figures S17&#x2013;S19</xref>; GxG: comparing <xref ref-type="fig" rid="F3">Figure 3</xref> with <xref ref-type="sec" rid="s12">Supplementary Figures S20&#x2013;S22</xref>).</p>
<p>For TG, <italic>APOA5 rs662799</italic> has been found to interact with cigarette smoking and alcohol consumption, based on individuals from Korea (<xref ref-type="bibr" rid="B23">Park and Kang, 2020</xref>). These interactions were also associated with metabolic syndrome (MetS) by analyzing subjects from Jilin Province of China (<xref ref-type="bibr" rid="B35">Wu et al., 2016</xref>). Because TG is one critical component of MetS, both studies (<xref ref-type="bibr" rid="B23">Park and Kang, 2020</xref>; <xref ref-type="bibr" rid="B35">Wu et al., 2016</xref>) were in line with our analysis results (<italic>APOA5 rs662799</italic> is on the second row of <xref ref-type="fig" rid="F2">Figure 2</xref>).</p>
<p>DRM and KW are the most recommended parametric and non-parametric vQTL methods. QUAIL appropriately preserves the FPR under normally or non-normally distributed traits. However, its power is never among the optimal choices, and its computational time is much longer than the other competitors. BF suffers from severe inflation in FPR when SNP&#x2019;s MAF &#x3c;0.2. We may alleviate this inflation in FPR by increasing the sample size to, say, <italic>N</italic> &#x223c; 150,000. DRM and BF are based on a similar strategy. DRM regresses the deviation between individuals&#x2019; covariate-adjusted residuals and their group medians on the minor allele count of an SNP, whereas BF regresses the deviation between individuals&#x2019; covariate-adjusted residuals and their group medians on two indicator variables categorizing three genotypes. Therefore, DRM and BF have similar power. However, BF suffers from more inflation in FPR than DRM does, mainly when a small sample size is observed in one genotype. Although DGLM is the most powerful method given normally distributed traits, it is not valid (i.e., producing large FPR) for non-normally distributed traits. Adopting the rank-based INT transformation can address the outlier problem and adjust the FPR for kurtotic traits. However, INT cannot fully solve the heteroscedastic issue of skewed traits.</p>
<p>Recently, a robust Bayesian mixed model rooted in the one-way ANOVA has been developed (<xref ref-type="bibr" rid="B6">Fan et al., 2025</xref>). By specifying a likelihood based on a heavy-tailed distribution, this method can improve robustness in GxE detection. With Fan et al.&#x2018;s concept (<xref ref-type="bibr" rid="B6">Fan et al., 2025</xref>), the non-robust parametric tests can build a robust mixed-effect model.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>The datasets presented in this study can be found in online repositories. The names of the repository/repositories and accession number(s) can be found below: <ext-link ext-link-type="uri" xlink:href="https://www.twbiobank.org.tw/new_web/">https://www.twbiobank.org.tw/new_web/</ext-link>, TWBR10810-07.</p>
</sec>
<sec sec-type="ethics-statement" id="s6">
<title>Ethics statement</title>
<p>The studies involving humans were approved by National Taiwan University Hospital. The studies were conducted in accordance with the local legislation and institutional requirements. The participants provided their written informed consent to participate in this study. No potentially identifiable images or data are presented in this study.</p>
</sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>W-YL: Formal Analysis, Methodology, Validation, Project administration, Data curation, Supervision, Writing &#x2013; original draft, Conceptualization, Software, Funding acquisition, Investigation, Visualization, Resources, Writing &#x2013; review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This study was supported by the National Science and Technology Council of Taiwan (grant number 112-2628-B-002-024-MY3 to W.-Y.L.).</p>
</sec>
<ack>
<p>The authors thank the reviewers for their constructive comments and all the Taiwan Biobank investigators, staff, and participants for their contributions. This research has been conducted using the Taiwan Biobank Resource under application number TWBR10810-07.</p>
</ack>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s10">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="s12">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fgene.2025.1617504/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fgene.2025.1617504/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.pdf" id="SM1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Beyene</surname>
<given-names>H. B.</given-names>
</name>
<name>
<surname>Olshansky</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Aa</surname>
<given-names>T. S.</given-names>
</name>
<name>
<surname>Giles</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Huynh</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Cinel</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>High-coverage plasma lipidomics reveals novel sex-specific lipidomic fingerprints of age and BMI: evidence from two large population cohort studies</article-title>. <source>PLoS Biol. Sep.</source> <volume>18</volume>, <fpage>e3000870</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pbio.3000870</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bondell</surname>
<given-names>H. D.</given-names>
</name>
<name>
<surname>Reich</surname>
<given-names>B. J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Noncrossing quantile regression curve estimation</article-title>. <source>Biometrika</source> <volume>97</volume>, <fpage>825</fpage>&#x2013;<lpage>838</lpage>. <pub-id pub-id-type="doi">10.1093/biomet/asq048</pub-id>
<pub-id pub-id-type="pmid">22822254</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Breusch</surname>
<given-names>T. S.</given-names>
</name>
<name>
<surname>Pagan</surname>
<given-names>A. R.</given-names>
</name>
</person-group> (<year>1979</year>). <article-title>A simple test for heteroscedasticity and random coefficient variation</article-title>. <source>Econometrica</source> <volume>47</volume>, <fpage>1287</fpage>&#x2013;<lpage>1294</lpage>. <pub-id pub-id-type="doi">10.2307/1911963</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brown</surname>
<given-names>M. B.</given-names>
</name>
<name>
<surname>Forsythe</surname>
<given-names>A. B.</given-names>
</name>
</person-group> (<year>1974</year>). <article-title>The small sample behavior of some statistics which test the equality of several means</article-title>. <source>Technometrics</source> <volume>16</volume>, <fpage>129</fpage>&#x2013;<lpage>132</lpage>. <pub-id pub-id-type="doi">10.1080/00401706.1974.10489158</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Delaneau</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Zagury</surname>
<given-names>J. F.</given-names>
</name>
<name>
<surname>Marchini</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Improved whole-chromosome phasing for disease and population genetic studies</article-title>. <source>Nat. Methods</source> <volume>10</volume>, <fpage>5</fpage>&#x2013;<lpage>6</lpage>. <pub-id pub-id-type="doi">10.1038/nmeth.2307</pub-id>
<pub-id pub-id-type="pmid">23269371</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fan</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>S. G.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>W. Q.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Robust sparse Bayesian regression for longitudinal gene-environment interactions</article-title>. <source>J. Roy. Stat. Soc. C</source>, <fpage>qlaf027</fpage>. <pub-id pub-id-type="doi">10.1093/jrsssc/qlaf027</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Fox</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Weisberg</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). <source>An R companion to applied regression</source>. <publisher-loc>Thousand Oaks, United States</publisher-loc>: <publisher-name>Sage Publications</publisher-name>. <edition>Third Edition</edition>.</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Howie</surname>
<given-names>B. N.</given-names>
</name>
<name>
<surname>Donnelly</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Marchini</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>A flexible and accurate genotype imputation method for the next generation of genome-wide association studies</article-title>. <source>Plos Genet. Jun</source> <volume>5</volume>, <fpage>e1000529</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pgen.1000529</pub-id>
<pub-id pub-id-type="pmid">19543373</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kosugi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kamatani</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Harada</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Tomizuka</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Momozawa</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Morisaki</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Detection of trait-associated structural variations using short-read sequencing</article-title>. <source>Cell. Genom</source> <volume>14</volume> (<issue>3</issue>), <fpage>100328</fpage>. <pub-id pub-id-type="doi">10.1016/j.xgen.2023.100328</pub-id>
<pub-id pub-id-type="pmid">37388916</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kozlowski</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Gawelczyk</surname>
<given-names>A. T.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>Why are species&#x27; body size distributions usually skewed to the right?</article-title>. <source>Funct. Ecol.</source> <volume>16</volume>., <fpage>419</fpage>&#x2013;<lpage>432</lpage>. <pub-id pub-id-type="doi">10.1046/j.1365-2435.2002.00646.x</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kruskal</surname>
<given-names>W. H.</given-names>
</name>
<name>
<surname>Wallis</surname>
<given-names>W. A.</given-names>
</name>
</person-group> (<year>1952</year>). <article-title>Use of ranks in one-criterion variance analysis</article-title>. <source>J. Am. Stat. Assoc.</source> <volume>47</volume>, <fpage>583</fpage>&#x2013;<lpage>621</lpage>. <pub-id pub-id-type="doi">10.2307/2280779</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lara</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Amigo</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Association between education and blood lipid levels as income increases over a decade: a cohort study</article-title>. <source>BMC Public Health</source> <volume>18</volume>, <fpage>256</fpage>. <pub-id pub-id-type="doi">10.1186/s12889-018-5185-3</pub-id>
<pub-id pub-id-type="pmid">29482545</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Levene</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>1960</year>). <article-title>Robust tests for equality of variances</article-title>. In: <source>Contributions to probability and statistics; essays in honor of harold hotelling</source>. <publisher-loc>Redwood City, United States</publisher-loc>: <publisher-name>Stanford University Press</publisher-name>. p. <fpage>278</fpage>&#x2013;<lpage>292</lpage>.</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lin</surname>
<given-names>W. Y.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A large-scale observational study linking various kinds of physical exercise to lipoprotein-lipid profile</article-title>. <source>J. Int. Soc. Sport Nutr.</source> <volume>18</volume>, <fpage>35</fpage>. <pub-id pub-id-type="doi">10.1186/s12970-021-00436-2</pub-id>
<pub-id pub-id-type="pmid">33971893</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lin</surname>
<given-names>W. Y.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Detecting gene-environment interactions from multiple continuous traits</article-title>. <source>Bioinformatics</source> <volume>40</volume>, <fpage>btae419</fpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btae419</pub-id>
<pub-id pub-id-type="pmid">38917408</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ma</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Niu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ni</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Cigarette smoke exposure impairs lipid metabolism by decreasing low-density lipoprotein receptor expression in hepatocytes</article-title>. <source>Lipids Health Dis.</source> <volume>19</volume>, <fpage>88</fpage>. <pub-id pub-id-type="doi">10.1186/s12944-020-01276-w</pub-id>
<pub-id pub-id-type="pmid">32384892</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Manichaikul</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Mychaleckyj</surname>
<given-names>J. C.</given-names>
</name>
<name>
<surname>Rich</surname>
<given-names>S. S.</given-names>
</name>
<name>
<surname>Daly</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Sale</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>W. M.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Robust relationship inference in genome-wide association studies</article-title>. <source>Bioinformatics</source> <volume>26</volume>, <fpage>2867</fpage>&#x2013;<lpage>2873</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btq559</pub-id>
<pub-id pub-id-type="pmid">20926424</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Marderstein</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>Davenport</surname>
<given-names>E. R.</given-names>
</name>
<name>
<surname>Kulm</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Van Hout</surname>
<given-names>C. V.</given-names>
</name>
<name>
<surname>Elemento</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Clark</surname>
<given-names>A. G.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Leveraging phenotypic variability to identify genetic interactions in human phenotypes</article-title>. <source>Am. J. Hum. Genet.</source> <volume>108</volume>, <fpage>49</fpage>&#x2013;<lpage>67</lpage>. <pub-id pub-id-type="doi">10.1016/j.ajhg.2020.11.016</pub-id>
<pub-id pub-id-type="pmid">33326753</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Marees</surname>
<given-names>A. T.</given-names>
</name>
<name>
<surname>de Kluiver</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Stringer</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Vorspan</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Curis</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Marie-Claire</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>A tutorial on conducting genome-wide association studies: quality control and statistical analysis</article-title>. <source>Int. J. Methods Psychiatr. Res.</source> <volume>27</volume>, <fpage>e1608</fpage>. <pub-id pub-id-type="doi">10.1002/mpr.1608</pub-id>
<pub-id pub-id-type="pmid">29484742</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>McDonald</surname>
<given-names>J. H.</given-names>
</name>
</person-group> (<year>2014</year>). <source>Handbook of biological statistics</source>. <publisher-loc>Baltimore, Maryland</publisher-loc>: <publisher-name>Sparky House Publishing</publisher-name>. <edition>3rd Edition</edition>.</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Miao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Schmitz</surname>
<given-names>L. L.</given-names>
</name>
<name>
<surname>Fletcher</surname>
<given-names>J. M.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>A quantile integral linear model to quantify genetic effects on phenotypic variability</article-title>. <source>Proc. Natl. Acad. Sci. U. S. A.</source> <volume>119</volume>, <fpage>e2212959119</fpage>. <pub-id pub-id-type="doi">10.1073/pnas.2212959119</pub-id>
<pub-id pub-id-type="pmid">36122202</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pare</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Cook</surname>
<given-names>N. R.</given-names>
</name>
<name>
<surname>Ridker</surname>
<given-names>P. M.</given-names>
</name>
<name>
<surname>Chasman</surname>
<given-names>D. I.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>On the use of variance per genotype as a tool to identify quantitative trait interaction effects: a report from the women&#x27;s genome health Study</article-title>. <source>Plos Genet. Jun</source> <volume>6</volume>, <fpage>e1000981</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pgen.1000981</pub-id>
<pub-id pub-id-type="pmid">20585554</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Park</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Alcohol, carbohydrate, and calcium intakes and smoking interactions with APOA5 rs662799 and rs2266788 were associated with elevated plasma triglyceride concentrations in a cross-sectional study of Korean adults</article-title>. <source>J. Acad. Nutr. Diet.</source> <volume>120</volume>, <fpage>1318</fpage>&#x2013;<lpage>1329</lpage>. <pub-id pub-id-type="doi">10.1016/j.jand.2020.01.009</pub-id>
<pub-id pub-id-type="pmid">32335043</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Purcell</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Neale</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Todd-Brown</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Thomas</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Ferreira</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Bender</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2007</year>). <article-title>PLINK: a tool set for whole-genome association and population-based linkage analyses</article-title>. <source>Am. J. Hum. Genet.</source> <volume>81</volume>, <fpage>559</fpage>&#x2013;<lpage>575</lpage>. <pub-id pub-id-type="doi">10.1086/519795</pub-id>
<pub-id pub-id-type="pmid">17701901</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ronnegard</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Valdar</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Detecting major genetic loci controlling phenotypic variability in experimental crosses</article-title>. <source>Genet. Jun</source> <volume>188</volume>, <fpage>435</fpage>&#x2013;<lpage>447</lpage>. <pub-id pub-id-type="doi">10.1534/genetics.111.127068</pub-id>
<pub-id pub-id-type="pmid">21467569</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ronnegard</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Valdar</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Recent developments in statistical methods for detecting genetic loci affecting phenotypic variability</article-title>. <source>Bmc Genet.</source> <volume>13</volume>, <fpage>63</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2156-13-63</pub-id>
<pub-id pub-id-type="pmid">22827487</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shi</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Genome-wide variance quantitative trait locus analysis suggests small interaction effects in blood pressure traits</article-title>. <source>Sci. Rep.</source> <volume>12</volume>, <fpage>12649</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-022-16908-7</pub-id>
<pub-id pub-id-type="pmid">35879408</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Smyth</surname>
<given-names>G. K.</given-names>
</name>
</person-group> (<year>1989</year>). <article-title>Generalized linear-models with varying dispersion</article-title>. <source>J. Roy. Stat. Soc. B</source> <volume>51</volume>, <fpage>47</fpage>&#x2013;<lpage>60</lpage>. <pub-id pub-id-type="doi">10.1111/j.2517-6161.1989.tb01747.x</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Steiner</surname>
<given-names>J. L.</given-names>
</name>
<name>
<surname>Lang</surname>
<given-names>C. H.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Alcohol, adipose tissue and lipid dysregulation</article-title>. <source>Biomol. Mar.</source> <volume>7</volume>, <fpage>16</fpage>. <pub-id pub-id-type="doi">10.3390/biom7010016</pub-id>
<pub-id pub-id-type="pmid">28212318</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tharu</surname>
<given-names>B. P.</given-names>
</name>
<name>
<surname>Tsokos</surname>
<given-names>C. P.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>A statistical study of serum cholesterol level by gender and race</article-title>. <source>J. Res. Health Sci. Jul</source> <volume>25</volume> (<issue>17</issue>), <fpage>e00386</fpage>.<pub-id pub-id-type="pmid">28878106</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Uffelmann</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Q. Q.</given-names>
</name>
<name>
<surname>Munung</surname>
<given-names>N. S.</given-names>
</name>
<name>
<surname>de Vries</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Okada</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Martin</surname>
<given-names>A. R.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Genome-wide association studies</article-title>. <source>Nat. Rev. Method Prime</source> <volume>1</volume>, <fpage>59</fpage>&#x2013;<lpage>21</lpage>. <pub-id pub-id-type="doi">10.1038/s43586-021-00056-9</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>H. W.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>F. T.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Kemper</surname>
<given-names>K. E.</given-names>
</name>
<name>
<surname>Xue</surname>
<given-names>A. L.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Genotype-by-environment interactions inferred from genetic effects on phenotypic variability in the UK Biobank</article-title>. <source>Sci. Adv.</source> <volume>5</volume>, <fpage>eaaw3538</fpage>. <pub-id pub-id-type="doi">10.1126/sciadv.aaw3538</pub-id>
<pub-id pub-id-type="pmid">31453325</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wei</surname>
<given-names>C. Y.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>J. H.</given-names>
</name>
<name>
<surname>Yeh</surname>
<given-names>E. C.</given-names>
</name>
<name>
<surname>Tsai</surname>
<given-names>M. F.</given-names>
</name>
<name>
<surname>Kao</surname>
<given-names>H. J.</given-names>
</name>
<name>
<surname>Lo</surname>
<given-names>C. Z.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Genetic profiles of 103,106 individuals in the Taiwan biobank provide insights into the health and history of Han Chinese</article-title>. <source>Npj Genom Med.</source> <volume>6</volume>, <fpage>10</fpage>. <pub-id pub-id-type="doi">10.1038/s41525-021-00178-9</pub-id>
<pub-id pub-id-type="pmid">33574314</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Westerman</surname>
<given-names>K. E.</given-names>
</name>
<name>
<surname>Majarian</surname>
<given-names>T. D.</given-names>
</name>
<name>
<surname>Giulianini</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Jang</surname>
<given-names>D. K.</given-names>
</name>
<name>
<surname>Miao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Florez</surname>
<given-names>J. C.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Variance-quantitative trait loci enable systematic discovery of gene-environment interactions for cardiometabolic serum biomarkers</article-title>. <source>Nat. Commun.</source> <volume>13</volume>, <fpage>3993</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-022-31625-5</pub-id>
<pub-id pub-id-type="pmid">35810165</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>Y. H.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>Y. Q.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>T. C.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S. B.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>Y. L.</given-names>
</name>
<name>
<surname>Qi</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Interactions of environmental factors and APOA1-APOC3-APOA4-APOA5 gene cluster gene polymorphisms with metabolic syndrome</article-title>. <source>Plos One</source> <volume>11</volume>, <fpage>e0147946</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0147946</pub-id>
<pub-id pub-id-type="pmid">26824674</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Young</surname>
<given-names>A. I.</given-names>
</name>
<name>
<surname>Wauthier</surname>
<given-names>F. L.</given-names>
</name>
<name>
<surname>Donnelly</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Identifying loci affecting trait variability and detecting interactions in genome-wide association studies</article-title>. <source>Nat. Genet. Nov.</source> <volume>50</volume>, <fpage>1608</fpage>&#x2013;<lpage>1614</lpage>. <pub-id pub-id-type="doi">10.1038/s41588-018-0225-6</pub-id>
<pub-id pub-id-type="pmid">30323177</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Bell</surname>
<given-names>J. T.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Detecting genetic effects on phenotype variability to capture gene-by-environment interactions: a systematic method comparison</article-title>. <source>G3 (Bethesda)</source> <volume>14</volume>, <fpage>jkae022</fpage>. <pub-id pub-id-type="doi">10.1093/g3journal/jkae022</pub-id>
<pub-id pub-id-type="pmid">38289865</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Gene-Environment interaction: a variable selection perspective</article-title>. <source>Methods Mol. Biol.</source> <volume>2212</volume>, <fpage>191</fpage>&#x2013;<lpage>223</lpage>. <pub-id pub-id-type="doi">10.1007/978-1-0716-0947-7_13</pub-id>
<pub-id pub-id-type="pmid">33733358</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>